{
  "data": {
    "category": {
      "area": "models",
      "capabilities": [
        "inference.decision"
      ],
      "description": "Models that answer typed questions with probabilities instead of text, a yes or no, one label from a set or a level on a rubric, for routing, triage and checks inside agents and workflows. Compared on accuracy, calibration, context, latency, price and whether the weights are open.",
      "json": "https://www.anchorterminal.com/categories/decision-models.json",
      "name": "Decision models",
      "slug": "decision-models",
      "test": "A fixed set of labelled decisions (yes or no, one label from a set, a level on a rubric) sent to every model in the same request shape, with the same states and questions. We check accuracy against the labels, calibration (expected calibration error, Brier score and how often an answer given 0.9 or more is wrong), whether answers move when the option order changes, latency and the cost per 1,000 decisions.",
      "title": "Decision models for AI agents",
      "toolCount": 12,
      "tools": [
        "openai-decisions-api",
        "decider",
        "convai-laya",
        "jaredpalmer-kev",
        "vela",
        "cloudflare-clef",
        "pplx-decider",
        "typesafe-jev",
        "strands-decider",
        "gliclass",
        "celeris-1-decision",
        "liquid-d1"
      ],
      "url": "https://www.anchorterminal.com/categories/decision-models"
    },
    "faq": [
      {
        "answer": "OpenAI Decisions API has the highest benchmark score of the 12 ranked decision models, 71.5 (BB). Decider is second with 69.5 (B).",
        "question": "What are the highest-rated decision models for AI agents?"
      },
      {
        "answer": "1 of the 12 ranked here grade BB or better, the bar for agent-ready on the Anchor benchmark.",
        "question": "How many decision models are agent-ready?"
      },
      {
        "answer": "None of the ranked listings here accepts x402 for its main call yet.",
        "question": "Which decision models accept x402 payments?"
      },
      {
        "answer": "By the Anchor benchmark score out of 100, a weighted mean of the scored categories minus deductions for negative events, from public evidence re-checked as vendors change. Listings cannot pay for a place. The latest assessment behind this page is from 9 October 2026.",
        "question": "How is this list ranked?"
      }
    ],
    "howToChoose": [
      {
        "label": "Calibration of stated probabilities",
        "detail": "Check calibration on labelled cases, since a high stated probability that often turns out wrong sends an agent down the wrong branch."
      },
      {
        "label": "Stable answers to option order",
        "detail": "Check whether answers stay the same when the order of the options changes, since a routing label that depends on position sends work to the wrong queue."
      },
      {
        "label": "Accuracy on your own labels",
        "detail": "Measure accuracy on labelled decisions drawn from your own workflow, since the answer labels an agent acts on are the ones that count."
      },
      {
        "label": "Cost per 1,000 decisions",
        "detail": "Compare the cost per 1,000 decisions for your own label mix, since output length and retries add to the list price."
      }
    ],
    "picks": [
      {
        "also": {
          "name": "Decider",
          "slug": "decider",
          "why": "B, 69.5/100"
        },
        "name": "OpenAI Decisions API",
        "need": "Highest score overall",
        "slug": "openai-decisions-api",
        "why": "BB, 71.5/100 on the benchmark"
      },
      {
        "name": "Decider",
        "need": "Reliability",
        "slug": "decider",
        "why": "90/100 on reliability, against 53 for the overall leader"
      },
      {
        "name": "pplx-decider",
        "need": "Schema \u0026 documentation",
        "slug": "pplx-decider",
        "why": "90/100 on schema \u0026 documentation, against 89 for the overall leader"
      },
      {
        "name": "Decider",
        "need": "Maintenance \u0026 community",
        "slug": "decider",
        "why": "87/100 on maintenance \u0026 community, against 77 for the overall leader"
      },
      {
        "name": "Clef",
        "need": "Transparency \u0026 trust",
        "slug": "cloudflare-clef",
        "why": "86/100 on transparency \u0026 trust, against 83 for the overall leader"
      },
      {
        "also": {
          "name": "Laya",
          "slug": "convai-laya",
          "why": "self-hosted, Apache-2 licence"
        },
        "name": "Decider",
        "need": "Self-hosting under an open licence",
        "slug": "decider",
        "why": "self-hosted, Apache-2 licence"
      }
    ],
    "ranked": 12,
    "shortlist": [
      {
        "bestFor": "High-volume yes or no checks, routing among fixed options and rubric scoring over text and images, for teams already on an OpenAI key.",
        "grade": "BB",
        "name": "OpenAI Decisions API",
        "position": 1,
        "price": "Pay per use",
        "score": 71.5,
        "slug": "openai-decisions-api",
        "strengths": [
          "Three typed question kinds, predicate, choice and score, with per-option probabilities and up to 200 questions about one input in a call",
          "$0.10 per million input tokens on `gpt-6-luna`, with no output, cache-read or cache-write charge",
          "Text and up to 128 inline images in one request"
        ],
        "url": "https://www.anchorterminal.com/tools/openai-decisions-api",
        "verdict": "Typed predicate, choice and score answers with probabilities, up to 200 questions a call, at $0.10 per million input tokens with no output charge. It is a public beta released on 6 October 2026, with one model alias, no dated snapshot, no Batch route and no SLA found.",
        "weaknesses": [
          "Public beta released on 6 October 2026. The guide expects general availability in the coming weeks and gives no date",
          "One model alias, `gpt-6-luna`, with no dated snapshot to pin",
          "Images must be inline base64 data URLs. Hosted image URLs and `file_id` inputs aren't accepted"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Local classification, routing, triage and checks where a team wants open weights in several sizes and a Jev-shaped route.",
        "grade": "B",
        "name": "Decider",
        "position": 2,
        "price": "Free · OSS",
        "score": 69.5,
        "slug": "decider",
        "strengths": [
          "Apache-2.0 code and weights, with the training code, data builders and per-version measurements in the repository",
          "Sizes from 0.8B to 35B parameters, with GGUF files for CPU and builds for CUDA, Apple silicon and vLLM",
          "`POST /v1/systemone` follows TypeSafe's wire format, and the README says TypeSafe's SDKs work with `TYPESAFE_BASE_URL` set to the local server"
        ],
        "url": "https://www.anchorterminal.com/tools/decider",
        "verdict": "An Apache-2.0 decision model family with a dated changelog, passing CI, 21 package releases since 22 September 2026 and model cards that list measured regressions. One person maintains it, the local server has no authentication option, states over 32,768 tokens are cut without an error, and no security policy is published.",
        "weaknesses": [
          "The local server has no authentication option. It binds to 127.0.0.1 since 1.7.1",
          "States over 32,768 tokens are truncated (`DECIDER_MAX_STATE_TOKENS`) without an error",
          "No SECURITY.md, disclosure policy or security contact was found in the repository"
        ],
        "where": "local",
        "x402": "no"
      },
      {
        "bestFor": "Fast, cheap classification and routing on short text in many languages, as a base to fine-tune on your own labels, and as an MCP or LangGraph routing step.",
        "grade": "B",
        "name": "Laya",
        "position": 3,
        "price": "Free · OSS",
        "score": 69.2,
        "slug": "convai-laya",
        "strengths": [
          "Apache-2.0 code and weights, installed with `pip install laya`, with Python 3.10 to 3.13 tested in CI",
          "421M and 322M-parameter encoders that the README times at 32.8 to 39.5 ms for one question on a Tesla T4",
          "A Jev-compatible HTTP server with a batch route for up to 64 states, an 8-tool MCP server, and LangChain, LlamaIndex and CrewAI wrappers"
        ],
        "url": "https://www.anchorterminal.com/tools/convai-laya",
        "verdict": "Apache-2.0 code and weights, installed with `pip install laya`, with Python 3.10 to 3.13 tested in CI. Base checkpoints score 0.362 and 0.352 on the maintainers' typed-decisions benchmark against a 0.318 random baseline, so it needs fine-tuning.",
        "weaknesses": [
          "Base checkpoints score 0.362 and 0.352 on the maintainers' typed-decisions benchmark against a 0.318 random baseline, so it needs fine-tuning",
          "Choice options share a 192 or 256-token budget, and the README reports 0.425 on Banking77's 77 labels",
          "512 tokens of context on the English checkpoint and 1,024 by default on the others"
        ],
        "where": "local",
        "x402": "no"
      },
      {
        "bestFor": "Self-hosted classification, routing, triage and rubric scoring where a probability matters, especially for teams already calling Jev who want the same API on their own hardware.",
        "grade": "B",
        "name": "Kev",
        "position": 4,
        "price": "Free · OSS",
        "score": 67.4,
        "slug": "jaredpalmer-kev",
        "strengths": [
          "Apache-2.0 code, adapters and heads on Apache-2.0 Qwen bases, with release tarballs and SHA-256 checksums for the 0.8B, 4B and 9B models",
          "The same `/v1/systemone` request and answer shapes as Jev, and the README says TypeSafe's Python SDK works against it unchanged",
          "A fitted temperature per checkpoint, with Brier scores, calibration error and confident-error rates published for each model"
        ],
        "url": "https://www.anchorterminal.com/tools/jaredpalmer-kev",
        "verdict": "Apache-2.0 code, adapters and heads on Apache-2.0 Qwen bases, with release tarballs and SHA-256 checksums for the 0.8B, 4B and 9B models. No package. `pip install kev` installs an unrelated 2021 ORM, so Kev runs from a Git clone with uv.",
        "weaknesses": [
          "No package. `pip install kev` installs an unrelated 2021 ORM, so Kev runs from a Git clone with uv",
          "Kev-0.8B, 4B and 9B are validated to 8,192 tokens of state, though the server accepts 65,536",
          "Jared Palmer wrote 312 of the 333 commits we cloned"
        ],
        "where": "local",
        "x402": "no"
      },
      {
        "bestFor": "Self-hosted routing and guardrail checks in one call, where span offsets for personal data or unsupported claims matter.",
        "grade": "B",
        "name": "Vela 2.0",
        "position": 5,
        "price": "Free · OSS",
        "score": 66.5,
        "slug": "vela",
        "strengths": [
          "Five question types in one request (choice, noul, score, set and span), with span answers as labelled character offsets and a probability each",
          "Apache-2.0 weights, code and documentation, ungated on Hugging Face, with safetensors files and a SHA256SUMS manifest in the three decoder repositories",
          "Two serving routes. A bundled FastAPI server on `POST /v1/systemone`, and the router's model runtime with an OpenAPI 3.0.3 contract and Prometheus metrics"
        ],
        "url": "https://www.anchorterminal.com/tools/vela",
        "verdict": "One self-hosted call answers routing, prompt-attack, personal-data and unsupported-claim questions with probabilities and character offsets, under Apache-2.0 with SHA-256 manifests. The models are days old and carry no Hub version tags, and the three larger sizes keep 74 to 89 per cent of their Decision 2.0 bases on the Jev Decision Index by the authors' figures.",
        "weaknesses": [
          "No version tags on the four Hub repositories, and the 4B and 9B weights were replaced in place on 3 October 2026",
          "Loading with `transformers` needs `trust_remote_code=True`, which runs Python from the model repository",
          "The router's `vllm-sr serve MODEL` engine mode is newer than the 0.4.0 stable release and needs the development channel"
        ],
        "where": "local",
        "x402": "no"
      },
      {
        "bestFor": "Classification, routing and rubric scoring over text, JSON and images inside Cloudflare, or self-hosted where data can't leave.",
        "grade": "B",
        "name": "Clef",
        "position": 6,
        "price": "Freemium",
        "score": 66.1,
        "slug": "cloudflare-clef",
        "strengths": [
          "Apache-2.0 weights for both models on Hugging Face, ungated, with no account needed to download",
          "$0.24 per million input tokens for Clef and $0.09 for Clef-flash, with 10,000 free neurons a day on Workers AI",
          "Typed answers with a probability for every allowed option, 1 to 64 questions a call, and up to 4 images"
        ],
        "url": "https://www.anchorterminal.com/tools/cloudflare-clef",
        "verdict": "Apache-2.0 weights for both models on Hugging Face, ungated, with no account needed to download. Released on 1 October 2026, with no service record and no entry in the Workers AI changelog we read.",
        "weaknesses": [
          "Released on 1 October 2026, with no service record and no entry in the Workers AI changelog we read",
          "No limitations, failure modes or prompt-injection guidance in the model page or the launch post",
          "No SLA found for Workers AI, and Clef shares the Text Generation default of 300 requests a minute"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "High-volume classification, routing and rubric scoring over text and images where a team wants a low hosted price and the option to run the same weights itself.",
        "grade": "B",
        "name": "pplx-decider",
        "position": 7,
        "price": "Pay per use",
        "score": 62.9,
        "slug": "pplx-decider",
        "strengths": [
          "Weights for v1.1 and v1 are public on Hugging Face under Apache-2.0, with training code, config and a data manifest in the repository",
          "One request takes up to 128 questions about one `state`, mixing `noul`, `choice` and `score`, with under 262,144 input tokens",
          "$0.02 per million input tokens with free output and no per-request fee, published without a login"
        ],
        "url": "https://www.anchorterminal.com/tools/pplx-decider",
        "verdict": "The hosted Decisions API has an OpenAPI description, typed questions, documented limits and a published price of $0.02 per million input tokens, and the v1.1 weights are public under Apache-2.0. The API is days old, the official Python SDK has no method for it, and the governing terms and privacy notice could not be read on 9 October 2026.",
        "weaknesses": [
          "The Decisions API was released in October 2026, so its incident record is days long, and Perplexity's FAQ says it gives no uptime guarantee",
          "The official Python SDK at 0.43.8 has no Decisions method. The guide uses a plain HTTP client",
          "An image over 2,048 tiles is not rejected. The request waits about a minute and returns 504"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "High-volume yes or no answers, labelling, routing and rubric scoring where a probability is more useful than prose, such as ticket triage, invoice checks or picking a tool or skill from a list.",
        "grade": "B",
        "name": "Jev",
        "position": 8,
        "price": "Pay per use",
        "score": 62.1,
        "slug": "typesafe-jev",
        "strengths": [
          "Typed answers with probabilities for `noul`, `choice` and `score` questions, many per call, with no text to parse",
          "$0.042 per million input tokens, with output tokens free",
          "A public OpenAPI 3.1 document, llms.txt with Markdown twins, and Python and TypeScript SDKs that retry 408, 429 and 5xx with backoff"
        ],
        "url": "https://www.anchorterminal.com/tools/typesafe-jev",
        "verdict": "Typed answers with probabilities for `noul`, `choice` and `score` questions, many per call, with no text to parse. Early access behind a waitlist, with no free tier or free credits found.",
        "weaknesses": [
          "Early access behind a waitlist, with no free tier or free credits found",
          "No SLA, and the customer agreement promises only commercially reasonable efforts to give notice of API changes",
          "Published limits of 40 requests a second can change without notice"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Cheap, local classification, routing, triage and tool-call checks on short text inside Strands or other Python agents, and for teams who want to retrain a decision model from a published recipe.",
        "grade": "C",
        "name": "Strands Decider 2B",
        "position": 9,
        "price": "Free · OSS",
        "score": 61.3,
        "slug": "strands-decider",
        "strengths": [
          "Apache-2.0 code and weights, with the training recipe, data inventory and evaluation logs published",
          "Runs on CUDA, Apple silicon (MPS or MLX) or CPU, with a v19 median of 115 ms a question on an RTX 3090 per the README",
          "Brier score and expected calibration error published for each released checkpoint"
        ],
        "url": "https://www.anchorterminal.com/tools/strands-decider",
        "verdict": "A 1.9B-parameter Apache-2.0 decision model that runs on a laptop GPU, an Apple silicon Mac or a CPU, with its training data, recipe and per-version results published. It's an experimental 0.1.0 release with a 4,096-token window that cuts long states by default, and its local server has no authentication.",
        "weaknesses": [
          "Version 0.1.0, described as experimental in its package metadata, with no changelog file",
          "A 4,096-token window, and by default an over-long state is cut to fit without an error",
          "The local server has no authentication option"
        ],
        "where": "local",
        "x402": "no"
      },
      {
        "bestFor": "Topic, intent and sentiment routing over a known label set on the owner's own CPU or GPU, where many labels must be scored at once.",
        "grade": "D",
        "name": "GLiClass",
        "position": 10,
        "price": "Free · OSS",
        "score": 49.9,
        "slug": "gliclass",
        "strengths": [
          "Apache-2.0 code and weights, with `train.py`, the training datasets named on the model cards and an arXiv paper (2508.07662)",
          "One forward pass scores every label. The base v3.0 card reports 51.6 examples a second averaged over 1 to 128 labels on an A6000 (vendor figures)",
          "Single-label (softmax) and multi-label (sigmoid) modes, hierarchical label sets, task prompts and few-shot examples in one pipeline call"
        ],
        "url": "https://www.anchorterminal.com/tools/gliclass",
        "verdict": "An Apache-2.0 classifier that scores a whole label set in one encoder pass on the owner's hardware, with single-label, multi-label, hierarchical and few-shot modes. It returns label scores with no calibration claim, the bundled server has no authentication, and the last three test runs on the main branch, on 24 September 2026, failed.",
        "weaknesses": [
          "No calibration evidence. The cards report F1 only, and the docs tell users to calibrate thresholds on their own traffic",
          "The bundled server has no authentication, and `python -m gliclass.serve` binds to 0.0.0.0 by default",
          "The last three runs of the Tests workflow on main, all on 24 September 2026, failed"
        ],
        "where": "local",
        "x402": "no"
      }
    ],
    "updated": "2026-10-09"
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/best/decision-models/",
    "json": "https://www.anchorterminal.com/best/decision-models/index.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/best/decision-models/index.md",
    "slim": "https://www.anchorterminal.com/best/decision-models/index.min.md"
  },
  "markdown": "The 10 highest-scoring of 12 decision models on the Anchor benchmark, with a pick for each need and where each one falls short. Scores come from public evidence, re-checked as vendors change.\n\n- Ranked: 12 · agent-ready (BB or better): 1 · accept x402: 0 · hosted endpoints: 6\n- Full ranked table: https://www.anchorterminal.com/categories/decision-models.md\n- Head-to-head comparisons: https://www.anchorterminal.com/compare/decision-models/index.md (66)\n- Methodology: https://www.anchorterminal.com/benchmark/index.md\n\n## The shortlist\n\n| # | Tool | Grade | Score | Best for | Price | Where |\n| --- | --- | --- | --- | --- | --- | --- |\n| 1 | [OpenAI Decisions API](https://www.anchorterminal.com/tools/openai-decisions-api.md) | BB | 71.5 | High-volume yes or no checks, routing among fixed options and rubric scoring over text and images, for teams already on an OpenAI key. | Pay per use | hosted |\n| 2 | [Decider](https://www.anchorterminal.com/tools/decider.md) | B | 69.5 | Local classification, routing, triage and checks where a team wants open weights in several sizes and a Jev-shaped route. | Free · OSS | local |\n| 3 | [Laya](https://www.anchorterminal.com/tools/convai-laya.md) | B | 69.2 | Fast, cheap classification and routing on short text in many languages, as a base to fine-tune on your own labels, and as an MCP or LangGraph routing step. | Free · OSS | local |\n| 4 | [Kev](https://www.anchorterminal.com/tools/jaredpalmer-kev.md) | B | 67.4 | Self-hosted classification, routing, triage and rubric scoring where a probability matters, especially for teams already calling Jev who want the same API on their own hardware. | Free · OSS | local |\n| 5 | [Vela 2.0](https://www.anchorterminal.com/tools/vela.md) | B | 66.5 | Self-hosted routing and guardrail checks in one call, where span offsets for personal data or unsupported claims matter. | Free · OSS | local |\n| 6 | [Clef](https://www.anchorterminal.com/tools/cloudflare-clef.md) | B | 66.1 | Classification, routing and rubric scoring over text, JSON and images inside Cloudflare, or self-hosted where data can't leave. | Freemium | hosted |\n| 7 | [pplx-decider](https://www.anchorterminal.com/tools/pplx-decider.md) | B | 62.9 | High-volume classification, routing and rubric scoring over text and images where a team wants a low hosted price and the option to run the same weights itself. | Pay per use | hosted |\n| 8 | [Jev](https://www.anchorterminal.com/tools/typesafe-jev.md) | B | 62.1 | High-volume yes or no answers, labelling, routing and rubric scoring where a probability is more useful than prose, such as ticket triage, invoice checks or picking a tool or skill from a list. | Pay per use | hosted |\n| 9 | [Strands Decider 2B](https://www.anchorterminal.com/tools/strands-decider.md) | C | 61.3 | Cheap, local classification, routing, triage and tool-call checks on short text inside Strands or other Python agents, and for teams who want to retrain a decision model from a published recipe. | Free · OSS | local |\n| 10 | [GLiClass](https://www.anchorterminal.com/tools/gliclass.md) | D | 49.9 | Topic, intent and sentiment routing over a known label set on the owner's own CPU or GPU, where many labels must be scored at once. | Free · OSS | local |\n\n## Picks by need\n\n- Highest score overall: [OpenAI Decisions API](https://www.anchorterminal.com/tools/openai-decisions-api.md), BB, 71.5/100 on the benchmark. Also [Decider](https://www.anchorterminal.com/tools/decider.md), B, 69.5/100.\n- Reliability: [Decider](https://www.anchorterminal.com/tools/decider.md), 90/100 on reliability, against 53 for the overall leader.\n- Schema \u0026 documentation: [pplx-decider](https://www.anchorterminal.com/tools/pplx-decider.md), 90/100 on schema \u0026 documentation, against 89 for the overall leader.\n- Maintenance \u0026 community: [Decider](https://www.anchorterminal.com/tools/decider.md), 87/100 on maintenance \u0026 community, against 77 for the overall leader.\n- Transparency \u0026 trust: [Clef](https://www.anchorterminal.com/tools/cloudflare-clef.md), 86/100 on transparency \u0026 trust, against 83 for the overall leader.\n- Self-hosting under an open licence: [Decider](https://www.anchorterminal.com/tools/decider.md), self-hosted, Apache-2 licence. Also [Laya](https://www.anchorterminal.com/tools/convai-laya.md), self-hosted, Apache-2 licence.\n\n## How to choose\n\n- Calibration of stated probabilities: Check calibration on labelled cases, since a high stated probability that often turns out wrong sends an agent down the wrong branch.\n- Stable answers to option order: Check whether answers stay the same when the order of the options changes, since a routing label that depends on position sends work to the wrong queue.\n- Accuracy on your own labels: Measure accuracy on labelled decisions drawn from your own workflow, since the answer labels an agent acts on are the ones that count.\n- Cost per 1,000 decisions: Compare the cost per 1,000 decisions for your own label mix, since output length and retries add to the list price.\n\n- How the benchmark tests this category: A fixed set of labelled decisions (yes or no, one label from a set, a level on a rubric) sent to every model in the same request shape, with the same states and questions. We check accuracy against the labels, calibration (expected calibration error, Brier score and how often an answer given 0.9 or more is wrong), whether answers move when the option order changes, latency and the cost per 1,000 decisions.\n\n## Each one in detail\n\n### 1. OpenAI Decisions API, BB 71.5/100\n\nThe Decisions API is an OpenAI endpoint, in public beta since 6 October 2026, that reads text and images and returns typed answers with probabilities, a predicate, a choice from a fixed set or a rubric score.\n\n- Verdict: Typed predicate, choice and score answers with probabilities, up to 200 questions a call, at $0.10 per million input tokens with no output charge. It is a public beta released on 6 October 2026, with one model alias, no dated snapshot, no Batch route and no SLA found.\n- Choose it for: High-volume yes or no checks, routing among fixed options and rubric scoring over text and images, for teams already on an OpenAI key.\n- Strength: Three typed question kinds, predicate, choice and score, with per-option probabilities and up to 200 questions about one input in a call\n- Strength: $0.10 per million input tokens on `gpt-6-luna`, with no output, cache-read or cache-write charge\n- Strength: Text and up to 128 inline images in one request\n- Weakness: Public beta released on 6 October 2026. The guide expects general availability in the coming weeks and gives no date\n- Weakness: One model alias, `gpt-6-luna`, with no dated snapshot to pin\n- Weakness: Images must be inline base64 data URLs. Hosted image URLs and `file_id` inputs aren't accepted\n- Price: Pay per use · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/openai-decisions-api.md\n\n### 2. Decider, B 69.5/100\n\nDecider is a family of open-weight decision models by Mark Marosi (Mapika), from 0.8B to 35B parameters under Apache-2.0. It answers typed yes or no, choice and score questions with probabilities, and runs locally from the `decider-ai` Python package.\n\n- Verdict: An Apache-2.0 decision model family with a dated changelog, passing CI, 21 package releases since 22 September 2026 and model cards that list measured regressions. One person maintains it, the local server has no authentication option, states over 32,768 tokens are cut without an error, and no security policy is published.\n- Choose it for: Local classification, routing, triage and checks where a team wants open weights in several sizes and a Jev-shaped route.\n- Strength: Apache-2.0 code and weights, with the training code, data builders and per-version measurements in the repository\n- Strength: Sizes from 0.8B to 35B parameters, with GGUF files for CPU and builds for CUDA, Apple silicon and vLLM\n- Strength: `POST /v1/systemone` follows TypeSafe's wire format, and the README says TypeSafe's SDKs work with `TYPESAFE_BASE_URL` set to the local server\n- Weakness: The local server has no authentication option. It binds to 127.0.0.1 since 1.7.1\n- Weakness: States over 32,768 tokens are truncated (`DECIDER_MAX_STATE_TOKENS`) without an error\n- Weakness: No SECURITY.md, disclosure policy or security contact was found in the repository\n- Price: Free · OSS · Auth: None · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/decider.md\n- Against #1: https://www.anchorterminal.com/compare/decider-vs-openai-decisions-api.md\n\n### 3. Laya, B 69.2/100\n\nOpen-source decision engine from Convai Innovations, released under Apache-2.0.\n\n- Verdict: Apache-2.0 code and weights, installed with `pip install laya`, with Python 3.10 to 3.13 tested in CI. Base checkpoints score 0.362 and 0.352 on the maintainers' typed-decisions benchmark against a 0.318 random baseline, so it needs fine-tuning.\n- Choose it for: Fast, cheap classification and routing on short text in many languages, as a base to fine-tune on your own labels, and as an MCP or LangGraph routing step.\n- Strength: Apache-2.0 code and weights, installed with `pip install laya`, with Python 3.10 to 3.13 tested in CI\n- Strength: 421M and 322M-parameter encoders that the README times at 32.8 to 39.5 ms for one question on a Tesla T4\n- Strength: A Jev-compatible HTTP server with a batch route for up to 64 states, an 8-tool MCP server, and LangChain, LlamaIndex and CrewAI wrappers\n- Weakness: Base checkpoints score 0.362 and 0.352 on the maintainers' typed-decisions benchmark against a 0.318 random baseline, so it needs fine-tuning\n- Weakness: Choice options share a 192 or 256-token budget, and the README reports 0.425 on Banking77's 77 labels\n- Weakness: 512 tokens of context on the English checkpoint and 1,024 by default on the others\n- Price: Free · OSS · Auth: None · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/convai-laya.md\n- Against #1: https://www.anchorterminal.com/compare/convai-laya-vs-openai-decisions-api.md\n\n### 4. Kev, B 67.4/100\n\nKev is a family of four open-weight decision models by Jared Palmer, released together as Kev 1.0 on 1 October 2026 under Apache-2.0.\n\n- Verdict: Apache-2.0 code, adapters and heads on Apache-2.0 Qwen bases, with release tarballs and SHA-256 checksums for the 0.8B, 4B and 9B models. No package. `pip install kev` installs an unrelated 2021 ORM, so Kev runs from a Git clone with uv.\n- Choose it for: Self-hosted classification, routing, triage and rubric scoring where a probability matters, especially for teams already calling Jev who want the same API on their own hardware.\n- Strength: Apache-2.0 code, adapters and heads on Apache-2.0 Qwen bases, with release tarballs and SHA-256 checksums for the 0.8B, 4B and 9B models\n- Strength: The same `/v1/systemone` request and answer shapes as Jev, and the README says TypeSafe's Python SDK works against it unchanged\n- Strength: A fitted temperature per checkpoint, with Brier scores, calibration error and confident-error rates published for each model\n- Weakness: No package. `pip install kev` installs an unrelated 2021 ORM, so Kev runs from a Git clone with uv\n- Weakness: Kev-0.8B, 4B and 9B are validated to 8,192 tokens of state, though the server accepts 65,536\n- Weakness: Jared Palmer wrote 312 of the 333 commits we cloned\n- Price: Free · OSS · Auth: None · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/jaredpalmer-kev.md\n- Against #1: https://www.anchorterminal.com/compare/jaredpalmer-kev-vs-openai-decisions-api.md\n\n### 5. Vela 2.0, B 66.5/100\n\nVela 2.0 is a family of four open-weight decision models from the vLLM Semantic Router project and KR Labs, released on 6 October 2026 under Apache-2.0 for routing, safety checks, personal-data spans and hallucination checks.\n\n- Verdict: One self-hosted call answers routing, prompt-attack, personal-data and unsupported-claim questions with probabilities and character offsets, under Apache-2.0 with SHA-256 manifests. The models are days old and carry no Hub version tags, and the three larger sizes keep 74 to 89 per cent of their Decision 2.0 bases on the Jev Decision Index by the authors' figures.\n- Choose it for: Self-hosted routing and guardrail checks in one call, where span offsets for personal data or unsupported claims matter.\n- Strength: Five question types in one request (choice, noul, score, set and span), with span answers as labelled character offsets and a probability each\n- Strength: Apache-2.0 weights, code and documentation, ungated on Hugging Face, with safetensors files and a SHA256SUMS manifest in the three decoder repositories\n- Strength: Two serving routes. A bundled FastAPI server on `POST /v1/systemone`, and the router's model runtime with an OpenAPI 3.0.3 contract and Prometheus metrics\n- Weakness: No version tags on the four Hub repositories, and the 4B and 9B weights were replaced in place on 3 October 2026\n- Weakness: Loading with `transformers` needs `trust_remote_code=True`, which runs Python from the model repository\n- Weakness: The router's `vllm-sr serve MODEL` engine mode is newer than the 0.4.0 stable release and needs the development channel\n- Price: Free · OSS · Auth: None · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/vela.md\n- Against #1: https://www.anchorterminal.com/compare/openai-decisions-api-vs-vela.md\n\n### 6. Clef, B 66.1/100\n\nClef and Clef-flash are Cloudflare's open-weight decision models, released on 1 October 2026 under Apache-2.0 and hosted on Workers AI.\n\n- Verdict: Apache-2.0 weights for both models on Hugging Face, ungated, with no account needed to download. Released on 1 October 2026, with no service record and no entry in the Workers AI changelog we read.\n- Choose it for: Classification, routing and rubric scoring over text, JSON and images inside Cloudflare, or self-hosted where data can't leave.\n- Strength: Apache-2.0 weights for both models on Hugging Face, ungated, with no account needed to download\n- Strength: $0.24 per million input tokens for Clef and $0.09 for Clef-flash, with 10,000 free neurons a day on Workers AI\n- Strength: Typed answers with a probability for every allowed option, 1 to 64 questions a call, and up to 4 images\n- Weakness: Released on 1 October 2026, with no service record and no entry in the Workers AI changelog we read\n- Weakness: No limitations, failure modes or prompt-injection guidance in the model page or the launch post\n- Weakness: No SLA found for Workers AI, and Clef shares the Text Generation default of 300 requests a minute\n- Price: Freemium · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/cloudflare-clef.md\n- Against #1: https://www.anchorterminal.com/compare/cloudflare-clef-vs-openai-decisions-api.md\n\n### 7. pplx-decider, B 62.9/100\n\npplx-decider is Perplexity's multimodal decision model, a fine-tune of Qwen3.8-27B. It answers yes or no, choice and score questions about text, JSON and images with probabilities, through Perplexity's hosted Decisions API or from open weights on Hugging Face.\n\n- Verdict: The hosted Decisions API has an OpenAPI description, typed questions, documented limits and a published price of $0.02 per million input tokens, and the v1.1 weights are public under Apache-2.0. The API is days old, the official Python SDK has no method for it, and the governing terms and privacy notice could not be read on 9 October 2026.\n- Choose it for: High-volume classification, routing and rubric scoring over text and images where a team wants a low hosted price and the option to run the same weights itself.\n- Strength: Weights for v1.1 and v1 are public on Hugging Face under Apache-2.0, with training code, config and a data manifest in the repository\n- Strength: One request takes up to 128 questions about one `state`, mixing `noul`, `choice` and `score`, with under 262,144 input tokens\n- Strength: $0.02 per million input tokens with free output and no per-request fee, published without a login\n- Weakness: The Decisions API was released in October 2026, so its incident record is days long, and Perplexity's FAQ says it gives no uptime guarantee\n- Weakness: The official Python SDK at 0.43.8 has no Decisions method. The guide uses a plain HTTP client\n- Weakness: An image over 2,048 tiles is not rejected. The request waits about a minute and returns 504\n- Price: Pay per use · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/pplx-decider.md\n- Against #1: https://www.anchorterminal.com/compare/openai-decisions-api-vs-pplx-decider.md\n\n### 8. Jev, B 62.1/100\n\nJev is TypeSafe AI's first System One model, a closed decision model behind an HTTP API.\n\n- Verdict: Typed answers with probabilities for `noul`, `choice` and `score` questions, many per call, with no text to parse. Early access behind a waitlist, with no free tier or free credits found.\n- Choose it for: High-volume yes or no answers, labelling, routing and rubric scoring where a probability is more useful than prose, such as ticket triage, invoice checks or picking a tool or skill from a list.\n- Strength: Typed answers with probabilities for `noul`, `choice` and `score` questions, many per call, with no text to parse\n- Strength: $0.042 per million input tokens, with output tokens free\n- Strength: A public OpenAPI 3.1 document, llms.txt with Markdown twins, and Python and TypeScript SDKs that retry 408, 429 and 5xx with backoff\n- Weakness: Early access behind a waitlist, with no free tier or free credits found\n- Weakness: No SLA, and the customer agreement promises only commercially reasonable efforts to give notice of API changes\n- Weakness: Published limits of 40 requests a second can change without notice\n- Price: Pay per use · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/typesafe-jev.md\n- Against #1: https://www.anchorterminal.com/compare/openai-decisions-api-vs-typesafe-jev.md\n\n### 9. Strands Decider 2B, C 61.3/100\n\nStrands Decider 2B is an open-weight decision model from AWS's Strands Labs, released on 1 October 2026 under Apache-2.0. It answers typed yes or no, choice and score questions with probabilities, and runs locally from a Python package.\n\n- Verdict: A 1.9B-parameter Apache-2.0 decision model that runs on a laptop GPU, an Apple silicon Mac or a CPU, with its training data, recipe and per-version results published. It's an experimental 0.1.0 release with a 4,096-token window that cuts long states by default, and its local server has no authentication.\n- Choose it for: Cheap, local classification, routing, triage and tool-call checks on short text inside Strands or other Python agents, and for teams who want to retrain a decision model from a published recipe.\n- Strength: Apache-2.0 code and weights, with the training recipe, data inventory and evaluation logs published\n- Strength: Runs on CUDA, Apple silicon (MPS or MLX) or CPU, with a v19 median of 115 ms a question on an RTX 3090 per the README\n- Strength: Brier score and expected calibration error published for each released checkpoint\n- Weakness: Version 0.1.0, described as experimental in its package metadata, with no changelog file\n- Weakness: A 4,096-token window, and by default an over-long state is cut to fit without an error\n- Weakness: The local server has no authentication option\n- Price: Free · OSS · Auth: None · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/strands-decider.md\n- Against #1: https://www.anchorterminal.com/compare/openai-decisions-api-vs-strands-decider.md\n\n### 10. GLiClass, D 49.9/100\n\nGLiClass is an open-source Python library and family of open-weight zero-shot text classifiers from Knowledgator. It scores every candidate label in one forward pass and runs locally through a pipeline or a Ray Serve endpoint.\n\n- Verdict: An Apache-2.0 classifier that scores a whole label set in one encoder pass on the owner's hardware, with single-label, multi-label, hierarchical and few-shot modes. It returns label scores with no calibration claim, the bundled server has no authentication, and the last three test runs on the main branch, on 24 September 2026, failed.\n- Choose it for: Topic, intent and sentiment routing over a known label set on the owner's own CPU or GPU, where many labels must be scored at once.\n- Strength: Apache-2.0 code and weights, with `train.py`, the training datasets named on the model cards and an arXiv paper (2508.07662)\n- Strength: One forward pass scores every label. The base v3.0 card reports 51.6 examples a second averaged over 1 to 128 labels on an A6000 (vendor figures)\n- Strength: Single-label (softmax) and multi-label (sigmoid) modes, hierarchical label sets, task prompts and few-shot examples in one pipeline call\n- Weakness: No calibration evidence. The cards report F1 only, and the docs tell users to calibrate thresholds on their own traffic\n- Weakness: The bundled server has no authentication, and `python -m gliclass.serve` binds to 0.0.0.0 by default\n- Weakness: The last three runs of the Tests workflow on main, all on 24 September 2026, failed\n- Price: Free · OSS · Auth: None · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/gliclass.md\n- Against #1: https://www.anchorterminal.com/compare/gliclass-vs-openai-decisions-api.md\n\n2 more are ranked in the full table: https://www.anchorterminal.com/categories/decision-models.md\n\n## Head to head\n\n- [Decider vs OpenAI Decisions API](https://www.anchorterminal.com/compare/decider-vs-openai-decisions-api.md)\n- [Laya vs OpenAI Decisions API](https://www.anchorterminal.com/compare/convai-laya-vs-openai-decisions-api.md)\n- [Kev vs OpenAI Decisions API](https://www.anchorterminal.com/compare/jaredpalmer-kev-vs-openai-decisions-api.md)\n- [OpenAI Decisions API vs Vela 2.0](https://www.anchorterminal.com/compare/openai-decisions-api-vs-vela.md)\n- [Laya vs Decider](https://www.anchorterminal.com/compare/convai-laya-vs-decider.md)\n- [Decider vs Kev](https://www.anchorterminal.com/compare/decider-vs-jaredpalmer-kev.md)\n- [Decider vs Vela 2.0](https://www.anchorterminal.com/compare/decider-vs-vela.md)\n- [Laya vs Kev](https://www.anchorterminal.com/compare/convai-laya-vs-jaredpalmer-kev.md)\n- [Laya vs Vela 2.0](https://www.anchorterminal.com/compare/convai-laya-vs-vela.md)\n- [Kev vs Vela 2.0](https://www.anchorterminal.com/compare/jaredpalmer-kev-vs-vela.md)\n\n## Questions\n\n### What are the highest-rated decision models for AI agents?\n\nOpenAI Decisions API has the highest benchmark score of the 12 ranked decision models, 71.5 (BB). Decider is second with 69.5 (B).\n\n### How many decision models are agent-ready?\n\n1 of the 12 ranked here grade BB or better, the bar for agent-ready on the Anchor benchmark.\n\n### Which decision models accept x402 payments?\n\nNone of the ranked listings here accepts x402 for its main call yet.\n\n### How is this list ranked?\n\nBy the Anchor benchmark score out of 100, a weighted mean of the scored categories minus deductions for negative events, from public evidence re-checked as vendors change. Listings cannot pay for a place. The latest assessment behind this page is from 9 October 2026.\n\n## How this list is made\n\nThe order is the Anchor benchmark score, the same number as on each listing. Each listing is graded from public evidence against the benchmark checklist, and the picks are worked out from those grades, prices and facts. No listing pays for its place, and paid audits or listing help never change a score.\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-10",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Best of",
        "url": "https://www.anchorterminal.com/best/"
      },
      {
        "name": "Decision models",
        "url": ""
      }
    ],
    "description": "OpenAI Decisions API (BB), Decider (B) and Laya (B) lead the 12 ranked decision models. Picks by need, strengths, weaknesses and prices from the Anchor benchmark.",
    "facts": [
      "OpenAI Decisions API BB",
      "Decider B",
      "Laya B"
    ],
    "h1": "Best decision models for AI agents",
    "image": "https://www.anchorterminal.com/assets/og/best-decision-models.png",
    "path": "/best/decision-models/",
    "published": "",
    "section": "tools",
    "title": "Best decision models for AI agents in 2026, ranked | Anchor Terminal",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/best/decision-models/"
  },
  "tokens": {
    "markdown": 6100,
    "slim": 1480
  },
  "version": 1
}
