{
  "benchmark": {
    "name": "Anchor Agent Tool Benchmark",
    "shortName": "Agent Tool Benchmark",
    "version": "0.3",
    "agentReadyGrade": "BB",
    "maxNegative": -15,
    "categories": [
      {
        "key": "reliability",
        "name": "Reliability",
        "weight": 16,
        "description": "Does the tool answer, and does it keep answering the same way? In this run it's assessed from public evidence (90 days of status history, documented rate limits and overload handling, SLAs, general availability). Our own probes from three regions will replace that evidence as they accumulate.",
        "signals": [
          "30-day availability of the endpoint (remote) or of a clean start plus tools/list (local)",
          "Error rate on a fixed set of representative calls",
          "Timeout rate and behaviour under back-pressure (429 with Retry-After, not silent hangs)",
          "Consistency, meaning the same input gives the same shape of output across the window",
          "Graceful degradation when upstream systems are down"
        ]
      },
      {
        "key": "performance",
        "name": "Performance",
        "weight": 10,
        "description": "How long an agent waits. Latency is measured per call, not per page load, because a slow tool costs an agent a whole turn.",
        "signals": [
          "p50 and p95 latency of representative calls",
          "Cold-start time for local servers and first-call time for hosted ones",
          "Streaming or partial results where the task is long-running",
          "Payload size discipline (results sized for a context window rather than a data warehouse)"
        ],
        "pending": true,
        "pendingNote": "Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes."
      },
      {
        "key": "schema",
        "name": "Schema \u0026 documentation",
        "weight": 13,
        "description": "Can a model understand what the tool does from the tool definition alone? We grade the text a model sees.",
        "signals": [
          "Tool descriptions state purpose, when to use, and when not to",
          "Input schemas with types, enums, constraints and required fields; no free-form JSON blobs",
          "Examples in descriptions or docs; documented error responses",
          "Versioning of the tool surface and a changelog",
          "Machine-readable docs (llms.txt, OpenAPI, registry server.json)"
        ]
      },
      {
        "key": "ergonomics",
        "name": "Agent ergonomics",
        "weight": 13,
        "description": "How much of an agent's context and how many round trips the tool consumes to get a job done.",
        "signals": [
          "Context cost, the tokens tools/list consumes before any work starts",
          "Pagination, filtering and output-size controls",
          "Actionable error messages an agent can recover from without a human",
          "Idempotency and safe retries; read-only variants and tool annotations (readOnlyHint, destructiveHint)",
          "Sensible defaults and few mandatory parameters; toolsets or meta-tools when the surface is large"
        ]
      },
      {
        "key": "security",
        "name": "Security \u0026 auth",
        "weight": 14,
        "description": "Can an operator give an agent this tool without giving it the keys to everything?",
        "signals": [
          "OAuth 2.1 with scoped, revocable, short-lived credentials; no secrets in URLs",
          "Read-only and scoped modes; confirmation for destructive actions",
          "Prompt-injection mitigations for tools that return untrusted content",
          "Audit logging and per-call visibility for the operator",
          "Handling of advisories (disclosed, patched, communicated)"
        ]
      },
      {
        "key": "payments",
        "name": "Payments \u0026 pricing",
        "weight": 10,
        "description": "Can an agent start using the tool, and pay for it, without a human in the loop? A machine payment protocol on the tool's own endpoints is the largest single signal.",
        "signals": [
          "A machine payment protocol (x402, MPP or L402) on the tool's own endpoints (40 points)",
          "Transparent, per-call pricing with units, published without a login (20 points)",
          "A free tier or trial that doesn't require a card (20 points)",
          "Autonomous onboarding, meaning an agent can obtain access without a human signup flow (20 points)"
        ]
      },
      {
        "key": "tasks",
        "name": "Task success",
        "weight": 10,
        "description": "Does an agent finish the job? A fixed suite of representative tasks per category, run monthly with a fixed reference model through the tool. For data providers, half of this category is the data-quality score, because a clean API over thin data still fails the task.",
        "signals": [
          "For data providers, the data-quality score from /benchmark/#data-quality (half the category)",
          "Pass rate on the category task suite (pass@1 and pass@4)",
          "Retries, turns and tokens per completed task",
          "Recovery, whether an error on the first attempt is fixable from the tool's own error message",
          "Consistency across reference models"
        ],
        "pending": true,
        "pendingNote": "Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored."
      },
      {
        "key": "maintenance",
        "name": "Maintenance \u0026 community",
        "weight": 7,
        "description": "Is anyone home? Release cadence and responsiveness predict how a tool behaves after the protocol moves under it.",
        "signals": [
          "Release cadence in the last 90 days and time since the last release",
          "Issue and pull-request responsiveness",
          "Presence in the official MCP registry under a verified namespace",
          "Package health (current SDK versions, no pinned-and-forgotten dependencies)"
        ]
      },
      {
        "key": "transparency",
        "name": "Transparency \u0026 trust",
        "weight": 7,
        "description": "Can an operator find out who runs the tool, what it does with data, and what changed? Half of this category is the provenance score, computed from checked facts. Legal entity, domain age, whether the endpoint sits on the vendor's own domain, terms, privacy policy, status page, changelog and security.txt.",
        "signals": [
          "Provenance score, computed from the checks on /benchmark/#provenance (half the category)",
          "Source availability and licence clarity",
          "Data handling and retention statements that agree with each other",
          "Deprecation notices with dates",
          "Telemetry disclosed and opt-out documented"
        ]
      }
    ],
    "negative": {
      "key": "negative",
      "name": "Negative events",
      "description": "Deductions of up to 15 points for incidents in the last 12 months. Security incidents, breaking changes shipped without notice, silent deprecations, unresolved advisories, or misleading listings. Deductions decay over time and are lifted early when the vendor documents the fix.",
      "examples": [
        "Confirmed data exfiltration path via the tool (up to -15)",
        "Breaking tool rename or schema change without a deprecation window (-3 to -8)",
        "Endpoint removed while still advertised in a registry or docs (-3 to -6)",
        "Telemetry enabled without disclosure (-2 to -5)"
      ]
    },
    "grades": [
      {
        "grade": "AA",
        "min": 85,
        "label": "Exceptional. Agent-ready, with agent-native payments or equivalent autonomy."
      },
      {
        "grade": "A",
        "min": 78,
        "label": "Excellent. Agent-ready; minor gaps."
      },
      {
        "grade": "BB",
        "min": 70,
        "label": "Good. Agent-ready with documented caveats."
      },
      {
        "grade": "B",
        "min": 62,
        "label": "Usable. Needs operator supervision or workarounds."
      },
      {
        "grade": "C",
        "min": 54,
        "label": "Mixed. Material gaps in one or more categories."
      },
      {
        "grade": "D",
        "min": 46,
        "label": "Poor. Not recommended for autonomous use."
      },
      {
        "grade": "E",
        "min": 38,
        "label": "Very poor."
      },
      {
        "grade": "F",
        "min": 0,
        "label": "Failing or unverifiable."
      }
    ],
    "principles": [
      "We score what a model sees and what an agent experiences, not marketing pages.",
      "Every score has a category, a weight and a dated history. Changes are explained.",
      "Vendors can't pay for placement. Reports are paid; rankings aren't.",
      "We don't host tools we rank. letme passes calls through and measures; it isn't a marketplace, and selling through it has no effect on a grade.",
      "Agent-native payments are weighted up on purpose. A tool an agent can't pay for is a tool an agent can't use alone.",
      "Methodology is versioned. Re-runs are published with the version that produced them.",
      "Who stands behind a listing is checked and published, line by line, so anyone can recompute the provenance score from the same sources.",
      "A reviewer on the panel never grades the company whose model it runs on.",
      "Payment protocols are graded but not ranked against tools. You choose a protocol by who you're paying, not by its score."
    ]
  },
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-05",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.3",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "run": {
    "version": "0.3",
    "runLabel": "October 2026 research run",
    "runDate": "2026-10-01",
    "window": "2026-09-30/2026-10-02",
    "probeRegions": [
      "eu-west (London)",
      "us-east (Virginia)",
      "ap-southeast (Singapore)"
    ],
    "note": "Every grade comes from public evidence checked between 30 September and 2 October 2026 by research agents running on Anthropic's Claude models, against the checklist published on /benchmark/. Each score carries its reason and sources on the listing. Performance and Task success need our probes and task suites, which haven't run yet, so they're pending and their weight is shared across the seven assessed categories. Panel reviews are desk reviews written from the same evidence; nobody called, paid for or timed a tool to write them."
  },
  "scoring": {
    "assessedWeight": 80,
    "effectiveWeights": {
      "ergonomics": 16.25,
      "maintenance": 8.75,
      "payments": 12.5,
      "performance": 0,
      "reliability": 20,
      "schema": 16.25,
      "security": 17.5,
      "tasks": 0,
      "transparency": 8.75
    },
    "pending": [
      "performance",
      "tasks"
    ],
    "total": "score = sum of (category score × weight) over the assessed categories ÷ 80, then the negative events (0 to -15). A pending category has no score and its weight is shared across the assessed ones until it's scored."
  },
  "stats": {
    "agentReady": 105,
    "avgScore": 59.7,
    "gradeCounts": {
      "A": 17,
      "AA": 1,
      "B": 120,
      "BB": 87,
      "C": 108,
      "D": 67,
      "E": 36,
      "F": 26
    },
    "notGraded": 0,
    "reviews": 1214,
    "tools": 462,
    "x402": 28
  }
}
