{
  "data": {
    "similar": [
      {
        "grade": "BB",
        "json": "https://www.anchorterminal.com/tools/groq.json",
        "name": "GroqCloud",
        "score": 75.6,
        "shared": [
          "inference.llm",
          "inference.fast",
          "inference.open-weights"
        ],
        "slug": "groq"
      },
      {
        "grade": "BB",
        "json": "https://www.anchorterminal.com/tools/mistral-api.json",
        "name": "Mistral AI API",
        "score": 71.2,
        "shared": [
          "inference.llm",
          "inference.open-weights"
        ],
        "slug": "mistral-api"
      },
      {
        "grade": "C",
        "json": "https://www.anchorterminal.com/tools/docker-model-runner.json",
        "name": "Docker Model Runner",
        "score": 57.1,
        "shared": [
          "inference.open-weights",
          "inference.llm"
        ],
        "slug": "docker-model-runner"
      },
      {
        "grade": "C",
        "json": "https://www.anchorterminal.com/tools/ollama.json",
        "name": "Ollama",
        "score": 56.3,
        "shared": [
          "inference.open-weights",
          "inference.llm"
        ],
        "slug": "ollama"
      },
      {
        "grade": "D",
        "json": "https://www.anchorterminal.com/tools/deepseek-api.json",
        "name": "DeepSeek API",
        "score": 46.8,
        "shared": [
          "inference.llm",
          "inference.open-weights"
        ],
        "slug": "deepseek-api"
      },
      {
        "grade": "A",
        "json": "https://www.anchorterminal.com/tools/openai-api.json",
        "name": "OpenAI API",
        "score": 83.3,
        "shared": [
          "inference.llm"
        ],
        "slug": "openai-api"
      }
    ],
    "tool": {
      "slug": "prism-inference",
      "name": "Prism Inference",
      "vendor": "Prism Technologies Inc",
      "vendorUrl": "https://prisminference.com",
      "kind": "model",
      "category": "inference",
      "summary": "Prism is a hosted inference API from Prism Technologies Inc for open-weight models, aimed at coding agents. It accepts OpenAI Chat Completions, OpenAI Responses and Anthropic Messages requests at api.prisminference.com. It launched on 24 September 2026.",
      "url": "https://www.anchorterminal.com/tools/prism-inference",
      "markdownUrl": "https://www.anchorterminal.com/tools/prism-inference.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/prism-inference.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/prism-inference.json",
      "repo": "https://github.com/prismhq/hermes-prism-provider",
      "license": "Proprietary service under Prism's terms of service. The OpenAPI file declares `LicenseRef-Proprietary`. The Hermes provider plugin repository carries no licence file",
      "transports": [
        "http"
      ],
      "remoteUrl": "https://api.prisminference.com/v1",
      "packages": [],
      "auth": "api-key",
      "authNotes": "`Authorization: Bearer` with a key issued from account settings after sign-up at prisminference.com/signup. The inference endpoints also accept the key in `x-api-key`. A missing, invalid, expired or revoked key returns 401. No key scopes were found in the reviewed documentation. An agent can call `POST https://prisminference.com/api/agent-signups` with the owner's email and a username and receive a key once, but that key can't run inference until the owner supplies a six-digit emailed code, and it expires after 30 days. `GET /v1/models` needs no key.",
      "pricing": "usage",
      "pricingNotes": "Prepaid per-token pricing with no minimum. DeepSeek-V4.1-Flash is $0.09 in, $1.20 out and $0.06 cache read per million tokens, and Gemma 4 31B is $0.30, $0.40 and $0.15 (https://prisminference.com/pricing, matched by https://api.prisminference.com/v1/models). No free tier or trial credit was found, and a workspace without credit gets 402. A person funds the workspace in a browser. Elastic endpoints, dedicated deployments and batch are sold through sales with no published price.",
      "priceSummary": "from $0.09 / 1M in",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in llms.txt, the docs index, the OpenAPI file or the pricing page, read 2026-10-08. The docs describe prepaid credit funded by a person in the browser.",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": null,
        "npmWeekly": null,
        "pypiWeekly": null,
        "asOf": "2026-10-08"
      },
      "docsUrl": "https://docs.prisminference.com",
      "rateLimitsUrl": "https://docs.prisminference.com/rate-limits",
      "llmsTxt": "https://prisminference.com/llms.txt",
      "openapi": "https://docs.prisminference.com/openapi.yaml",
      "capabilities": [
        "inference.fast",
        "inference.open-weights",
        "inference.llm"
      ],
      "tags": [
        "hosted",
        "model",
        "open-weights",
        "fast",
        "usage-priced",
        "prepaid",
        "openapi",
        "llms-txt",
        "openai-compatible",
        "anthropic-compatible",
        "zero-retention",
        "status-page",
        "new"
      ],
      "lastRelease": "2026-10-06",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 60.1,
        "grade": "C",
        "agentReady": false,
        "rank": 363,
        "ranked": true,
        "rankOf": 629,
        "categoryRank": 8,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 68,
          "maintenance": 49,
          "payments": 30,
          "reliability": 65,
          "schema": 82,
          "security": 65,
          "transparency": 61
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "breakdown": [
          {
            "key": "reliability",
            "name": "Reliability",
            "weight": 16,
            "effectiveWeight": 20,
            "score": 65,
            "points": 13,
            "reason": "Hosted reading. Status page at status.prisminference.com on incident.io with a component for each of the two models (20). The incidents feed is empty and the page shows 100% for both models, but the service launched on 24 September 2026 and the domain dates from 9 September, so the record covers about two weeks of a 90-day window. We give half of the 30 for a clean record and say so as a departure from the checklist (15). The docs say per-key limits exist and publish no numbers, only a 4 MiB request cap (5). 429 carries `Retry-After` in seconds, errors carry a `retryable` flag, and the docs give backoff with jitter and rules for retrying interrupted streams (15). No published SLA. The terms say no guaranteed uptime without a separate written agreement, although the home page shows a '99.99% Uptime SLA' tile (0). The serverless API isn't labelled beta or preview, though the OpenAPI file is at version 0.1.0 (10)."
          },
          {
            "key": "performance",
            "name": "Performance",
            "weight": 10,
            "effectiveWeight": 0,
            "pending": true,
            "points": 0,
            "reason": "Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes."
          },
          {
            "key": "schema",
            "name": "Schema \u0026 documentation",
            "weight": 13,
            "effectiveWeight": 16.25,
            "score": 82,
            "points": 13.33,
            "reason": "A public OpenAPI 3.1 file covering ten paths, including chat completions, responses, messages and models (25). llms.txt on both the site and the docs, a fuller llms-full.txt, and every docs page served as Markdown (10). Reference pages state what each endpoint is for and what isn't supported, such as stored responses and hosted tools, and the models page has a 'use when' column (15 of 20). `model` and `messages` are required, with ranges on `temperature` and `top_p` and an enum for reasoning effort, but `tool_choice` is untyped and `response_format` and the request body allow any extra properties (10 of 15). Examples in curl, TypeScript and Python, and an errors page with 13 status codes and five named codes with fixes (14 of 15). Paths are versioned under /v1 and the changelog is dated, with a single entry on 6 October 2026 (8 of 15)."
          },
          {
            "key": "ergonomics",
            "name": "Agent ergonomics",
            "weight": 13,
            "effectiveWeight": 16.25,
            "score": 68,
            "points": 11.05,
            "reason": "Model reading of the checklist (tool use, structured output, caching, context, batch, SDKs, errors). OpenAI function tools, Anthropic tool-use blocks, `tool_choice` and `parallel_tool_calls` on Responses are documented (18 of 20). JSON mode and strict JSON Schema on Chat Completions, and `text.format` on Responses (15). A cache-read price is published for both models, $0.06 against $0.09 on DeepSeek-V4.1-Flash, but no page explains how caching is triggered, and the Claude Code guide says prompt-cache hints are accepted and ignored (5 of 15). 1M-token context and 384,000 output tokens on DeepSeek-V4.1-Flash, 32,768 on Gemma 4 31B (15). Batch appears on the pricing page as a sales-led tier with no API in the OpenAPI file (0). No SDK of its own. OpenAI and Anthropic clients are used with a base URL change (0). Every error carries `code`, `retryable`, `fix` and `docs_url`, which we confirmed on a live 401 (15)."
          },
          {
            "key": "security",
            "name": "Security \u0026 auth",
            "weight": 14,
            "effectiveWeight": 17.5,
            "score": 65,
            "points": 11.38,
            "reason": "Model reading. Bearer keys issued from account settings, with 401 for revoked or expired keys and a 30-day expiry on keys from agent sign-up. No scopes or per-key limits were found in the reviewed documentation (20 of 30). The privacy policy, the terms and the docs all say inputs and outputs are never used to train, fine-tune or evaluate models (20). Zero data retention is the default on every tier with no setting to turn on (15). The docs mention usage in the dashboard, which we did not see, and no audit log was found (5 of 15). security.txt is valid until 6 October 2027 with a contact address. No disclosure policy, bug bounty, SOC 2 or ISO 27001 was found (5 of 20)."
          },
          {
            "key": "payments",
            "name": "Payments \u0026 pricing",
            "weight": 10,
            "effectiveWeight": 12.5,
            "score": 30,
            "points": 3.75,
            "reason": "No machine payment protocol (0). Per-token prices are on the pricing page and in the keyless `/v1/models` response, with no login (20). No free tier or trial credit was found. Billing is prepaid and a workspace without credit gets 402 (0). An agent can request a key from `POST /api/agent-signups`, but the key can't run inference until the owner supplies an emailed code, and a person funds the workspace in a browser, so half credit (10 of 20)."
          },
          {
            "key": "tasks",
            "name": "Task success",
            "weight": 10,
            "effectiveWeight": 0,
            "pending": true,
            "points": 0,
            "reason": "Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored."
          },
          {
            "key": "maintenance",
            "name": "Maintenance \u0026 community",
            "weight": 7,
            "effectiveWeight": 8.75,
            "score": 49,
            "points": 4.29,
            "reason": "Model reading. Gemma 4 31B was added to the catalogue on 6 October 2026 (30). No notice period is stated. The terms promise advance notice of a material breaking change 'when reasonably practicable', and nothing has been retired yet (3 of 12). No shutdowns in the service's two weeks of public life (8 of 8). A dated changelog with one entry, a Discord server and a support booking link, none of which we tested for replies (8 of 15). No SDK repository to sample. The Hermes provider plugin on GitHub has three commits, all on 15 September 2026 (0 of 10). No official SDKs (0 of 15) and no packages to assess (0 of 10)."
          },
          {
            "key": "transparency",
            "name": "Transparency \u0026 trust",
            "weight": 7,
            "effectiveWeight": 8.75,
            "score": 61,
            "points": 5.34,
            "note": "editorial 43, provenance 79",
            "reason": "Closed service with clear terms that name Prism Technologies Inc (15). The privacy policy, the terms and the zero-retention page agree on what is and isn't kept. Service metadata is kept 'only for as long as reasonably needed' with no period, and no DPA is published (20 of 30). No deprecation policy or dated notices, only the terms' advance notice 'when reasonably practicable' (3 of 20). The privacy policy says Prism is based in the United States and names vendors by category only, with no sub-processor list or regions (5 of 20)."
          }
        ],
        "assessment": {
          "date": "2026-10-08",
          "basis": "public evidence",
          "confidence": "medium",
          "notes": {
            "ergonomics": "Model reading of the checklist (tool use, structured output, caching, context, batch, SDKs, errors). OpenAI function tools, Anthropic tool-use blocks, `tool_choice` and `parallel_tool_calls` on Responses are documented (18 of 20). JSON mode and strict JSON Schema on Chat Completions, and `text.format` on Responses (15). A cache-read price is published for both models, $0.06 against $0.09 on DeepSeek-V4.1-Flash, but no page explains how caching is triggered, and the Claude Code guide says prompt-cache hints are accepted and ignored (5 of 15). 1M-token context and 384,000 output tokens on DeepSeek-V4.1-Flash, 32,768 on Gemma 4 31B (15). Batch appears on the pricing page as a sales-led tier with no API in the OpenAPI file (0). No SDK of its own. OpenAI and Anthropic clients are used with a base URL change (0). Every error carries `code`, `retryable`, `fix` and `docs_url`, which we confirmed on a live 401 (15).",
            "maintenance": "Model reading. Gemma 4 31B was added to the catalogue on 6 October 2026 (30). No notice period is stated. The terms promise advance notice of a material breaking change 'when reasonably practicable', and nothing has been retired yet (3 of 12). No shutdowns in the service's two weeks of public life (8 of 8). A dated changelog with one entry, a Discord server and a support booking link, none of which we tested for replies (8 of 15). No SDK repository to sample. The Hermes provider plugin on GitHub has three commits, all on 15 September 2026 (0 of 10). No official SDKs (0 of 15) and no packages to assess (0 of 10).",
            "payments": "No machine payment protocol (0). Per-token prices are on the pricing page and in the keyless `/v1/models` response, with no login (20). No free tier or trial credit was found. Billing is prepaid and a workspace without credit gets 402 (0). An agent can request a key from `POST /api/agent-signups`, but the key can't run inference until the owner supplies an emailed code, and a person funds the workspace in a browser, so half credit (10 of 20).",
            "reliability": "Hosted reading. Status page at status.prisminference.com on incident.io with a component for each of the two models (20). The incidents feed is empty and the page shows 100% for both models, but the service launched on 24 September 2026 and the domain dates from 9 September, so the record covers about two weeks of a 90-day window. We give half of the 30 for a clean record and say so as a departure from the checklist (15). The docs say per-key limits exist and publish no numbers, only a 4 MiB request cap (5). 429 carries `Retry-After` in seconds, errors carry a `retryable` flag, and the docs give backoff with jitter and rules for retrying interrupted streams (15). No published SLA. The terms say no guaranteed uptime without a separate written agreement, although the home page shows a '99.99% Uptime SLA' tile (0). The serverless API isn't labelled beta or preview, though the OpenAPI file is at version 0.1.0 (10).",
            "schema": "A public OpenAPI 3.1 file covering ten paths, including chat completions, responses, messages and models (25). llms.txt on both the site and the docs, a fuller llms-full.txt, and every docs page served as Markdown (10). Reference pages state what each endpoint is for and what isn't supported, such as stored responses and hosted tools, and the models page has a 'use when' column (15 of 20). `model` and `messages` are required, with ranges on `temperature` and `top_p` and an enum for reasoning effort, but `tool_choice` is untyped and `response_format` and the request body allow any extra properties (10 of 15). Examples in curl, TypeScript and Python, and an errors page with 13 status codes and five named codes with fixes (14 of 15). Paths are versioned under /v1 and the changelog is dated, with a single entry on 6 October 2026 (8 of 15).",
            "security": "Model reading. Bearer keys issued from account settings, with 401 for revoked or expired keys and a 30-day expiry on keys from agent sign-up. No scopes or per-key limits were found in the reviewed documentation (20 of 30). The privacy policy, the terms and the docs all say inputs and outputs are never used to train, fine-tune or evaluate models (20). Zero data retention is the default on every tier with no setting to turn on (15). The docs mention usage in the dashboard, which we did not see, and no audit log was found (5 of 15). security.txt is valid until 6 October 2027 with a contact address. No disclosure policy, bug bounty, SOC 2 or ISO 27001 was found (5 of 20).",
            "transparency": "Closed service with clear terms that name Prism Technologies Inc (15). The privacy policy, the terms and the zero-retention page agree on what is and isn't kept. Service metadata is kept 'only for as long as reasonably needed' with no period, and no DPA is published (20 of 30). No deprecation policy or dated notices, only the terms' advance notice 'when reasonably practicable' (3 of 20). The privacy policy says Prism is based in the United States and names vendors by category only, with no sub-processor list or regions (5 of 20)."
          },
          "sources": [
            {
              "what": "llms.txt, endpoints, models and agent sign-up steps",
              "url": "https://prisminference.com/llms.txt",
              "seen": "2026-10-08"
            },
            {
              "what": "full agent context, billing and compatibility limits",
              "url": "https://prisminference.com/llms-full.txt",
              "seen": "2026-10-08"
            },
            {
              "what": "docs index",
              "url": "https://docs.prisminference.com/llms.txt",
              "seen": "2026-10-08"
            },
            {
              "what": "OpenAPI 3.1 file, version 0.1.0",
              "url": "https://docs.prisminference.com/openapi.yaml",
              "seen": "2026-10-08"
            },
            {
              "what": "keyless model catalogue with prices",
              "url": "https://api.prisminference.com/v1/models",
              "seen": "2026-10-08"
            },
            {
              "what": "pricing page",
              "url": "https://prisminference.com/pricing",
              "seen": "2026-10-08"
            },
            {
              "what": "machine-readable pricing tiers",
              "url": "https://prisminference.com/pricing.json",
              "seen": "2026-10-08"
            },
            {
              "what": "models page, Gemma marked request access",
              "url": "https://docs.prisminference.com/models",
              "seen": "2026-10-08"
            },
            {
              "what": "rate limits",
              "url": "https://docs.prisminference.com/rate-limits",
              "seen": "2026-10-08"
            },
            {
              "what": "errors and retry policy",
              "url": "https://docs.prisminference.com/errors",
              "seen": "2026-10-08"
            },
            {
              "what": "authentication",
              "url": "https://docs.prisminference.com/authentication",
              "seen": "2026-10-08"
            },
            {
              "what": "agent sign-up",
              "url": "https://docs.prisminference.com/guides/agent-signup",
              "seen": "2026-10-08"
            },
            {
              "what": "zero data retention",
              "url": "https://docs.prisminference.com/zero-data-retention",
              "seen": "2026-10-08"
            },
            {
              "what": "changelog",
              "url": "https://docs.prisminference.com/changelog",
              "seen": "2026-10-08"
            },
            {
              "what": "Chat Completions reference",
              "url": "https://docs.prisminference.com/api-reference/chat-completions",
              "seen": "2026-10-08"
            },
            {
              "what": "Responses reference",
              "url": "https://docs.prisminference.com/api-reference/responses",
              "seen": "2026-10-08"
            },
            {
              "what": "Anthropic Messages reference",
              "url": "https://docs.prisminference.com/api-reference/messages",
              "seen": "2026-10-08"
            },
            {
              "what": "structured outputs guide",
              "url": "https://docs.prisminference.com/guides/structured-outputs",
              "seen": "2026-10-08"
            },
            {
              "what": "Claude Code guide",
              "url": "https://docs.prisminference.com/guides/claude-code",
              "seen": "2026-10-08"
            },
            {
              "what": "SDKs and clients",
              "url": "https://docs.prisminference.com/sdks",
              "seen": "2026-10-08"
            },
            {
              "what": "privacy policy, updated 9 September 2026",
              "url": "https://prisminference.com/privacy",
              "seen": "2026-10-08"
            },
            {
              "what": "terms of service, updated 9 September 2026",
              "url": "https://prisminference.com/terms",
              "seen": "2026-10-08"
            },
            {
              "what": "home page, speed and SLA claims",
              "url": "https://prisminference.com/",
              "seen": "2026-10-08"
            },
            {
              "what": "status page",
              "url": "https://status.prisminference.com/",
              "seen": "2026-10-08"
            },
            {
              "what": "status incidents feed (empty)",
              "url": "https://status.prisminference.com/api/v2/incidents.json",
              "seen": "2026-10-08"
            },
            {
              "what": "security.txt",
              "url": "https://prisminference.com/.well-known/security.txt",
              "seen": "2026-10-08"
            },
            {
              "what": "unauthenticated request, live error envelope",
              "url": "https://api.prisminference.com/v1/chat/completions",
              "seen": "2026-10-08"
            },
            {
              "what": "YC directory, team of 2",
              "url": "https://www.ycombinator.com/companies/prism",
              "seen": "2026-10-08"
            },
            {
              "what": "YC launch post, 24 September 2026",
              "url": "https://www.ycombinator.com/launches/UOa-prism-lightning-fast-inference-for-coding-agents",
              "seen": "2026-10-08"
            },
            {
              "what": "RDAP, domain registered 9 September 2026",
              "url": "https://rdap.org/domain/prisminference.com",
              "seen": "2026-10-08"
            },
            {
              "what": "Hermes provider plugin repository",
              "url": "https://github.com/prismhq/hermes-prism-provider",
              "seen": "2026-10-08"
            }
          ],
          "openQuestions": [
            "unchecked: the API was not called with a key and no account was opened, so tool use, structured output, caching and speed are as documented, not observed",
            "unchecked: the sign-up page and dashboard, which render only with JavaScript. Whether new accounts get any free credit, and what key controls and usage views the dashboard has, were not seen",
            "unchecked: the agent sign-up endpoints, which we did not call because they create an account",
            "unchecked: the throughput claim of 550 tokens a second, which is the vendor's own figure",
            "Whether Gemma 4 31B is open to every account. The docs say request access, and llms.txt and the keyless catalogue say available",
            "How prompt caching is triggered. A cache-read price is published and no caching page was found",
            "The status page shows 100% from July 2026, before the domain was registered on 9 September 2026, so what the early part of that window measures is unclear",
            "Judgement calls. Half marks for a clean incident record that is about two weeks long, half marks for agent sign-up that still needs an emailed code and browser funding, and a deduction of 2 for the home page SLA tile and the pricing page's 'no rate limits' line",
            "The scout listed both models as available and had not read the pricing page. Prices are public, and the docs mark Gemma as request access"
          ]
        },
        "negative": -2,
        "negativeNotes": [
          "2026-10-08. The home page shows a '99.99% Uptime SLA' tile, and the pricing page says 'No minimums, no rate limits' above the per-token table. The terms of 9 September 2026 say the services have no guaranteed uptime or service credit unless a separate written agreement says otherwise, and the docs describe per-key rate limits that return 429. No SLA document was found. A misleading claim, with the smallest deduction because the terms and docs state the real position (https://prisminference.com/, https://prisminference.com/pricing, https://prisminference.com/terms, https://docs.prisminference.com/rate-limits)."
        ],
        "verdict": "Three wire formats, a public OpenAPI 3.1 file, per-token prices in a keyless catalogue and zero data retention by default on every tier. The service launched on 24 September 2026 with two models, one of them by request, from a two-person company. No rate-limit numbers, SLA document, free tier or deprecation policy was found.",
        "bestFor": "Coding agents that want DeepSeek-V4.1-Flash at a low input price, with no retention, through whichever of the three wire formats the harness already speaks.",
        "strengths": [
          "One key works across OpenAI Chat Completions, OpenAI Responses and Anthropic Messages, with a public OpenAPI 3.1 file, llms.txt and Markdown docs",
          "Zero data retention is the default on every tier, and the privacy policy, terms and docs all say inputs and outputs are never used for training",
          "Every error carries a stable `code`, a `retryable` flag, a `fix` hint and a `docs_url`, and 429 carries `Retry-After` in seconds",
          "`GET /v1/models` answers without a key and returns context length, maximum output and per-token prices for each model",
          "DeepSeek-V4.1-Flash is listed with a 1M-token context and 384,000 output tokens at $0.09 in and $1.20 out per million"
        ],
        "weaknesses": [
          "Two models. The docs mark Gemma 4 31B as request access per organisation, while llms.txt and the keyless catalogue list it as available",
          "No rate-limit numbers are published. The docs say per-key limits exist, and the pricing page says 'no rate limits'",
          "The home page shows a '99.99% Uptime SLA' tile, while the terms say there is no guaranteed uptime without a separate written agreement",
          "No free tier found. Billing is prepaid, a person funds the workspace in a browser, and the agent sign-up key needs an emailed code",
          "The service launched on 24 September 2026. The changelog has one entry, and no deprecation policy, sub-processor list or certification was found"
        ],
        "agentNotes": [
          "Call `GET https://api.prisminference.com/v1/models` at start-up, with no key, and use only ids it returns. Expect 403 on `gemma-4-31b` without organisation access",
          "Use base URL `https://api.prisminference.com/v1` for OpenAI clients and `https://api.prisminference.com` with no `/v1` for Anthropic clients",
          "Read `error.retryable` before retrying, and wait for `Retry-After` on 429, which covers both key limits and model capacity",
          "Send `reasoning_effort: \"none\"` or `low` when latency matters. Reasoning is on by default and its tokens are billed as output",
          "Keep conversation state yourself and send `store: false` on Responses. `previous_response_id`, stored responses and hosted tools aren't supported"
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "C",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 60.1
          }
        ],
        "editorialScores": {
          "ergonomics": 68,
          "maintenance": 49,
          "payments": 30,
          "reliability": 65,
          "schema": 82,
          "security": 65,
          "transparency": 43
        },
        "provenanceScore": 79
      },
      "connect": {
        "install": "pip install openai   # or: npm install openai, base URL https://api.prisminference.com/v1. Anthropic SDKs use https://api.prisminference.com with no /v1",
        "http": "curl \"https://api.prisminference.com/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $PRISM_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\":\"deepseek-v4.1-flash\",\"messages\":[{\"role\":\"user\",\"content\":\"Return pong.\"}]}'",
        "claudeCode": "export ANTHROPIC_BASE_URL=https://api.prisminference.com\nexport ANTHROPIC_AUTH_TOKEN=$PRISM_API_KEY\nexport ANTHROPIC_MODEL=deepseek-v4.1-flash\nexport ANTHROPIC_SMALL_FAST_MODEL=gemma-4-31b"
      },
      "letme": {
        "capability": "https://letme.dev/inference.fast",
        "tool": "https://letme.dev/prism-inference"
      },
      "notable": [
        "Launched on Y Combinator's launch board on 24 September 2026. The company is in the Spring 2025 batch with a team of 2, and earlier launched a session-replay analyser and a short-video maker under the same name (https://www.ycombinator.com/companies/prism)",
        "prisminference.com was registered on 9 September 2026 per RDAP (https://rdap.org/domain/prisminference.com)",
        "Prism says it serves DeepSeek-V4.1-Flash at 550 tokens a second, 'up to 5.8x faster' than major providers at 95 to 247. This is a vendor claim that we did not measure (https://prisminference.com/)",
        "Zero data retention is the default for every request and tier. Inputs and outputs are not written to storage, logs, analytics or backups, and are not used for training (https://docs.prisminference.com/zero-data-retention)",
        "The Responses endpoint is stateless. `store: true`, `previous_response_id`, background jobs and hosted tools such as web search aren't supported (https://prisminference.com/llms-full.txt)",
        "The docs mark `gemma-4-31b` as request access per organisation, while llms.txt and the keyless catalogue list it as available (https://docs.prisminference.com/models)",
        "The home page shows a '99.99% Uptime SLA' tile. The terms say there is no guaranteed uptime unless a separate written agreement says otherwise (https://prisminference.com/terms)"
      ],
      "area": "models",
      "details": [
        {
          "label": "Age",
          "value": "Launched 24 September 2026. Domain registered 9 September 2026. OpenAPI file at version 0.1.0. One changelog entry, 6 October 2026"
        },
        {
          "label": "Endpoints",
          "value": "POST /v1/chat/completions, /v1/responses, /v1/messages and /v1/messages/count_tokens. GET /v1/models, /v1/models/{model} and /health"
        },
        {
          "label": "Models",
          "value": "`deepseek-v4.1-flash` (text and image input, 1M context, 384,000 output tokens) and `gemma-4-31b` (text and image, 32,768 context, 8,192 output, BF16, request access per the docs)"
        },
        {
          "label": "Free tier",
          "value": "None found. Prepaid credit, 402 when the workspace has none"
        },
        {
          "label": "Trains on API data",
          "value": "No, per the privacy policy, the terms and the docs"
        },
        {
          "label": "Data retention",
          "value": "Zero retention of inputs and outputs by default on every tier. Service metadata (key id, model, token counts, status, latency, source IP) kept for as long as reasonably needed, with no period stated"
        },
        {
          "label": "Data location",
          "value": "Prism is based in the United States and may process data there and in other countries. No region list or sub-processor list found"
        },
        {
          "label": "Rate limits",
          "value": "Per key, no numbers published. 429 with `Retry-After` in seconds, also used when a model is at capacity. Requests are capped at 4 MiB"
        },
        {
          "label": "Errors",
          "value": "OpenAI-style envelope with `code`, `retryable`, `fix` and `docs_url`. 13 status codes documented with retry advice"
        },
        {
          "label": "Structured output",
          "value": "JSON mode and strict JSON Schema on Chat Completions, `text.format` on Responses"
        },
        {
          "label": "Reasoning",
          "value": "On by default. `reasoning_effort` none, low, medium or high, and reasoning tokens are billed as output"
        },
        {
          "label": "Speed",
          "value": "550 tokens a second on DeepSeek-V4.1-Flash per the vendor's home page (547 in the launch post). Not measured by us"
        },
        {
          "label": "SDKs",
          "value": "None of its own. OpenAI and Anthropic client libraries, plus provider plugins for Hermes and OpenClaw and setup guides for Claude Code, Codex, Cursor and OpenCode"
        },
        {
          "label": "SLA",
          "value": "None published. The terms say no guaranteed uptime without a separate written agreement, and dedicated deployments list a negotiated latency SLA"
        },
        {
          "label": "Status",
          "value": "status.prisminference.com on incident.io, one component per model, no incidents posted"
        },
        {
          "label": "Company",
          "value": "Prism Technologies Inc, San Francisco, Y Combinator Spring 2025, team of 2 per the YC directory"
        }
      ],
      "models": [
        {
          "id": "deepseek-v4.1-flash",
          "name": "DeepSeek-V4.1-Flash",
          "inputPer1M": 0.09,
          "outputPer1M": 1.2,
          "contextTokens": 1000000,
          "role": "default",
          "note": "cache read $0.06. Text and image input, up to 384,000 output tokens"
        },
        {
          "id": "gemma-4-31b",
          "name": "Gemma 4 31B",
          "inputPer1M": 0.3,
          "outputPer1M": 0.4,
          "contextTokens": 32768,
          "role": "fast",
          "note": "cache read $0.15. Served at BF16. The docs say access is granted per organisation on request"
        }
      ],
      "provenance": {
        "legalEntity": "Prism Technologies Inc",
        "domain": "prisminference.com",
        "domainRegistered": "2026-09-09",
        "endpointOnVendorDomain": true,
        "terms": "https://prisminference.com/terms",
        "privacy": "https://prisminference.com/privacy",
        "statusPage": "https://status.prisminference.com",
        "changelog": "https://docs.prisminference.com/changelog",
        "securityTxt": "valid",
        "checked": "2026-10-08",
        "notes": [
          "The terms and the privacy policy, both last updated 9 September 2026, name Prism Technologies Inc. The terms are governed by California law with arbitration in San Francisco.",
          "RDAP gives a registration date of 2026-09-09 for prisminference.com, with Name.com as registrar.",
          "security.txt has a Contact line (founders@prisminference.com), an Expires date of 2027-10-06 and a Canonical line. No disclosure policy or bug bounty is named.",
          "The status page runs on incident.io with one component for each model and no component for the API or the website. Its incidents feed was empty on 8 October 2026.",
          "The YC directory lists Prism in the Spring 2025 batch, in San Francisco, with a team of 2. The home page links an X account named prism_videos, from the company's earlier video product."
        ],
        "score": 79,
        "checks": [
          {
            "check": "Legal entity named",
            "value": "Prism Technologies Inc",
            "points": 20,
            "max": 20,
            "state": "ok"
          },
          {
            "check": "Domain age",
            "value": "prisminference.com, registered 2026-09-09 (under a year)",
            "points": 0,
            "max": 15,
            "state": "no"
          },
          {
            "check": "Endpoint on the vendor's domain",
            "value": "api.prisminference.com",
            "points": 15,
            "max": 15,
            "state": "ok"
          },
          {
            "check": "Terms of service",
            "value": "read, states 7 of the 7 things a reader expects, and has 1 clause that costs points",
            "points": 8,
            "max": 10,
            "state": "part"
          },
          {
            "check": "Privacy policy",
            "value": "read, states 3 of the 8 things a reader expects",
            "points": 6.3,
            "max": 10,
            "state": "part"
          },
          {
            "check": "Status page",
            "value": "status.prisminference.com",
            "points": 10,
            "max": 10,
            "state": "ok"
          },
          {
            "check": "Changelog",
            "value": "published",
            "points": 10,
            "max": 10,
            "state": "ok"
          },
          {
            "check": "security.txt",
            "value": "valid",
            "points": 10,
            "max": 10,
            "state": "ok"
          }
        ],
        "policies": [
          {
            "kind": "terms",
            "url": "https://prisminference.com/terms",
            "state": "read",
            "readAt": "2026-10-08",
            "statedDate": "2026-09-09",
            "words": 2045,
            "points": 8,
            "max": 10,
            "expected": [
              {
                "key": "terms.date",
                "label": "Gives the date it was last updated",
                "found": true,
                "quote": "Last updated: September 9, 2026",
                "says": "Last updated 2026-09-09"
              },
              {
                "key": "terms.law",
                "label": "Names the governing law or courts",
                "found": true,
                "quote": "For a dispute that is not subject to arbitration, the parties consent to the exclusive jurisdiction of state and federal courts located in San Francisco, California.",
                "says": "Disputes go to the courts of San Francisco, California"
              },
              {
                "key": "terms.liability",
                "label": "States a limit on its liability",
                "found": true,
                "quote": "For free Services, Prism's total liability arising out of or relating to the Services or these Terms will not exceed the lesser of $500 or the amount you paid Prism during the 12 months before the event giving rise to the claim.",
                "says": "Capped at the lesser of $500 and the fees paid in the 12 months before the claim"
              },
              {
                "key": "terms.termination",
                "label": "Says how the agreement or account can be ended",
                "found": true,
                "quote": "We may reject requests or restrict, suspend, or terminate access when we reasonably believe use creates security, legal, financial, or operational risk or violates these Terms."
              },
              {
                "key": "terms.changes",
                "label": "Says how changes to the terms are announced",
                "found": true,
                "quote": "We may revise these Terms by posting an updated version.",
                "says": "Changes are posted, with no other notice named"
              },
              {
                "key": "terms.use",
                "label": "Lists what users may not do",
                "found": true,
                "quote": "You may not share credentials outside your organization, publish API keys, or bypass account or plan limits."
              },
              {
                "key": "terms.sla",
                "label": "Refers to a service level or uptime commitment",
                "found": true,
                "quote": "Unless a separate written agreement states otherwise, the Services have no guaranteed uptime, latency, throughput, model availability, support response time, or service credit."
              }
            ],
            "toKnow": [
              {
                "key": "terms.automated",
                "label": "Restricts automated access",
                "found": true,
                "quote": "You may not reverse engineer the Services to extract model weights or confidential technology, scrape or overload the Services, resell access without written authorization, use false account information, or use Inputs that you lack the right to provide.",
                "costsPoints": true
              },
              {
                "key": "terms.arbitration",
                "label": "Requires arbitration or waives class actions",
                "found": true,
                "quote": "Section 13 requires most disputes to be resolved by individual binding arbitration and includes a class-action waiver."
              }
            ],
            "notes": [
              {
                "date": "2026-10-08",
                "text": "Prism says API inputs and outputs are processed transiently, not kept after processing and not used to train or improve models.",
                "quote": "We process them transiently to fulfill each request, do not persist them after processing, and do not use them to train or improve models."
              },
              {
                "date": "2026-10-08",
                "text": "Liability for paid services is capped at the lesser of 10,000 US dollars or the amount paid in the 12 months before the claim.",
                "quote": "For paid Services, the cap is the lesser of $10,000 or the amount you paid Prism during that 12-month period."
              },
              {
                "date": "2026-10-08",
                "text": "A customer may reject the arbitration agreement by email within 30 days of first accepting the terms.",
                "quote": "You may reject this arbitration agreement by emailing founders@prisminference.com within 30 days after you first accept these Terms."
              }
            ]
          },
          {
            "kind": "privacy",
            "url": "https://prisminference.com/privacy",
            "state": "read",
            "readAt": "2026-10-08",
            "statedDate": "2026-09-09",
            "words": 1218,
            "points": 6.3,
            "max": 10,
            "expected": [
              {
                "key": "privacy.date",
                "label": "Gives the date it was last updated",
                "found": true,
                "quote": "Last updated: September 9, 2026",
                "says": "Last updated 2026-09-09"
              },
              {
                "key": "privacy.collected",
                "label": "Says what personal data is collected",
                "found": true,
                "quote": "We collect information you provide when you register or manage an account, such as your name, email address, authentication identifiers, organization membership, plan, API-key records, and support communications."
              },
              {
                "key": "privacy.retention",
                "label": "Says how long data is kept",
                "found": true,
                "quote": "We keep service metadata only for as long as reasonably needed for metering, billing, security, abuse prevention, reliability, and enforcement.",
                "says": "For as long as needed, with no period named"
              },
              {
                "key": "privacy.processors",
                "label": "Says who else receives the data",
                "found": false
              },
              {
                "key": "privacy.sale",
                "label": "Says whether personal data is sold or shared for advertising",
                "found": false
              },
              {
                "key": "privacy.rights",
                "label": "Says what rights people have over their data",
                "found": false
              },
              {
                "key": "privacy.contact",
                "label": "Gives a privacy contact",
                "found": false
              },
              {
                "key": "privacy.transfers",
                "label": "Says where data is transferred or stored",
                "found": false
              }
            ],
            "notes": [
              {
                "date": "2026-10-08",
                "text": "Content a customer includes in a support request may be kept with the support record, outside the zero retention policy for API content.",
                "quote": "If you voluntarily include an Input, Output, or other content in a support request, that copy becomes part of the support communication and may be retained with the support record."
              },
              {
                "date": "2026-10-08",
                "text": "Prism says it does not use inputs or outputs to train, fine-tune, evaluate or otherwise improve its models.",
                "quote": "We do not use Inputs or Outputs to train, fine-tune, evaluate, or otherwise improve our models."
              }
            ]
          }
        ]
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/prism-inference.json",
      "live": {
        "slug": "prism-inference",
        "probe": {
          "target": "https://api.prisminference.com/v1",
          "method": "get",
          "lastAt": "2026-10-08T19:08:56.350063654Z",
          "lastOk": true,
          "lastStatus": 200,
          "lastMs": 134,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 142,
          "p95ms24h": 447,
          "samples24h": 42,
          "samples30d": 42,
          "days": [
            {
              "date": "2026-10-08",
              "probes": 42,
              "ok": 42
            }
          ]
        },
        "vendorStatus": {
          "page": "https://status.prisminference.com",
          "indicator": "none",
          "summary": "All Systems Operational",
          "checkedAt": "2026-10-08T19:06:55.435401601Z"
        },
        "versions": [
          {
            "registry": "github",
            "name": "prismhq/hermes-prism-provider",
            "version": "v1.0.2",
            "released": "2026-09-16",
            "seenAt": "2026-10-08T16:26:20.638377705Z"
          }
        ],
        "githubStars": 0,
        "securityTxt": {
          "url": "https://prisminference.com/.well-known/security.txt",
          "state": "valid",
          "expires": "2027-10-06T00:00:00.000Z",
          "checkedAt": "2026-10-08T15:38:58.310470594Z"
        },
        "pages": [
          {
            "url": "https://docs.prisminference.com/changelog",
            "kind": "changelog",
            "status": 200,
            "checkedAt": "2026-10-08T18:19:15.357311357Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "184a0fafdb8c"
          },
          {
            "url": "https://prisminference.com/pricing",
            "kind": "pricing",
            "status": 200,
            "checkedAt": "2026-10-08T18:23:20.103073255Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "dffb5d16a3c8"
          },
          {
            "url": "https://prisminference.com/privacy",
            "kind": "privacy",
            "status": 200,
            "checkedAt": "2026-10-08T18:23:22.266572016Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "64ea7b8f1b7b"
          },
          {
            "url": "https://prisminference.com/terms",
            "kind": "terms",
            "status": 200,
            "checkedAt": "2026-10-08T18:23:24.296415884Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "27dd15c69f49"
          }
        ],
        "updatedAt": "2026-10-08T19:08:56.350063654Z"
      }
    },
    "verify": {
      "accepts": "a page on prisminference.com or one of its subdomains, or the README of github.com/prismhq/hermes-prism-provider",
      "badgeUrl": "https://www.anchorterminal.com/badges/prism-inference.svg",
      "body": {
        "slug": "prism-inference",
        "url": "the page with the badge or the link"
      },
      "docs": "https://www.anchorterminal.com/builders/#verify",
      "effect": "none, it never changes a grade, rank or review",
      "endpoint": "https://www.anchorterminal.com/api/v1/verify",
      "listingUrl": "https://www.anchorterminal.com/tools/prism-inference",
      "mcpTool": "verify_listing",
      "recheck": "weekly; two failed checks in a row and it lapses, a later pass restores it",
      "snippets": {
        "html": "\u003ca href=\"https://www.anchorterminal.com/tools/prism-inference\"\u003e\u003cimg src=\"https://www.anchorterminal.com/badges/prism-inference.svg\" alt=\"Prism Inference on Anchor Terminal\" height=\"20\"\u003e\u003c/a\u003e",
        "markdown": "[![Prism Inference on Anchor Terminal](https://www.anchorterminal.com/badges/prism-inference.svg)](https://www.anchorterminal.com/tools/prism-inference)",
        "link": "\u003ca href=\"https://www.anchorterminal.com/tools/prism-inference\"\u003ePrism Inference on Anchor Terminal\u003c/a\u003e"
      }
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/tools/prism-inference",
    "json": "https://www.anchorterminal.com/tools/prism-inference.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/tools/prism-inference.md",
    "slim": "https://www.anchorterminal.com/tools/prism-inference.min.md"
  },
  "markdown": "## Overview\n\n**Grade C · 60.1/100 · rank #363 of 629 · #8 in Model APIs \u0026 inference · not agent-ready · confidence medium**\n\n\n## Assessment\n\nThree wire formats, a public OpenAPI 3.1 file, per-token prices in a keyless catalogue and zero data retention by default on every tier. The service launched on 24 September 2026 with two models, one of them by request, from a two-person company. No rate-limit numbers, SLA document, free tier or deprecation policy was found.\n\n## Facts\n\n| Field | Value |\n| --- | --- |\n| Vendor | Prism Technologies Inc (https://prisminference.com) |\n| Kind | Model API |\n| Category | Model APIs \u0026 inference (https://www.anchorterminal.com/categories/inference) |\n| Transport | HTTP |\n| Endpoint | `https://api.prisminference.com/v1` |\n| Auth | API key · `Authorization: Bearer` with a key issued from account settings after sign-up at prisminference.com/signup. The inference endpoints also accept the key in `x-api-key`. A missing, invalid, expired or revoked key returns 401. No key scopes were found in the reviewed documentation. An agent can call `POST https://prisminference.com/api/agent-signups` with the owner's email and a username and receive a key once, but that key can't run inference until the owner supplies a six-digit emailed code, and it expires after 30 days. `GET /v1/models` needs no key. |\n| Pricing | Pay per use (from $0.09 / 1M in) · Prepaid per-token pricing with no minimum. DeepSeek-V4.1-Flash is $0.09 in, $1.20 out and $0.06 cache read per million tokens, and Gemma 4 31B is $0.30, $0.40 and $0.15 (https://prisminference.com/pricing, matched by https://api.prisminference.com/v1/models). No free tier or trial credit was found, and a workspace without credit gets 402. A person funds the workspace in a browser. Elastic endpoints, dedicated deployments and batch are sold through sales with no published price. |\n| x402 | No · No x402, MPP or L402 in llms.txt, the docs index, the OpenAPI file or the pricing page, read 2026-10-08. The docs describe prepaid credit funded by a person in the browser. |\n| Licence | Proprietary service under Prism's terms of service. The OpenAPI file declares `LicenseRef-Proprietary`. The Hermes provider plugin repository carries no licence file |\n| Source | https://github.com/prismhq/hermes-prism-provider |\n| Docs | https://docs.prisminference.com |\n| llms.txt | https://prisminference.com/llms.txt |\n| Last release | 2026-10-06 |\n| Age | Launched 24 September 2026. Domain registered 9 September 2026. OpenAPI file at version 0.1.0. One changelog entry, 6 October 2026 |\n| Endpoints | POST /v1/chat/completions, /v1/responses, /v1/messages and /v1/messages/count_tokens. GET /v1/models, /v1/models/{model} and /health |\n| Models | `deepseek-v4.1-flash` (text and image input, 1M context, 384,000 output tokens) and `gemma-4-31b` (text and image, 32,768 context, 8,192 output, BF16, request access per the docs) |\n| Free tier | None found. Prepaid credit, 402 when the workspace has none |\n| Trains on API data | No, per the privacy policy, the terms and the docs |\n| Data retention | Zero retention of inputs and outputs by default on every tier. Service metadata (key id, model, token counts, status, latency, source IP) kept for as long as reasonably needed, with no period stated |\n| Data location | Prism is based in the United States and may process data there and in other countries. No region list or sub-processor list found |\n| Rate limits | Per key, no numbers published. 429 with `Retry-After` in seconds, also used when a model is at capacity. Requests are capped at 4 MiB |\n| Errors | OpenAI-style envelope with `code`, `retryable`, `fix` and `docs_url`. 13 status codes documented with retry advice |\n| Structured output | JSON mode and strict JSON Schema on Chat Completions, `text.format` on Responses |\n| Reasoning | On by default. `reasoning_effort` none, low, medium or high, and reasoning tokens are billed as output |\n| Speed | 550 tokens a second on DeepSeek-V4.1-Flash per the vendor's home page (547 in the launch post). Not measured by us |\n| SDKs | None of its own. OpenAI and Anthropic client libraries, plus provider plugins for Hermes and OpenClaw and setup guides for Claude Code, Codex, Cursor and OpenCode |\n| SLA | None published. The terms say no guaranteed uptime without a separate written agreement, and dedicated deployments list a negotiated latency SLA |\n| Status | status.prisminference.com on incident.io, one component per model, no incidents posted |\n| Company | Prism Technologies Inc, San Francisco, Y Combinator Spring 2025, team of 2 per the YC directory |\n| Capabilities | inference.fast, inference.open-weights, inference.llm |\n| Tags | hosted, model, open-weights, fast, usage-priced, prepaid, openapi, llms-txt, openai-compatible, anthropic-compatible, zero-retention, status-page, new |\n| JSON | https://www.anchorterminal.com/api/v1/tools/prism-inference.json |\n\n## Score breakdown (methodology v0.4, October 2026 research run)\n\nAssessed 2026-10-08 from public evidence against the published checklist (https://www.anchorterminal.com/benchmark/#checklist). Confidence: medium. Performance and Task success pending (no score, not in the total); the total is Σ(score × weight) ÷ 80 over the 7 assessed categories. \"This run\" is each category's share of the 100 points.\n\n| Category | Weight | This run | Score (0–100) | Points |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% | 20 | 65 | 13.0 |\n| Performance | 10% | pending | pending | n/a |\n| Schema \u0026 documentation | 13% | 16.2 | 82 | 13.3 |\n| Agent ergonomics | 13% | 16.2 | 68 | 11.1 |\n| Security \u0026 auth | 14% | 17.5 | 65 | 11.4 |\n| Payments \u0026 pricing | 10% | 12.5 | 30 | 3.8 |\n| Task success | 10% | pending | pending | n/a |\n| Maintenance \u0026 community | 7% | 8.8 | 49 | 4.3 |\n| Transparency \u0026 trust (editorial 43, provenance 79) | 7% | 8.8 | 61 | 5.3 |\n| Negative events | up to −15 | up to −15 | 2026-10-08. The home page shows a '99.99% Uptime SLA' tile, and the pricing page says 'No minimums, no rate limits' above the per-token table. The terms of 9 September 2026 say the services have no guaranteed uptime or service credit unless a separate written agreement says otherwise, and the docs describe per-key rate limits that return 429. No SLA document was found. A misleading claim, with the smallest deduction because the terms and docs state the real position (https://prisminference.com/, https://prisminference.com/pricing, https://prisminference.com/terms, https://docs.prisminference.com/rate-limits).  | -2 |\n| **Total** | | | | **60.1 → C** |\n\n### Why each score\n\n- Reliability 65: Hosted reading. Status page at status.prisminference.com on incident.io with a component for each of the two models (20). The incidents feed is empty and the page shows 100% for both models, but the service launched on 24 September 2026 and the domain dates from 9 September, so the record covers about two weeks of a 90-day window. We give half of the 30 for a clean record and say so as a departure from the checklist (15). The docs say per-key limits exist and publish no numbers, only a 4 MiB request cap (5). 429 carries `Retry-After` in seconds, errors carry a `retryable` flag, and the docs give backoff with jitter and rules for retrying interrupted streams (15). No published SLA. The terms say no guaranteed uptime without a separate written agreement, although the home page shows a '99.99% Uptime SLA' tile (0). The serverless API isn't labelled beta or preview, though the OpenAPI file is at version 0.1.0 (10).\n- Performance: Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes.\n- Schema \u0026 documentation 82: A public OpenAPI 3.1 file covering ten paths, including chat completions, responses, messages and models (25). llms.txt on both the site and the docs, a fuller llms-full.txt, and every docs page served as Markdown (10). Reference pages state what each endpoint is for and what isn't supported, such as stored responses and hosted tools, and the models page has a 'use when' column (15 of 20). `model` and `messages` are required, with ranges on `temperature` and `top_p` and an enum for reasoning effort, but `tool_choice` is untyped and `response_format` and the request body allow any extra properties (10 of 15). Examples in curl, TypeScript and Python, and an errors page with 13 status codes and five named codes with fixes (14 of 15). Paths are versioned under /v1 and the changelog is dated, with a single entry on 6 October 2026 (8 of 15).\n- Agent ergonomics 68: Model reading of the checklist (tool use, structured output, caching, context, batch, SDKs, errors). OpenAI function tools, Anthropic tool-use blocks, `tool_choice` and `parallel_tool_calls` on Responses are documented (18 of 20). JSON mode and strict JSON Schema on Chat Completions, and `text.format` on Responses (15). A cache-read price is published for both models, $0.06 against $0.09 on DeepSeek-V4.1-Flash, but no page explains how caching is triggered, and the Claude Code guide says prompt-cache hints are accepted and ignored (5 of 15). 1M-token context and 384,000 output tokens on DeepSeek-V4.1-Flash, 32,768 on Gemma 4 31B (15). Batch appears on the pricing page as a sales-led tier with no API in the OpenAPI file (0). No SDK of its own. OpenAI and Anthropic clients are used with a base URL change (0). Every error carries `code`, `retryable`, `fix` and `docs_url`, which we confirmed on a live 401 (15).\n- Security \u0026 auth 65: Model reading. Bearer keys issued from account settings, with 401 for revoked or expired keys and a 30-day expiry on keys from agent sign-up. No scopes or per-key limits were found in the reviewed documentation (20 of 30). The privacy policy, the terms and the docs all say inputs and outputs are never used to train, fine-tune or evaluate models (20). Zero data retention is the default on every tier with no setting to turn on (15). The docs mention usage in the dashboard, which we did not see, and no audit log was found (5 of 15). security.txt is valid until 6 October 2027 with a contact address. No disclosure policy, bug bounty, SOC 2 or ISO 27001 was found (5 of 20).\n- Payments \u0026 pricing 30: No machine payment protocol (0). Per-token prices are on the pricing page and in the keyless `/v1/models` response, with no login (20). No free tier or trial credit was found. Billing is prepaid and a workspace without credit gets 402 (0). An agent can request a key from `POST /api/agent-signups`, but the key can't run inference until the owner supplies an emailed code, and a person funds the workspace in a browser, so half credit (10 of 20).\n- Task success: Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored.\n- Maintenance \u0026 community 49: Model reading. Gemma 4 31B was added to the catalogue on 6 October 2026 (30). No notice period is stated. The terms promise advance notice of a material breaking change 'when reasonably practicable', and nothing has been retired yet (3 of 12). No shutdowns in the service's two weeks of public life (8 of 8). A dated changelog with one entry, a Discord server and a support booking link, none of which we tested for replies (8 of 15). No SDK repository to sample. The Hermes provider plugin on GitHub has three commits, all on 15 September 2026 (0 of 10). No official SDKs (0 of 15) and no packages to assess (0 of 10).\n- Transparency \u0026 trust 61: Closed service with clear terms that name Prism Technologies Inc (15). The privacy policy, the terms and the zero-retention page agree on what is and isn't kept. Service metadata is kept 'only for as long as reasonably needed' with no period, and no DPA is published (20 of 30). No deprecation policy or dated notices, only the terms' advance notice 'when reasonably practicable' (3 of 20). The privacy policy says Prism is based in the United States and names vendors by category only, with no sub-processor list or regions (5 of 20).\n\nFix list for a coding agent, everything this grade says the listing lacks, the biggest gain first (20 items): https://www.anchorterminal.com/fixes/prism-inference.md (JSON https://www.anchorterminal.com/fixes/prism-inference.json)\n\n### What we couldn't check\n\n- unchecked: the API was not called with a key and no account was opened, so tool use, structured output, caching and speed are as documented, not observed\n- unchecked: the sign-up page and dashboard, which render only with JavaScript. Whether new accounts get any free credit, and what key controls and usage views the dashboard has, were not seen\n- unchecked: the agent sign-up endpoints, which we did not call because they create an account\n- unchecked: the throughput claim of 550 tokens a second, which is the vendor's own figure\n- Whether Gemma 4 31B is open to every account. The docs say request access, and llms.txt and the keyless catalogue say available\n- How prompt caching is triggered. A cache-read price is published and no caching page was found\n- The status page shows 100% from July 2026, before the domain was registered on 9 September 2026, so what the early part of that window measures is unclear\n- Judgement calls. Half marks for a clean incident record that is about two weeks long, half marks for agent sign-up that still needs an emailed code and browser funding, and a deduction of 2 for the home page SLA tile and the pricing page's 'no rate limits' line\n- The scout listed both models as available and had not read the pricing page. Prices are public, and the docs mark Gemma as request access\n\n### Sources\n\n- llms.txt, endpoints, models and agent sign-up steps: \u003chttps://prisminference.com/llms.txt\u003e (seen 2026-10-08)\n- full agent context, billing and compatibility limits: \u003chttps://prisminference.com/llms-full.txt\u003e (seen 2026-10-08)\n- docs index: \u003chttps://docs.prisminference.com/llms.txt\u003e (seen 2026-10-08)\n- OpenAPI 3.1 file, version 0.1.0: \u003chttps://docs.prisminference.com/openapi.yaml\u003e (seen 2026-10-08)\n- keyless model catalogue with prices: \u003chttps://api.prisminference.com/v1/models\u003e (seen 2026-10-08)\n- pricing page: \u003chttps://prisminference.com/pricing\u003e (seen 2026-10-08)\n- machine-readable pricing tiers: \u003chttps://prisminference.com/pricing.json\u003e (seen 2026-10-08)\n- models page, Gemma marked request access: \u003chttps://docs.prisminference.com/models\u003e (seen 2026-10-08)\n- rate limits: \u003chttps://docs.prisminference.com/rate-limits\u003e (seen 2026-10-08)\n- errors and retry policy: \u003chttps://docs.prisminference.com/errors\u003e (seen 2026-10-08)\n- authentication: \u003chttps://docs.prisminference.com/authentication\u003e (seen 2026-10-08)\n- agent sign-up: \u003chttps://docs.prisminference.com/guides/agent-signup\u003e (seen 2026-10-08)\n- zero data retention: \u003chttps://docs.prisminference.com/zero-data-retention\u003e (seen 2026-10-08)\n- changelog: \u003chttps://docs.prisminference.com/changelog\u003e (seen 2026-10-08)\n- Chat Completions reference: \u003chttps://docs.prisminference.com/api-reference/chat-completions\u003e (seen 2026-10-08)\n- Responses reference: \u003chttps://docs.prisminference.com/api-reference/responses\u003e (seen 2026-10-08)\n- Anthropic Messages reference: \u003chttps://docs.prisminference.com/api-reference/messages\u003e (seen 2026-10-08)\n- structured outputs guide: \u003chttps://docs.prisminference.com/guides/structured-outputs\u003e (seen 2026-10-08)\n- Claude Code guide: \u003chttps://docs.prisminference.com/guides/claude-code\u003e (seen 2026-10-08)\n- SDKs and clients: \u003chttps://docs.prisminference.com/sdks\u003e (seen 2026-10-08)\n- privacy policy, updated 9 September 2026: \u003chttps://prisminference.com/privacy\u003e (seen 2026-10-08)\n- terms of service, updated 9 September 2026: \u003chttps://prisminference.com/terms\u003e (seen 2026-10-08)\n- home page, speed and SLA claims: \u003chttps://prisminference.com/\u003e (seen 2026-10-08)\n- status page: \u003chttps://status.prisminference.com/\u003e (seen 2026-10-08)\n- status incidents feed (empty): \u003chttps://status.prisminference.com/api/v2/incidents.json\u003e (seen 2026-10-08)\n- security.txt: \u003chttps://prisminference.com/.well-known/security.txt\u003e (seen 2026-10-08)\n- unauthenticated request, live error envelope: \u003chttps://api.prisminference.com/v1/chat/completions\u003e (seen 2026-10-08)\n- YC directory, team of 2: \u003chttps://www.ycombinator.com/companies/prism\u003e (seen 2026-10-08)\n- YC launch post, 24 September 2026: \u003chttps://www.ycombinator.com/launches/UOa-prism-lightning-fast-inference-for-coding-agents\u003e (seen 2026-10-08)\n- RDAP, domain registered 9 September 2026: \u003chttps://rdap.org/domain/prisminference.com\u003e (seen 2026-10-08)\n- Hermes provider plugin repository: \u003chttps://github.com/prismhq/hermes-prism-provider\u003e (seen 2026-10-08)\n\n## Who's behind it (provenance 79/100, checked 2026-10-08)\n\n| Check | Finding | Points |\n| --- | --- | --- |\n| Legal entity named | Prism Technologies Inc | 20/20 |\n| Domain age | prisminference.com, registered 2026-09-09 (under a year) | 0/15 |\n| Endpoint on the vendor's domain | api.prisminference.com | 15/15 |\n| Terms of service | read, states 7 of the 7 things a reader expects, and has 1 clause that costs points | 8/10 |\n| Privacy policy | read, states 3 of the 8 things a reader expects | 6.3/10 |\n| Status page | status.prisminference.com | 10/10 |\n| Changelog | published | 10/10 |\n| security.txt | valid | 10/10 |\n\nThe terms and the privacy policy, both last updated 9 September 2026, name Prism Technologies Inc. The terms are governed by California law with arbitration in San Francisco.\n\nRDAP gives a registration date of 2026-09-09 for prisminference.com, with Name.com as registrar.\n\nsecurity.txt has a Contact line (founders@prisminference.com), an Expires date of 2027-10-06 and a Canonical line. No disclosure policy or bug bounty is named.\n\nThe status page runs on incident.io with one component for each model and no component for the API or the website. Its incidents feed was empty on 8 October 2026.\n\nThe YC directory lists Prism in the Spring 2025 batch, in San Francisco, with a team of 2. The home page links an X account named prism_videos, from the company's earlier video product.\n\n### Terms and privacy, as read\n\nA reading by a fixed set of rules, each answered with the vendor's own sentence. Not legal advice.\n\n**Terms of service** (https://prisminference.com/terms), read 2026-10-08, dated 2026-09-09, states 7 of the 7 things a reader expects.\n\n- To know. Restricts automated access (costs points). \"You may not reverse engineer the Services to extract model weights or confidential technology, scrape or overload the Services, resell access without written authorization, use false account information, or use Inputs that you lack the right to provide.\"\n- To know. Requires arbitration or waives class actions. \"Section 13 requires most disputes to be resolved by individual binding arbitration and includes a class-action waiver.\"\n- Gives the date it was last updated. Last updated 2026-09-09.\n- Names the governing law or courts. Disputes go to the courts of San Francisco, California.\n- States a limit on its liability. Capped at the lesser of $500 and the fees paid in the 12 months before the claim.\n- Says how changes to the terms are announced. Changes are posted, with no other notice named.\n- Also in the text (2026-10-08). Prism says API inputs and outputs are processed transiently, not kept after processing and not used to train or improve models. \"We process them transiently to fulfill each request, do not persist them after processing, and do not use them to train or improve models.\"\n- Also in the text (2026-10-08). Liability for paid services is capped at the lesser of 10,000 US dollars or the amount paid in the 12 months before the claim. \"For paid Services, the cap is the lesser of $10,000 or the amount you paid Prism during that 12-month period.\"\n- Also in the text (2026-10-08). A customer may reject the arbitration agreement by email within 30 days of first accepting the terms. \"You may reject this arbitration agreement by emailing founders@prisminference.com within 30 days after you first accept these Terms.\"\n\n**Privacy policy** (https://prisminference.com/privacy), read 2026-10-08, dated 2026-09-09, states 3 of the 8 things a reader expects.\n\n- Gives the date it was last updated. Last updated 2026-09-09.\n- Says how long data is kept. For as long as needed, with no period named.\n- Not found in the text. Says who else receives the data.\n- Not found in the text. Says whether personal data is sold or shared for advertising.\n- Not found in the text. Says what rights people have over their data.\n- Not found in the text. Gives a privacy contact.\n- Not found in the text. Says where data is transferred or stored.\n- Also in the text (2026-10-08). Content a customer includes in a support request may be kept with the support record, outside the zero retention policy for API content. \"If you voluntarily include an Input, Output, or other content in a support request, that copy becomes part of the support communication and may be retained with the support record.\"\n- Also in the text (2026-10-08). Prism says it does not use inputs or outputs to train, fine-tune, evaluate or otherwise improve its models. \"We do not use Inputs or Outputs to train, fine-tune, evaluate, or otherwise improve our models.\"\n\n## Live (updated 2026-10-08 19:08 UTC)\n\n- Right now: up, HTTP 200, 134 ms, checked 2026-10-08 19:08 UTC (get on `https://api.prisminference.com/v1`)\n- Uptime 24h 100.0% (42 probes) · 30 days 100.0% (42 probes) · p50 142 ms · p95 447 ms\n- Vendor status page: none, All Systems Operational\n- github `prismhq/hermes-prism-provider` v1.0.2, released 2026-09-16\n- security.txt: valid, expires 2027-10-06T00:00:00.000Z\n- Watching changelog \u003chttps://docs.prisminference.com/changelog\u003e\n- Watching pricing \u003chttps://prisminference.com/pricing\u003e\n- Watching privacy \u003chttps://prisminference.com/privacy\u003e\n- Watching terms \u003chttps://prisminference.com/terms\u003e\n- Always current: https://www.anchorterminal.com/api/v1/live/prism-inference.json\n\n## Probe metrics\n\nNot measured yet. Our benchmark probes haven't run, so there's no availability, latency or error rate from a run and Performance is pending. Live uptime, where we poll the endpoint, is under Live and doesn't change the score.\n\n## Models and prices (per 1M tokens)\n\n| Model | Input | Output | Context | Role | Supports |\n| --- | --- | --- | --- | --- | --- |\n| `deepseek-v4.1-flash` DeepSeek-V4.1-Flash | $0.09 | $1.20 | 1M | default (cache read $0.06. Text and image input, up to 384,000 output tokens) | not checked |\n| `gemma-4-31b` Gemma 4 31B | $0.30 | $0.40 | 32k | fast (cache read $0.15. Served at BF16. The docs say access is granted per organisation on request) | not checked |\n Rate limits depend on your account tier: https://docs.prisminference.com/rate-limits\n\n## Strengths\n\n- One key works across OpenAI Chat Completions, OpenAI Responses and Anthropic Messages, with a public OpenAPI 3.1 file, llms.txt and Markdown docs\n- Zero data retention is the default on every tier, and the privacy policy, terms and docs all say inputs and outputs are never used for training\n- Every error carries a stable `code`, a `retryable` flag, a `fix` hint and a `docs_url`, and 429 carries `Retry-After` in seconds\n- `GET /v1/models` answers without a key and returns context length, maximum output and per-token prices for each model\n- DeepSeek-V4.1-Flash is listed with a 1M-token context and 384,000 output tokens at $0.09 in and $1.20 out per million\n\n## Weaknesses\n\n- Two models. The docs mark Gemma 4 31B as request access per organisation, while llms.txt and the keyless catalogue list it as available\n- No rate-limit numbers are published. The docs say per-key limits exist, and the pricing page says 'no rate limits'\n- The home page shows a '99.99% Uptime SLA' tile, while the terms say there is no guaranteed uptime without a separate written agreement\n- No free tier found. Billing is prepaid, a person funds the workspace in a browser, and the agent sign-up key needs an emailed code\n- The service launched on 24 September 2026. The changelog has one entry, and no deprecation policy, sub-processor list or certification was found\n\n## Before you call it (notes for agents)\n\n1. Call `GET https://api.prisminference.com/v1/models` at start-up, with no key, and use only ids it returns. Expect 403 on `gemma-4-31b` without organisation access\n2. Use base URL `https://api.prisminference.com/v1` for OpenAI clients and `https://api.prisminference.com` with no `/v1` for Anthropic clients\n3. Read `error.retryable` before retrying, and wait for `Retry-After` on 429, which covers both key limits and model capacity\n4. Send `reasoning_effort: \"none\"` or `low` when latency matters. Reasoning is on by default and its tokens are billed as output\n5. Keep conversation state yourself and send `store: false` on Responses. `previous_response_id`, stored responses and hosted tools aren't supported\n\n## Connect\n\nInstall:\n\n```bash\npip install openai   # or: npm install openai, base URL https://api.prisminference.com/v1. Anthropic SDKs use https://api.prisminference.com with no /v1\n```\n\nFirst request:\n\n```bash\ncurl \"https://api.prisminference.com/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $PRISM_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\":\"deepseek-v4.1-flash\",\"messages\":[{\"role\":\"user\",\"content\":\"Return pong.\"}]}'\n```\n\nClaude Code:\n\n```bash\nexport ANTHROPIC_BASE_URL=https://api.prisminference.com\nexport ANTHROPIC_AUTH_TOKEN=$PRISM_API_KEY\nexport ANTHROPIC_MODEL=deepseek-v4.1-flash\nexport ANTHROPIC_SMALL_FAST_MODEL=gemma-4-31b\n```\n\n## Similar tools\n\nRanked by shared capabilities, then score. Same-category tools with no shared capability key are listed last.\n\n| Tool | Grade | Score | Rank | Shared capabilities | x402 | Markdown |\n| --- | --- | --- | --- | --- | --- | --- |\n| GroqCloud | BB | 75.6 | 39 | inference.llm, inference.fast, inference.open-weights | no | https://www.anchorterminal.com/tools/groq.md |\n| Mistral AI API | BB | 71.2 | 112 | inference.llm, inference.open-weights | no | https://www.anchorterminal.com/tools/mistral-api.md |\n| Docker Model Runner | C | 57.1 | 428 | inference.open-weights, inference.llm | no | https://www.anchorterminal.com/tools/docker-model-runner.md |\n| Ollama | C | 56.3 | 436 | inference.open-weights, inference.llm | no | https://www.anchorterminal.com/tools/ollama.md |\n| DeepSeek API | D | 46.8 | 555 | inference.llm, inference.open-weights | no | https://www.anchorterminal.com/tools/deepseek-api.md |\n| OpenAI API | A | 83.3 | 4 | inference.llm | no | https://www.anchorterminal.com/tools/openai-api.md |\n\n## Panel reviews (0)\n\nReviewed by the Anchor panel (https://www.anchorterminal.com/reviewers/index.md): .\n\nDesk reviews, written from public documentation, pricing, terms, source and status history on 1 October 2026. No calls made. For a desk review, the outcome says whether the reviewer's questions could be answered from public material: success, partial or failure. How reviews work: https://www.anchorterminal.com/reviews/how-it-works.md\n\n## Notable\n\n- Launched on Y Combinator's launch board on 24 September 2026. The company is in the Spring 2025 batch with a team of 2, and earlier launched a session-replay analyser and a short-video maker under the same name (source: \u003chttps://www.ycombinator.com/companies/prism\u003e)\n- prisminference.com was registered on 9 September 2026 per RDAP (source: \u003chttps://rdap.org/domain/prisminference.com\u003e)\n- Prism says it serves DeepSeek-V4.1-Flash at 550 tokens a second, 'up to 5.8x faster' than major providers at 95 to 247. This is a vendor claim that we did not measure (source: \u003chttps://prisminference.com/\u003e)\n- Zero data retention is the default for every request and tier. Inputs and outputs are not written to storage, logs, analytics or backups, and are not used for training (source: \u003chttps://docs.prisminference.com/zero-data-retention\u003e)\n- The Responses endpoint is stateless. `store: true`, `previous_response_id`, background jobs and hosted tools such as web search aren't supported (source: \u003chttps://prisminference.com/llms-full.txt\u003e)\n- The docs mark `gemma-4-31b` as request access per organisation, while llms.txt and the keyless catalogue list it as available (source: \u003chttps://docs.prisminference.com/models\u003e)\n- The home page shows a '99.99% Uptime SLA' tile. The terms say there is no guaranteed uptime unless a separate written agreement says otherwise (source: \u003chttps://prisminference.com/terms\u003e)\n\n## Compare\n\n- [Claude API vs Prism Inference](https://www.anchorterminal.com/compare/anthropic-api-vs-prism-inference.md): BB 77.3 vs C 60.1\n- [Antseed vs Prism Inference](https://www.anchorterminal.com/compare/antseed-vs-prism-inference.md): C 55.1 vs C 60.1\n- [BlockRun.AI vs Prism Inference](https://www.anchorterminal.com/compare/blockrun-ai-vs-prism-inference.md): BB 72.4 vs C 60.1\n- [DeepSeek API vs Prism Inference](https://www.anchorterminal.com/compare/deepseek-api-vs-prism-inference.md): D 46.8 vs C 60.1\n- [Gemini Developer API vs Prism Inference](https://www.anchorterminal.com/compare/gemini-api-vs-prism-inference.md): B 65.5 vs C 60.1\n- [GroqCloud vs Prism Inference](https://www.anchorterminal.com/compare/groq-vs-prism-inference.md): BB 75.6 vs C 60.1\n- [Mistral AI API vs Prism Inference](https://www.anchorterminal.com/compare/mistral-api-vs-prism-inference.md): BB 71.2 vs C 60.1\n- [OpenAI API vs Prism Inference](https://www.anchorterminal.com/compare/openai-api-vs-prism-inference.md): A 83.3 vs C 60.1\n- [OpenRouter vs Prism Inference](https://www.anchorterminal.com/compare/openrouter-vs-prism-inference.md): B 68.5 vs C 60.1\n\n## Verify this listing\n\nFor the vendor. The badge or a plain link to this page verifies the listing, from a page on prisminference.com or one of its subdomains, or the README of github.com/prismhq/hermes-prism-provider. It shows the listing is the vendor's and that the vendor knows it's here, and it never changes a grade, rank or review. The vendor sends the page's address to `POST https://www.anchorterminal.com/api/v1/verify` as `{\"slug\": \"prism-inference\", \"url\": \"…\"}`, or calls the `verify_listing` tool at https://www.anchorterminal.com/mcp. We fetch the page once, then again every week; two failed checks in a row and the verification lapses, and a later pass restores it. What we check: https://www.anchorterminal.com/builders/index.md#verify\n\nHTML badge:\n\n```html\n\u003ca href=\"https://www.anchorterminal.com/tools/prism-inference\"\u003e\u003cimg src=\"https://www.anchorterminal.com/badges/prism-inference.svg\" alt=\"Prism Inference on Anchor Terminal\" height=\"20\"\u003e\u003c/a\u003e\n```\n\nMarkdown badge, for a README:\n\n```markdown\n[![Prism Inference on Anchor Terminal](https://www.anchorterminal.com/badges/prism-inference.svg)](https://www.anchorterminal.com/tools/prism-inference)\n```\n\nPlain link:\n\n```html\n\u003ca href=\"https://www.anchorterminal.com/tools/prism-inference\"\u003ePrism Inference on Anchor Terminal\u003c/a\u003e\n```\n\n## Share this listing\n\nFor the vendor. Sharing assets for social media, two PNGs of 1200 × 630 that say Prism Inference is listed on Anchor Terminal, with the vendor's logo and this page's address and no grade or score.\n\n- Dark: https://www.anchorterminal.com/assets/share/prism-inference-dark.png\n- Light: https://www.anchorterminal.com/assets/share/prism-inference-light.png\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-08",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Terminal",
        "url": "https://www.anchorterminal.com/tools/"
      },
      {
        "name": "Model APIs \u0026 inference",
        "url": "https://www.anchorterminal.com/categories/inference"
      },
      {
        "name": "Prism Inference",
        "url": ""
      }
    ],
    "description": "Prism is a hosted inference API from Prism Technologies Inc for open-weight models, aimed at coding agents. It accepts OpenAI Chat Completions, OpenAI Responses and Anthropic Messages requests at api.prisminference.com. It launched on 24 September 2026.",
    "facts": [
      "rank #363 of 629",
      "API key auth",
      "0 desk reviews"
    ],
    "h1": "Prism Inference",
    "image": "https://www.anchorterminal.com/assets/og/tools-prism-inference.png",
    "path": "/tools/prism-inference",
    "published": "2026-10-01",
    "section": "tools",
    "title": "Prism Inference review for AI agents, grade C (60.1/100)",
    "toc": null,
    "updated": "2026-10-08",
    "url": "https://www.anchorterminal.com/tools/prism-inference"
  },
  "tokens": {
    "markdown": 8050,
    "slim": 1830
  },
  "version": 1
}
