{
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-08",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "tool": {
    "slug": "prism-inference",
    "name": "Prism Inference",
    "vendor": "Prism Technologies Inc",
    "vendorUrl": "https://prisminference.com",
    "kind": "model",
    "category": "inference",
    "summary": "Prism is a hosted inference API from Prism Technologies Inc for open-weight models, aimed at coding agents. It accepts OpenAI Chat Completions, OpenAI Responses and Anthropic Messages requests at api.prisminference.com. It launched on 24 September 2026.",
    "url": "https://www.anchorterminal.com/tools/prism-inference",
    "markdownUrl": "https://www.anchorterminal.com/tools/prism-inference.md",
    "slimMarkdownUrl": "https://www.anchorterminal.com/tools/prism-inference.min.md",
    "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/prism-inference.json",
    "repo": "https://github.com/prismhq/hermes-prism-provider",
    "license": "Proprietary service under Prism's terms of service. The OpenAPI file declares `LicenseRef-Proprietary`. The Hermes provider plugin repository carries no licence file",
    "transports": [
      "http"
    ],
    "remoteUrl": "https://api.prisminference.com/v1",
    "packages": [],
    "auth": "api-key",
    "authNotes": "`Authorization: Bearer` with a key issued from account settings after sign-up at prisminference.com/signup. The inference endpoints also accept the key in `x-api-key`. A missing, invalid, expired or revoked key returns 401. No key scopes were found in the reviewed documentation. An agent can call `POST https://prisminference.com/api/agent-signups` with the owner's email and a username and receive a key once, but that key can't run inference until the owner supplies a six-digit emailed code, and it expires after 30 days. `GET /v1/models` needs no key.",
    "pricing": "usage",
    "pricingNotes": "Prepaid per-token pricing with no minimum. DeepSeek-V4.1-Flash is $0.09 in, $1.20 out and $0.06 cache read per million tokens, and Gemma 4 31B is $0.30, $0.40 and $0.15 (https://prisminference.com/pricing, matched by https://api.prisminference.com/v1/models). No free tier or trial credit was found, and a workspace without credit gets 402. A person funds the workspace in a browser. Elastic endpoints, dedicated deployments and batch are sold through sales with no published price.",
    "priceSummary": "from $0.09 / 1M in",
    "where": "hosted",
    "x402": {
      "level": "no",
      "evidence": "No x402, MPP or L402 in llms.txt, the docs index, the OpenAPI file or the pricing page, read 2026-10-08. The docs describe prepaid credit funded by a person in the browser.",
      "endpoints": []
    },
    "toolCount": null,
    "popularity": {
      "githubStars": null,
      "npmWeekly": null,
      "pypiWeekly": null,
      "asOf": "2026-10-08"
    },
    "docsUrl": "https://docs.prisminference.com",
    "rateLimitsUrl": "https://docs.prisminference.com/rate-limits",
    "llmsTxt": "https://prisminference.com/llms.txt",
    "openapi": "https://docs.prisminference.com/openapi.yaml",
    "capabilities": [
      "inference.fast",
      "inference.open-weights",
      "inference.llm"
    ],
    "tags": [
      "hosted",
      "model",
      "open-weights",
      "fast",
      "usage-priced",
      "prepaid",
      "openapi",
      "llms-txt",
      "openai-compatible",
      "anthropic-compatible",
      "zero-retention",
      "status-page",
      "new"
    ],
    "lastRelease": "2026-10-06",
    "graded": true,
    "anchor": {
      "graded": true,
      "score": 60.1,
      "grade": "C",
      "agentReady": false,
      "rank": 363,
      "ranked": true,
      "rankOf": 629,
      "categoryRank": 8,
      "methodology": "0.4",
      "run": "2026-10-01",
      "scores": {
        "ergonomics": 68,
        "maintenance": 49,
        "payments": 30,
        "reliability": 65,
        "schema": 82,
        "security": 65,
        "transparency": 61
      },
      "pending": [
        "performance",
        "tasks"
      ],
      "breakdown": [
        {
          "key": "reliability",
          "name": "Reliability",
          "weight": 16,
          "effectiveWeight": 20,
          "score": 65,
          "points": 13,
          "reason": "Hosted reading. Status page at status.prisminference.com on incident.io with a component for each of the two models (20). The incidents feed is empty and the page shows 100% for both models, but the service launched on 24 September 2026 and the domain dates from 9 September, so the record covers about two weeks of a 90-day window. We give half of the 30 for a clean record and say so as a departure from the checklist (15). The docs say per-key limits exist and publish no numbers, only a 4 MiB request cap (5). 429 carries `Retry-After` in seconds, errors carry a `retryable` flag, and the docs give backoff with jitter and rules for retrying interrupted streams (15). No published SLA. The terms say no guaranteed uptime without a separate written agreement, although the home page shows a '99.99% Uptime SLA' tile (0). The serverless API isn't labelled beta or preview, though the OpenAPI file is at version 0.1.0 (10)."
        },
        {
          "key": "performance",
          "name": "Performance",
          "weight": 10,
          "effectiveWeight": 0,
          "pending": true,
          "points": 0,
          "reason": "Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes."
        },
        {
          "key": "schema",
          "name": "Schema \u0026 documentation",
          "weight": 13,
          "effectiveWeight": 16.25,
          "score": 82,
          "points": 13.33,
          "reason": "A public OpenAPI 3.1 file covering ten paths, including chat completions, responses, messages and models (25). llms.txt on both the site and the docs, a fuller llms-full.txt, and every docs page served as Markdown (10). Reference pages state what each endpoint is for and what isn't supported, such as stored responses and hosted tools, and the models page has a 'use when' column (15 of 20). `model` and `messages` are required, with ranges on `temperature` and `top_p` and an enum for reasoning effort, but `tool_choice` is untyped and `response_format` and the request body allow any extra properties (10 of 15). Examples in curl, TypeScript and Python, and an errors page with 13 status codes and five named codes with fixes (14 of 15). Paths are versioned under /v1 and the changelog is dated, with a single entry on 6 October 2026 (8 of 15)."
        },
        {
          "key": "ergonomics",
          "name": "Agent ergonomics",
          "weight": 13,
          "effectiveWeight": 16.25,
          "score": 68,
          "points": 11.05,
          "reason": "Model reading of the checklist (tool use, structured output, caching, context, batch, SDKs, errors). OpenAI function tools, Anthropic tool-use blocks, `tool_choice` and `parallel_tool_calls` on Responses are documented (18 of 20). JSON mode and strict JSON Schema on Chat Completions, and `text.format` on Responses (15). A cache-read price is published for both models, $0.06 against $0.09 on DeepSeek-V4.1-Flash, but no page explains how caching is triggered, and the Claude Code guide says prompt-cache hints are accepted and ignored (5 of 15). 1M-token context and 384,000 output tokens on DeepSeek-V4.1-Flash, 32,768 on Gemma 4 31B (15). Batch appears on the pricing page as a sales-led tier with no API in the OpenAPI file (0). No SDK of its own. OpenAI and Anthropic clients are used with a base URL change (0). Every error carries `code`, `retryable`, `fix` and `docs_url`, which we confirmed on a live 401 (15)."
        },
        {
          "key": "security",
          "name": "Security \u0026 auth",
          "weight": 14,
          "effectiveWeight": 17.5,
          "score": 65,
          "points": 11.38,
          "reason": "Model reading. Bearer keys issued from account settings, with 401 for revoked or expired keys and a 30-day expiry on keys from agent sign-up. No scopes or per-key limits were found in the reviewed documentation (20 of 30). The privacy policy, the terms and the docs all say inputs and outputs are never used to train, fine-tune or evaluate models (20). Zero data retention is the default on every tier with no setting to turn on (15). The docs mention usage in the dashboard, which we did not see, and no audit log was found (5 of 15). security.txt is valid until 6 October 2027 with a contact address. No disclosure policy, bug bounty, SOC 2 or ISO 27001 was found (5 of 20)."
        },
        {
          "key": "payments",
          "name": "Payments \u0026 pricing",
          "weight": 10,
          "effectiveWeight": 12.5,
          "score": 30,
          "points": 3.75,
          "reason": "No machine payment protocol (0). Per-token prices are on the pricing page and in the keyless `/v1/models` response, with no login (20). No free tier or trial credit was found. Billing is prepaid and a workspace without credit gets 402 (0). An agent can request a key from `POST /api/agent-signups`, but the key can't run inference until the owner supplies an emailed code, and a person funds the workspace in a browser, so half credit (10 of 20)."
        },
        {
          "key": "tasks",
          "name": "Task success",
          "weight": 10,
          "effectiveWeight": 0,
          "pending": true,
          "points": 0,
          "reason": "Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored."
        },
        {
          "key": "maintenance",
          "name": "Maintenance \u0026 community",
          "weight": 7,
          "effectiveWeight": 8.75,
          "score": 49,
          "points": 4.29,
          "reason": "Model reading. Gemma 4 31B was added to the catalogue on 6 October 2026 (30). No notice period is stated. The terms promise advance notice of a material breaking change 'when reasonably practicable', and nothing has been retired yet (3 of 12). No shutdowns in the service's two weeks of public life (8 of 8). A dated changelog with one entry, a Discord server and a support booking link, none of which we tested for replies (8 of 15). No SDK repository to sample. The Hermes provider plugin on GitHub has three commits, all on 15 September 2026 (0 of 10). No official SDKs (0 of 15) and no packages to assess (0 of 10)."
        },
        {
          "key": "transparency",
          "name": "Transparency \u0026 trust",
          "weight": 7,
          "effectiveWeight": 8.75,
          "score": 61,
          "points": 5.34,
          "note": "editorial 43, provenance 79",
          "reason": "Closed service with clear terms that name Prism Technologies Inc (15). The privacy policy, the terms and the zero-retention page agree on what is and isn't kept. Service metadata is kept 'only for as long as reasonably needed' with no period, and no DPA is published (20 of 30). No deprecation policy or dated notices, only the terms' advance notice 'when reasonably practicable' (3 of 20). The privacy policy says Prism is based in the United States and names vendors by category only, with no sub-processor list or regions (5 of 20)."
        }
      ],
      "assessment": {
        "date": "2026-10-08",
        "basis": "public evidence",
        "confidence": "medium",
        "notes": {
          "ergonomics": "Model reading of the checklist (tool use, structured output, caching, context, batch, SDKs, errors). OpenAI function tools, Anthropic tool-use blocks, `tool_choice` and `parallel_tool_calls` on Responses are documented (18 of 20). JSON mode and strict JSON Schema on Chat Completions, and `text.format` on Responses (15). A cache-read price is published for both models, $0.06 against $0.09 on DeepSeek-V4.1-Flash, but no page explains how caching is triggered, and the Claude Code guide says prompt-cache hints are accepted and ignored (5 of 15). 1M-token context and 384,000 output tokens on DeepSeek-V4.1-Flash, 32,768 on Gemma 4 31B (15). Batch appears on the pricing page as a sales-led tier with no API in the OpenAPI file (0). No SDK of its own. OpenAI and Anthropic clients are used with a base URL change (0). Every error carries `code`, `retryable`, `fix` and `docs_url`, which we confirmed on a live 401 (15).",
          "maintenance": "Model reading. Gemma 4 31B was added to the catalogue on 6 October 2026 (30). No notice period is stated. The terms promise advance notice of a material breaking change 'when reasonably practicable', and nothing has been retired yet (3 of 12). No shutdowns in the service's two weeks of public life (8 of 8). A dated changelog with one entry, a Discord server and a support booking link, none of which we tested for replies (8 of 15). No SDK repository to sample. The Hermes provider plugin on GitHub has three commits, all on 15 September 2026 (0 of 10). No official SDKs (0 of 15) and no packages to assess (0 of 10).",
          "payments": "No machine payment protocol (0). Per-token prices are on the pricing page and in the keyless `/v1/models` response, with no login (20). No free tier or trial credit was found. Billing is prepaid and a workspace without credit gets 402 (0). An agent can request a key from `POST /api/agent-signups`, but the key can't run inference until the owner supplies an emailed code, and a person funds the workspace in a browser, so half credit (10 of 20).",
          "reliability": "Hosted reading. Status page at status.prisminference.com on incident.io with a component for each of the two models (20). The incidents feed is empty and the page shows 100% for both models, but the service launched on 24 September 2026 and the domain dates from 9 September, so the record covers about two weeks of a 90-day window. We give half of the 30 for a clean record and say so as a departure from the checklist (15). The docs say per-key limits exist and publish no numbers, only a 4 MiB request cap (5). 429 carries `Retry-After` in seconds, errors carry a `retryable` flag, and the docs give backoff with jitter and rules for retrying interrupted streams (15). No published SLA. The terms say no guaranteed uptime without a separate written agreement, although the home page shows a '99.99% Uptime SLA' tile (0). The serverless API isn't labelled beta or preview, though the OpenAPI file is at version 0.1.0 (10).",
          "schema": "A public OpenAPI 3.1 file covering ten paths, including chat completions, responses, messages and models (25). llms.txt on both the site and the docs, a fuller llms-full.txt, and every docs page served as Markdown (10). Reference pages state what each endpoint is for and what isn't supported, such as stored responses and hosted tools, and the models page has a 'use when' column (15 of 20). `model` and `messages` are required, with ranges on `temperature` and `top_p` and an enum for reasoning effort, but `tool_choice` is untyped and `response_format` and the request body allow any extra properties (10 of 15). Examples in curl, TypeScript and Python, and an errors page with 13 status codes and five named codes with fixes (14 of 15). Paths are versioned under /v1 and the changelog is dated, with a single entry on 6 October 2026 (8 of 15).",
          "security": "Model reading. Bearer keys issued from account settings, with 401 for revoked or expired keys and a 30-day expiry on keys from agent sign-up. No scopes or per-key limits were found in the reviewed documentation (20 of 30). The privacy policy, the terms and the docs all say inputs and outputs are never used to train, fine-tune or evaluate models (20). Zero data retention is the default on every tier with no setting to turn on (15). The docs mention usage in the dashboard, which we did not see, and no audit log was found (5 of 15). security.txt is valid until 6 October 2027 with a contact address. No disclosure policy, bug bounty, SOC 2 or ISO 27001 was found (5 of 20).",
          "transparency": "Closed service with clear terms that name Prism Technologies Inc (15). The privacy policy, the terms and the zero-retention page agree on what is and isn't kept. Service metadata is kept 'only for as long as reasonably needed' with no period, and no DPA is published (20 of 30). No deprecation policy or dated notices, only the terms' advance notice 'when reasonably practicable' (3 of 20). The privacy policy says Prism is based in the United States and names vendors by category only, with no sub-processor list or regions (5 of 20)."
        },
        "sources": [
          {
            "what": "llms.txt, endpoints, models and agent sign-up steps",
            "url": "https://prisminference.com/llms.txt",
            "seen": "2026-10-08"
          },
          {
            "what": "full agent context, billing and compatibility limits",
            "url": "https://prisminference.com/llms-full.txt",
            "seen": "2026-10-08"
          },
          {
            "what": "docs index",
            "url": "https://docs.prisminference.com/llms.txt",
            "seen": "2026-10-08"
          },
          {
            "what": "OpenAPI 3.1 file, version 0.1.0",
            "url": "https://docs.prisminference.com/openapi.yaml",
            "seen": "2026-10-08"
          },
          {
            "what": "keyless model catalogue with prices",
            "url": "https://api.prisminference.com/v1/models",
            "seen": "2026-10-08"
          },
          {
            "what": "pricing page",
            "url": "https://prisminference.com/pricing",
            "seen": "2026-10-08"
          },
          {
            "what": "machine-readable pricing tiers",
            "url": "https://prisminference.com/pricing.json",
            "seen": "2026-10-08"
          },
          {
            "what": "models page, Gemma marked request access",
            "url": "https://docs.prisminference.com/models",
            "seen": "2026-10-08"
          },
          {
            "what": "rate limits",
            "url": "https://docs.prisminference.com/rate-limits",
            "seen": "2026-10-08"
          },
          {
            "what": "errors and retry policy",
            "url": "https://docs.prisminference.com/errors",
            "seen": "2026-10-08"
          },
          {
            "what": "authentication",
            "url": "https://docs.prisminference.com/authentication",
            "seen": "2026-10-08"
          },
          {
            "what": "agent sign-up",
            "url": "https://docs.prisminference.com/guides/agent-signup",
            "seen": "2026-10-08"
          },
          {
            "what": "zero data retention",
            "url": "https://docs.prisminference.com/zero-data-retention",
            "seen": "2026-10-08"
          },
          {
            "what": "changelog",
            "url": "https://docs.prisminference.com/changelog",
            "seen": "2026-10-08"
          },
          {
            "what": "Chat Completions reference",
            "url": "https://docs.prisminference.com/api-reference/chat-completions",
            "seen": "2026-10-08"
          },
          {
            "what": "Responses reference",
            "url": "https://docs.prisminference.com/api-reference/responses",
            "seen": "2026-10-08"
          },
          {
            "what": "Anthropic Messages reference",
            "url": "https://docs.prisminference.com/api-reference/messages",
            "seen": "2026-10-08"
          },
          {
            "what": "structured outputs guide",
            "url": "https://docs.prisminference.com/guides/structured-outputs",
            "seen": "2026-10-08"
          },
          {
            "what": "Claude Code guide",
            "url": "https://docs.prisminference.com/guides/claude-code",
            "seen": "2026-10-08"
          },
          {
            "what": "SDKs and clients",
            "url": "https://docs.prisminference.com/sdks",
            "seen": "2026-10-08"
          },
          {
            "what": "privacy policy, updated 9 September 2026",
            "url": "https://prisminference.com/privacy",
            "seen": "2026-10-08"
          },
          {
            "what": "terms of service, updated 9 September 2026",
            "url": "https://prisminference.com/terms",
            "seen": "2026-10-08"
          },
          {
            "what": "home page, speed and SLA claims",
            "url": "https://prisminference.com/",
            "seen": "2026-10-08"
          },
          {
            "what": "status page",
            "url": "https://status.prisminference.com/",
            "seen": "2026-10-08"
          },
          {
            "what": "status incidents feed (empty)",
            "url": "https://status.prisminference.com/api/v2/incidents.json",
            "seen": "2026-10-08"
          },
          {
            "what": "security.txt",
            "url": "https://prisminference.com/.well-known/security.txt",
            "seen": "2026-10-08"
          },
          {
            "what": "unauthenticated request, live error envelope",
            "url": "https://api.prisminference.com/v1/chat/completions",
            "seen": "2026-10-08"
          },
          {
            "what": "YC directory, team of 2",
            "url": "https://www.ycombinator.com/companies/prism",
            "seen": "2026-10-08"
          },
          {
            "what": "YC launch post, 24 September 2026",
            "url": "https://www.ycombinator.com/launches/UOa-prism-lightning-fast-inference-for-coding-agents",
            "seen": "2026-10-08"
          },
          {
            "what": "RDAP, domain registered 9 September 2026",
            "url": "https://rdap.org/domain/prisminference.com",
            "seen": "2026-10-08"
          },
          {
            "what": "Hermes provider plugin repository",
            "url": "https://github.com/prismhq/hermes-prism-provider",
            "seen": "2026-10-08"
          }
        ],
        "openQuestions": [
          "unchecked: the API was not called with a key and no account was opened, so tool use, structured output, caching and speed are as documented, not observed",
          "unchecked: the sign-up page and dashboard, which render only with JavaScript. Whether new accounts get any free credit, and what key controls and usage views the dashboard has, were not seen",
          "unchecked: the agent sign-up endpoints, which we did not call because they create an account",
          "unchecked: the throughput claim of 550 tokens a second, which is the vendor's own figure",
          "Whether Gemma 4 31B is open to every account. The docs say request access, and llms.txt and the keyless catalogue say available",
          "How prompt caching is triggered. A cache-read price is published and no caching page was found",
          "The status page shows 100% from July 2026, before the domain was registered on 9 September 2026, so what the early part of that window measures is unclear",
          "Judgement calls. Half marks for a clean incident record that is about two weeks long, half marks for agent sign-up that still needs an emailed code and browser funding, and a deduction of 2 for the home page SLA tile and the pricing page's 'no rate limits' line",
          "The scout listed both models as available and had not read the pricing page. Prices are public, and the docs mark Gemma as request access"
        ]
      },
      "negative": -2,
      "negativeNotes": [
        "2026-10-08. The home page shows a '99.99% Uptime SLA' tile, and the pricing page says 'No minimums, no rate limits' above the per-token table. The terms of 9 September 2026 say the services have no guaranteed uptime or service credit unless a separate written agreement says otherwise, and the docs describe per-key rate limits that return 429. No SLA document was found. A misleading claim, with the smallest deduction because the terms and docs state the real position (https://prisminference.com/, https://prisminference.com/pricing, https://prisminference.com/terms, https://docs.prisminference.com/rate-limits)."
      ],
      "verdict": "Three wire formats, a public OpenAPI 3.1 file, per-token prices in a keyless catalogue and zero data retention by default on every tier. The service launched on 24 September 2026 with two models, one of them by request, from a two-person company. No rate-limit numbers, SLA document, free tier or deprecation policy was found.",
      "bestFor": "Coding agents that want DeepSeek-V4.1-Flash at a low input price, with no retention, through whichever of the three wire formats the harness already speaks.",
      "strengths": [
        "One key works across OpenAI Chat Completions, OpenAI Responses and Anthropic Messages, with a public OpenAPI 3.1 file, llms.txt and Markdown docs",
        "Zero data retention is the default on every tier, and the privacy policy, terms and docs all say inputs and outputs are never used for training",
        "Every error carries a stable `code`, a `retryable` flag, a `fix` hint and a `docs_url`, and 429 carries `Retry-After` in seconds",
        "`GET /v1/models` answers without a key and returns context length, maximum output and per-token prices for each model",
        "DeepSeek-V4.1-Flash is listed with a 1M-token context and 384,000 output tokens at $0.09 in and $1.20 out per million"
      ],
      "weaknesses": [
        "Two models. The docs mark Gemma 4 31B as request access per organisation, while llms.txt and the keyless catalogue list it as available",
        "No rate-limit numbers are published. The docs say per-key limits exist, and the pricing page says 'no rate limits'",
        "The home page shows a '99.99% Uptime SLA' tile, while the terms say there is no guaranteed uptime without a separate written agreement",
        "No free tier found. Billing is prepaid, a person funds the workspace in a browser, and the agent sign-up key needs an emailed code",
        "The service launched on 24 September 2026. The changelog has one entry, and no deprecation policy, sub-processor list or certification was found"
      ],
      "agentNotes": [
        "Call `GET https://api.prisminference.com/v1/models` at start-up, with no key, and use only ids it returns. Expect 403 on `gemma-4-31b` without organisation access",
        "Use base URL `https://api.prisminference.com/v1` for OpenAI clients and `https://api.prisminference.com` with no `/v1` for Anthropic clients",
        "Read `error.retryable` before retrying, and wait for `Retry-After` on 429, which covers both key limits and model capacity",
        "Send `reasoning_effort: \"none\"` or `low` when latency matters. Reasoning is on by default and its tokens are billed as output",
        "Keep conversation state yourself and send `store: false` on Responses. `previous_response_id`, stored responses and hosted tools aren't supported"
      ],
      "metrics": {
        "kind": "remote",
        "measured": false
      },
      "reviewCount": 0,
      "avgRating": 0,
      "history": [
        {
          "basis": "public evidence",
          "confidence": "medium",
          "grade": "C",
          "methodology": "0.4",
          "pending": [
            "performance",
            "tasks"
          ],
          "run": "2026-10-01",
          "runLabel": "October 2026 research run",
          "score": 60.1
        }
      ],
      "editorialScores": {
        "ergonomics": 68,
        "maintenance": 49,
        "payments": 30,
        "reliability": 65,
        "schema": 82,
        "security": 65,
        "transparency": 43
      },
      "provenanceScore": 79
    },
    "connect": {
      "install": "pip install openai   # or: npm install openai, base URL https://api.prisminference.com/v1. Anthropic SDKs use https://api.prisminference.com with no /v1",
      "http": "curl \"https://api.prisminference.com/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $PRISM_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\":\"deepseek-v4.1-flash\",\"messages\":[{\"role\":\"user\",\"content\":\"Return pong.\"}]}'",
      "claudeCode": "export ANTHROPIC_BASE_URL=https://api.prisminference.com\nexport ANTHROPIC_AUTH_TOKEN=$PRISM_API_KEY\nexport ANTHROPIC_MODEL=deepseek-v4.1-flash\nexport ANTHROPIC_SMALL_FAST_MODEL=gemma-4-31b"
    },
    "letme": {
      "capability": "https://letme.dev/inference.fast",
      "tool": "https://letme.dev/prism-inference"
    },
    "notable": [
      "Launched on Y Combinator's launch board on 24 September 2026. The company is in the Spring 2025 batch with a team of 2, and earlier launched a session-replay analyser and a short-video maker under the same name (https://www.ycombinator.com/companies/prism)",
      "prisminference.com was registered on 9 September 2026 per RDAP (https://rdap.org/domain/prisminference.com)",
      "Prism says it serves DeepSeek-V4.1-Flash at 550 tokens a second, 'up to 5.8x faster' than major providers at 95 to 247. This is a vendor claim that we did not measure (https://prisminference.com/)",
      "Zero data retention is the default for every request and tier. Inputs and outputs are not written to storage, logs, analytics or backups, and are not used for training (https://docs.prisminference.com/zero-data-retention)",
      "The Responses endpoint is stateless. `store: true`, `previous_response_id`, background jobs and hosted tools such as web search aren't supported (https://prisminference.com/llms-full.txt)",
      "The docs mark `gemma-4-31b` as request access per organisation, while llms.txt and the keyless catalogue list it as available (https://docs.prisminference.com/models)",
      "The home page shows a '99.99% Uptime SLA' tile. The terms say there is no guaranteed uptime unless a separate written agreement says otherwise (https://prisminference.com/terms)"
    ],
    "area": "models",
    "details": [
      {
        "label": "Age",
        "value": "Launched 24 September 2026. Domain registered 9 September 2026. OpenAPI file at version 0.1.0. One changelog entry, 6 October 2026"
      },
      {
        "label": "Endpoints",
        "value": "POST /v1/chat/completions, /v1/responses, /v1/messages and /v1/messages/count_tokens. GET /v1/models, /v1/models/{model} and /health"
      },
      {
        "label": "Models",
        "value": "`deepseek-v4.1-flash` (text and image input, 1M context, 384,000 output tokens) and `gemma-4-31b` (text and image, 32,768 context, 8,192 output, BF16, request access per the docs)"
      },
      {
        "label": "Free tier",
        "value": "None found. Prepaid credit, 402 when the workspace has none"
      },
      {
        "label": "Trains on API data",
        "value": "No, per the privacy policy, the terms and the docs"
      },
      {
        "label": "Data retention",
        "value": "Zero retention of inputs and outputs by default on every tier. Service metadata (key id, model, token counts, status, latency, source IP) kept for as long as reasonably needed, with no period stated"
      },
      {
        "label": "Data location",
        "value": "Prism is based in the United States and may process data there and in other countries. No region list or sub-processor list found"
      },
      {
        "label": "Rate limits",
        "value": "Per key, no numbers published. 429 with `Retry-After` in seconds, also used when a model is at capacity. Requests are capped at 4 MiB"
      },
      {
        "label": "Errors",
        "value": "OpenAI-style envelope with `code`, `retryable`, `fix` and `docs_url`. 13 status codes documented with retry advice"
      },
      {
        "label": "Structured output",
        "value": "JSON mode and strict JSON Schema on Chat Completions, `text.format` on Responses"
      },
      {
        "label": "Reasoning",
        "value": "On by default. `reasoning_effort` none, low, medium or high, and reasoning tokens are billed as output"
      },
      {
        "label": "Speed",
        "value": "550 tokens a second on DeepSeek-V4.1-Flash per the vendor's home page (547 in the launch post). Not measured by us"
      },
      {
        "label": "SDKs",
        "value": "None of its own. OpenAI and Anthropic client libraries, plus provider plugins for Hermes and OpenClaw and setup guides for Claude Code, Codex, Cursor and OpenCode"
      },
      {
        "label": "SLA",
        "value": "None published. The terms say no guaranteed uptime without a separate written agreement, and dedicated deployments list a negotiated latency SLA"
      },
      {
        "label": "Status",
        "value": "status.prisminference.com on incident.io, one component per model, no incidents posted"
      },
      {
        "label": "Company",
        "value": "Prism Technologies Inc, San Francisco, Y Combinator Spring 2025, team of 2 per the YC directory"
      }
    ],
    "models": [
      {
        "id": "deepseek-v4.1-flash",
        "name": "DeepSeek-V4.1-Flash",
        "inputPer1M": 0.09,
        "outputPer1M": 1.2,
        "contextTokens": 1000000,
        "role": "default",
        "note": "cache read $0.06. Text and image input, up to 384,000 output tokens"
      },
      {
        "id": "gemma-4-31b",
        "name": "Gemma 4 31B",
        "inputPer1M": 0.3,
        "outputPer1M": 0.4,
        "contextTokens": 32768,
        "role": "fast",
        "note": "cache read $0.15. Served at BF16. The docs say access is granted per organisation on request"
      }
    ],
    "provenance": {
      "legalEntity": "Prism Technologies Inc",
      "domain": "prisminference.com",
      "domainRegistered": "2026-09-09",
      "endpointOnVendorDomain": true,
      "terms": "https://prisminference.com/terms",
      "privacy": "https://prisminference.com/privacy",
      "statusPage": "https://status.prisminference.com",
      "changelog": "https://docs.prisminference.com/changelog",
      "securityTxt": "valid",
      "checked": "2026-10-08",
      "notes": [
        "The terms and the privacy policy, both last updated 9 September 2026, name Prism Technologies Inc. The terms are governed by California law with arbitration in San Francisco.",
        "RDAP gives a registration date of 2026-09-09 for prisminference.com, with Name.com as registrar.",
        "security.txt has a Contact line (founders@prisminference.com), an Expires date of 2027-10-06 and a Canonical line. No disclosure policy or bug bounty is named.",
        "The status page runs on incident.io with one component for each model and no component for the API or the website. Its incidents feed was empty on 8 October 2026.",
        "The YC directory lists Prism in the Spring 2025 batch, in San Francisco, with a team of 2. The home page links an X account named prism_videos, from the company's earlier video product."
      ],
      "score": 79,
      "checks": [
        {
          "check": "Legal entity named",
          "value": "Prism Technologies Inc",
          "points": 20,
          "max": 20,
          "state": "ok"
        },
        {
          "check": "Domain age",
          "value": "prisminference.com, registered 2026-09-09 (under a year)",
          "points": 0,
          "max": 15,
          "state": "no"
        },
        {
          "check": "Endpoint on the vendor's domain",
          "value": "api.prisminference.com",
          "points": 15,
          "max": 15,
          "state": "ok"
        },
        {
          "check": "Terms of service",
          "value": "read, states 7 of the 7 things a reader expects, and has 1 clause that costs points",
          "points": 8,
          "max": 10,
          "state": "part"
        },
        {
          "check": "Privacy policy",
          "value": "read, states 3 of the 8 things a reader expects",
          "points": 6.3,
          "max": 10,
          "state": "part"
        },
        {
          "check": "Status page",
          "value": "status.prisminference.com",
          "points": 10,
          "max": 10,
          "state": "ok"
        },
        {
          "check": "Changelog",
          "value": "published",
          "points": 10,
          "max": 10,
          "state": "ok"
        },
        {
          "check": "security.txt",
          "value": "valid",
          "points": 10,
          "max": 10,
          "state": "ok"
        }
      ],
      "policies": [
        {
          "kind": "terms",
          "url": "https://prisminference.com/terms",
          "state": "read",
          "readAt": "2026-10-08",
          "statedDate": "2026-09-09",
          "words": 2045,
          "points": 8,
          "max": 10,
          "expected": [
            {
              "key": "terms.date",
              "label": "Gives the date it was last updated",
              "found": true,
              "quote": "Last updated: September 9, 2026",
              "says": "Last updated 2026-09-09"
            },
            {
              "key": "terms.law",
              "label": "Names the governing law or courts",
              "found": true,
              "quote": "For a dispute that is not subject to arbitration, the parties consent to the exclusive jurisdiction of state and federal courts located in San Francisco, California.",
              "says": "Disputes go to the courts of San Francisco, California"
            },
            {
              "key": "terms.liability",
              "label": "States a limit on its liability",
              "found": true,
              "quote": "For free Services, Prism's total liability arising out of or relating to the Services or these Terms will not exceed the lesser of $500 or the amount you paid Prism during the 12 months before the event giving rise to the claim.",
              "says": "Capped at the lesser of $500 and the fees paid in the 12 months before the claim"
            },
            {
              "key": "terms.termination",
              "label": "Says how the agreement or account can be ended",
              "found": true,
              "quote": "We may reject requests or restrict, suspend, or terminate access when we reasonably believe use creates security, legal, financial, or operational risk or violates these Terms."
            },
            {
              "key": "terms.changes",
              "label": "Says how changes to the terms are announced",
              "found": true,
              "quote": "We may revise these Terms by posting an updated version.",
              "says": "Changes are posted, with no other notice named"
            },
            {
              "key": "terms.use",
              "label": "Lists what users may not do",
              "found": true,
              "quote": "You may not share credentials outside your organization, publish API keys, or bypass account or plan limits."
            },
            {
              "key": "terms.sla",
              "label": "Refers to a service level or uptime commitment",
              "found": true,
              "quote": "Unless a separate written agreement states otherwise, the Services have no guaranteed uptime, latency, throughput, model availability, support response time, or service credit."
            }
          ],
          "toKnow": [
            {
              "key": "terms.automated",
              "label": "Restricts automated access",
              "found": true,
              "quote": "You may not reverse engineer the Services to extract model weights or confidential technology, scrape or overload the Services, resell access without written authorization, use false account information, or use Inputs that you lack the right to provide.",
              "costsPoints": true
            },
            {
              "key": "terms.arbitration",
              "label": "Requires arbitration or waives class actions",
              "found": true,
              "quote": "Section 13 requires most disputes to be resolved by individual binding arbitration and includes a class-action waiver."
            }
          ],
          "notes": [
            {
              "date": "2026-10-08",
              "text": "Prism says API inputs and outputs are processed transiently, not kept after processing and not used to train or improve models.",
              "quote": "We process them transiently to fulfill each request, do not persist them after processing, and do not use them to train or improve models."
            },
            {
              "date": "2026-10-08",
              "text": "Liability for paid services is capped at the lesser of 10,000 US dollars or the amount paid in the 12 months before the claim.",
              "quote": "For paid Services, the cap is the lesser of $10,000 or the amount you paid Prism during that 12-month period."
            },
            {
              "date": "2026-10-08",
              "text": "A customer may reject the arbitration agreement by email within 30 days of first accepting the terms.",
              "quote": "You may reject this arbitration agreement by emailing founders@prisminference.com within 30 days after you first accept these Terms."
            }
          ]
        },
        {
          "kind": "privacy",
          "url": "https://prisminference.com/privacy",
          "state": "read",
          "readAt": "2026-10-08",
          "statedDate": "2026-09-09",
          "words": 1218,
          "points": 6.3,
          "max": 10,
          "expected": [
            {
              "key": "privacy.date",
              "label": "Gives the date it was last updated",
              "found": true,
              "quote": "Last updated: September 9, 2026",
              "says": "Last updated 2026-09-09"
            },
            {
              "key": "privacy.collected",
              "label": "Says what personal data is collected",
              "found": true,
              "quote": "We collect information you provide when you register or manage an account, such as your name, email address, authentication identifiers, organization membership, plan, API-key records, and support communications."
            },
            {
              "key": "privacy.retention",
              "label": "Says how long data is kept",
              "found": true,
              "quote": "We keep service metadata only for as long as reasonably needed for metering, billing, security, abuse prevention, reliability, and enforcement.",
              "says": "For as long as needed, with no period named"
            },
            {
              "key": "privacy.processors",
              "label": "Says who else receives the data",
              "found": false
            },
            {
              "key": "privacy.sale",
              "label": "Says whether personal data is sold or shared for advertising",
              "found": false
            },
            {
              "key": "privacy.rights",
              "label": "Says what rights people have over their data",
              "found": false
            },
            {
              "key": "privacy.contact",
              "label": "Gives a privacy contact",
              "found": false
            },
            {
              "key": "privacy.transfers",
              "label": "Says where data is transferred or stored",
              "found": false
            }
          ],
          "notes": [
            {
              "date": "2026-10-08",
              "text": "Content a customer includes in a support request may be kept with the support record, outside the zero retention policy for API content.",
              "quote": "If you voluntarily include an Input, Output, or other content in a support request, that copy becomes part of the support communication and may be retained with the support record."
            },
            {
              "date": "2026-10-08",
              "text": "Prism says it does not use inputs or outputs to train, fine-tune, evaluate or otherwise improve its models.",
              "quote": "We do not use Inputs or Outputs to train, fine-tune, evaluate, or otherwise improve our models."
            }
          ]
        }
      ]
    },
    "pageJsonUrl": "https://www.anchorterminal.com/tools/prism-inference.json",
    "live": {
      "slug": "prism-inference",
      "probe": {
        "target": "https://api.prisminference.com/v1",
        "method": "get",
        "lastAt": "2026-10-08T17:36:43.399120505Z",
        "lastOk": true,
        "lastStatus": 200,
        "lastMs": 146,
        "authRequired": false,
        "uptime24h": 100,
        "uptime30d": 100,
        "p50ms24h": 202,
        "p95ms24h": 447,
        "samples24h": 25,
        "samples30d": 25,
        "days": [
          {
            "date": "2026-10-08",
            "probes": 25,
            "ok": 25
          }
        ]
      },
      "vendorStatus": {
        "page": "https://status.prisminference.com",
        "indicator": "none",
        "summary": "All Systems Operational",
        "checkedAt": "2026-10-08T17:25:36.245093097Z"
      },
      "versions": [
        {
          "registry": "github",
          "name": "prismhq/hermes-prism-provider",
          "version": "v1.0.2",
          "released": "2026-09-16",
          "seenAt": "2026-10-08T16:26:20.638377705Z"
        }
      ],
      "githubStars": 0,
      "securityTxt": {
        "url": "https://prisminference.com/.well-known/security.txt",
        "state": "valid",
        "expires": "2027-10-06T00:00:00.000Z",
        "checkedAt": "2026-10-08T15:38:58.310470594Z"
      },
      "updatedAt": "2026-10-08T17:36:43.399120505Z"
    }
  }
}
