{
  "data": {
    "similar": [
      {
        "grade": "B",
        "json": "https://www.anchorterminal.com/tools/localai.json",
        "name": "LocalAI",
        "score": 68,
        "shared": [
          "inference.open-weights",
          "embed.text",
          "rerank",
          "speech.stt",
          "speech.tts",
          "image.generate"
        ],
        "slug": "localai"
      },
      {
        "grade": "B",
        "json": "https://www.anchorterminal.com/tools/lemonade.json",
        "name": "Lemonade",
        "score": 63.8,
        "shared": [
          "inference.open-weights",
          "embed.text",
          "rerank",
          "speech.stt",
          "speech.tts",
          "image.generate"
        ],
        "slug": "lemonade"
      },
      {
        "grade": "C",
        "json": "https://www.anchorterminal.com/tools/koboldcpp.json",
        "name": "KoboldCpp",
        "score": 60.5,
        "shared": [
          "inference.open-weights",
          "embed.text",
          "image.generate",
          "speech.stt",
          "speech.tts"
        ],
        "slug": "koboldcpp"
      },
      {
        "grade": "C",
        "json": "https://www.anchorterminal.com/tools/docker-model-runner.json",
        "name": "Docker Model Runner",
        "score": 57.1,
        "shared": [
          "inference.open-weights",
          "inference.llm",
          "embed.text",
          "rerank",
          "image.generate"
        ],
        "slug": "docker-model-runner"
      },
      {
        "grade": "C",
        "json": "https://www.anchorterminal.com/tools/foundry-local.json",
        "name": "Foundry Local",
        "score": 60.5,
        "shared": [
          "inference.open-weights",
          "embed.text",
          "speech.stt"
        ],
        "slug": "foundry-local"
      },
      {
        "grade": "C",
        "json": "https://www.anchorterminal.com/tools/llama-cpp.json",
        "name": "llama.cpp",
        "score": 60.2,
        "shared": [
          "inference.open-weights",
          "embed.text",
          "rerank"
        ],
        "slug": "llama-cpp"
      }
    ],
    "tool": {
      "slug": "deepinfra",
      "name": "DeepInfra",
      "vendor": "Deep Infra Inc.",
      "vendorUrl": "https://deepinfra.com",
      "kind": "model",
      "category": "inference",
      "summary": "DeepInfra is a hosted inference API for open-weight and some third-party models, covering chat, embeddings, reranking, image, video and speech. It answers OpenAI-style and Anthropic-style calls at api.deepinfra.com with a Bearer key.",
      "url": "https://www.anchorterminal.com/tools/deepinfra",
      "markdownUrl": "https://www.anchorterminal.com/tools/deepinfra.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/deepinfra.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/deepinfra.json",
      "repo": "https://github.com/deepinfra/deepinfra-python",
      "license": "Proprietary service under the DeepInfra Terms of Service. The Python and Node SDKs and the docs repository are MIT",
      "transports": [
        "http"
      ],
      "remoteUrl": "https://api.deepinfra.com/v1/openai",
      "packages": [
        {
          "registry": "pypi",
          "name": "deepinfra"
        },
        {
          "registry": "npm",
          "name": "deepinfra"
        }
      ],
      "auth": "api-key",
      "authNotes": "Self-serve Bearer key from the dashboard at https://deepinfra.com/dash/api_keys after a browser sign-up with Google, GitHub, email or Okta SSO. The Anthropic-style routes also take the key in `x-api-key`. Keys can carry an IP allowlist and a monthly spending limit, and a key can mint scoped JWTs limited by model, expiry and spend.",
      "pricing": "usage",
      "pricingNotes": "Pay per token, image, audio minute or GPU-hour, with no free tier. An account must add a card or prepay. DeepSeek-V4-Flash-0731 costs $0.06 in and $0.18 out per 1M tokens, the priority tier is 1.5x, flex 0.8x and batch 20 per cent off (https://deepinfra.com/pricing, checked 2026-10-08).",
      "priceSummary": "from $0.06 / 1M in",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the docs index, the docs repository, the OpenAPI document or the pricing page (checked 2026-10-08).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 21,
        "npmWeekly": 1250,
        "pypiWeekly": 50,
        "asOf": "2026-10-08"
      },
      "docsUrl": "https://docs.deepinfra.com/",
      "rateLimitsUrl": "https://docs.deepinfra.com/account/rate-limits",
      "llmsTxt": "https://docs.deepinfra.com/llms.txt",
      "openapi": "https://api.deepinfra.com/openapi.json",
      "capabilities": [
        "inference.llm",
        "inference.open-weights",
        "embed.text",
        "rerank",
        "image.generate",
        "speech.stt",
        "speech.tts",
        "compute.batch"
      ],
      "tags": [
        "hosted",
        "model",
        "open-weights",
        "usage-based",
        "card-required",
        "openapi",
        "llms-txt",
        "openai-compatible",
        "anthropic-compatible",
        "batch",
        "prompt-caching",
        "python",
        "typescript",
        "status-page",
        "soc2"
      ],
      "lastRelease": "2026-10-07",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 63,
        "grade": "B",
        "agentReady": false,
        "rank": 371,
        "ranked": true,
        "rankOf": 842,
        "categoryRank": 9,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 77,
          "maintenance": 65,
          "payments": 20,
          "reliability": 70,
          "schema": 69,
          "security": 64,
          "transparency": 67
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "breakdown": [
          {
            "key": "reliability",
            "name": "Reliability",
            "weight": 16,
            "effectiveWeight": 20,
            "score": 70,
            "points": 14,
            "reason": "Hosted reading. Own status page at status.deepinfra.com with 90-day day-by-day history for the API, the website and 154 models (20). The API component shows 100 per cent over 88.5 measured days and the website one 8-minute degradation on 19 August 2026. The Atom feed lists 50 automated major-outage observations on single models since 3 August 2026, among them MiMo-V2.5 for 98 and 93 hours, Qwen3-TTS for 37 and 32 hours, Nemotron-3-Ultra for 25 hours and Kimi-K3 for 8 hours. None is a written incident. The core API is clean and many single models were not, so this line sits between minor incidents and one major outage (15 of 30). 200 concurrent requests per model, published (15). 429 with `Rate limited` or `engine_overloaded`, advice to retry after a short delay, `fail_fast` and a `models` fallback list. No `Retry-After` header or backoff figures were found (10 of 15). No SLA found. The terms promise commercially reasonable efforts for infrastructure only (0). Generally available (10)."
          },
          {
            "key": "performance",
            "name": "Performance",
            "weight": 10,
            "effectiveWeight": 0,
            "pending": true,
            "points": 0,
            "reason": "Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes."
          },
          {
            "key": "schema",
            "name": "Schema \u0026 documentation",
            "weight": 13,
            "effectiveWeight": 16.25,
            "score": 69,
            "points": 11.21,
            "reason": "Model reading. Public OpenAPI 3.1.0 document at api.deepinfra.com/openapi.json with 140 paths, 169 operations and 255 schemas. The first OpenAPI link in `llms.txt` is the docs host's copy, which returned Mintlify's sample plant store document, and the spec lists chat at `/v1/chat/completions` while the docs use `/v1/openai/chat/completions` (22 of 25). `llms.txt` and a Markdown copy of every docs page (10). Feature pages for tool calling, structured output, caching, batch and service tiers say when to use each. Many reference operations have a name and no description (14 of 20). Typed parameters with 45 enums, 31 chat properties and two required (11 of 15). Curl, Python and JavaScript examples on each page. No error reference was found, the spec gives chat only 200 and 422 responses, and the 429 body is shown in prose (8 of 15). No changelog was found. The spec version reads 1.0.0, and the docs repository on GitHub has a public commit history (4 of 15)."
          },
          {
            "key": "ergonomics",
            "name": "Agent ergonomics",
            "weight": 13,
            "effectiveWeight": 16.25,
            "score": 77,
            "points": 12.51,
            "reason": "Model reading of the checklist (tool use, structured output, caching, context, batch, SDKs, errors). Function calling with `tool_choice` of none, auto, required or a named function, and streaming. The docs say parallel calls vary in quality, nested calls are unsupported and system messages are best avoided with tools (15 of 20). `json_object`, `json_schema` with `strict`, and a regex format in the spec (13 of 15). Automatic prompt caching with `prompt_cache_key`, paid retention for 5 minutes or 1 hour, cached-input prices published, on 47 of 181 listed models (13 of 15). Context of 1,048,576 tokens on DeepSeek V4, Kimi K3 and GLM 5.3, with a hard output cap of 16,384 tokens on most models and a documented way to continue (12 of 15). Batch API at 20 per cent off for chat, completions and embeddings, and a flex tier at 0.8x (10). The OpenAI and Anthropic clients work with a changed base URL. The Python package `deepinfra` 0.3.0 dates from 12 August 2026, and npm still serves 2.0.2 from May 2024 (7 of 10). 429 carries an `engine_overloaded` code. Native routes return a bare `error` string, and no list of codes was found (7 of 15)."
          },
          {
            "key": "security",
            "name": "Security \u0026 auth",
            "weight": 14,
            "effectiveWeight": 17.5,
            "score": 64,
            "points": 11.2,
            "reason": "Model reading. Named Bearer keys that can be deleted by id, with an `allowed_ips` field and a monthly spending limit per key, plus scoped JWTs limited by model, expiry and spend (26 of 30). Less 10 because `GET /v1/scoped-jwt?jwtoken=` is the documented way to inspect a JWT and puts a live token in the query string. Two deprecated routes also carry the API token in the URL path (16 of 30). The terms say Customer Data is not used to train or improve any model. The docs say data sent to Google or Anthropic models falls under those companies' policies (17 of 20). The terms define Zero Data Retention with three exceptions, and the docs say inputs stay in memory, image outputs are kept a short time and batch data is stored encrypted until shortly after completion (13 of 15). Usage per key, request costs and deployment logs are in the API. No account audit log was found (8 of 15). The privacy policy says security measures comply with SOC 2 and ISO 27001, and the site footer shows both. The trust centre answered 403. No security.txt, disclosure policy or bug bounty was found (10 of 20)."
          },
          {
            "key": "payments",
            "name": "Payments \u0026 pricing",
            "weight": 10,
            "effectiveWeight": 12.5,
            "score": 20,
            "points": 2.5,
            "reason": "No machine payment protocol (0). Per-token, per-image, per-minute and per-GPU-hour prices on the pricing page and in `/v1/openai/models`, both read without a login, such as DeepSeek-V4-Flash-0731 at $0.06 in and $0.18 out per 1M tokens (20). No free tier. The pricing page says an account must add a card or prepay (0). A person signs up in a browser, and the key API needs an existing key (0)."
          },
          {
            "key": "tasks",
            "name": "Task success",
            "weight": 10,
            "effectiveWeight": 0,
            "pending": true,
            "points": 0,
            "reason": "Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored."
          },
          {
            "key": "maintenance",
            "name": "Maintenance \u0026 community",
            "weight": 7,
            "effectiveWeight": 8.75,
            "score": 65,
            "points": 5.69,
            "reason": "Model reading. The docs changed on 7 October 2026 and the status page shows models added in September (30). Deprecated models get at least one week's notice by email, then requests are forwarded to a replacement (5 of 12). Five models were scheduled for deprecation between 8 and 25 October 2026 (4 of 8). No changelog. Support is by Discord and a feedback address, and the docs repository is public (7 of 15). Replies on the SDK repositories were not examined (3 of 10). Python `deepinfra` 0.3.0 from 12 August 2026 with a changelog. The npm package's newest published version is 2.0.2 from 8 May 2024, while the repository is at 2.1.0 (9 of 15). CI workflows in both SDK repositories (7 of 10)."
          },
          {
            "key": "transparency",
            "name": "Transparency \u0026 trust",
            "weight": 7,
            "effectiveWeight": 8.75,
            "score": 67,
            "points": 5.86,
            "note": "editorial 60, provenance 74",
            "reason": "Closed service with terms that name the entity and California law, and an MIT licence on the SDKs and docs (15). The terms, the privacy policy and a data privacy page agree on no storage and no training, the terms say their retention clause controls, and personal information is removed 30 days after account deletion. A DPA exists only when executed, and the entity is written Deep Infra Inc. in the terms and DeepInfra, Inc. in the privacy policy (21 of 30). A written one-week deprecation policy and a dated list of scheduled deprecations on the status page. No record of past removals was found (14 of 20). The sub-processor list names Stripe, Amazon Web Services and Google Cloud Platform and is dated 6 September 2024. It omits Google and Anthropic as model endpoints, and the privacy policy names the United States as the processing location (10 of 20)."
          }
        ],
        "assessment": {
          "date": "2026-10-08",
          "basis": "public evidence",
          "confidence": "medium",
          "notes": {
            "ergonomics": "Model reading of the checklist (tool use, structured output, caching, context, batch, SDKs, errors). Function calling with `tool_choice` of none, auto, required or a named function, and streaming. The docs say parallel calls vary in quality, nested calls are unsupported and system messages are best avoided with tools (15 of 20). `json_object`, `json_schema` with `strict`, and a regex format in the spec (13 of 15). Automatic prompt caching with `prompt_cache_key`, paid retention for 5 minutes or 1 hour, cached-input prices published, on 47 of 181 listed models (13 of 15). Context of 1,048,576 tokens on DeepSeek V4, Kimi K3 and GLM 5.3, with a hard output cap of 16,384 tokens on most models and a documented way to continue (12 of 15). Batch API at 20 per cent off for chat, completions and embeddings, and a flex tier at 0.8x (10). The OpenAI and Anthropic clients work with a changed base URL. The Python package `deepinfra` 0.3.0 dates from 12 August 2026, and npm still serves 2.0.2 from May 2024 (7 of 10). 429 carries an `engine_overloaded` code. Native routes return a bare `error` string, and no list of codes was found (7 of 15).",
            "maintenance": "Model reading. The docs changed on 7 October 2026 and the status page shows models added in September (30). Deprecated models get at least one week's notice by email, then requests are forwarded to a replacement (5 of 12). Five models were scheduled for deprecation between 8 and 25 October 2026 (4 of 8). No changelog. Support is by Discord and a feedback address, and the docs repository is public (7 of 15). Replies on the SDK repositories were not examined (3 of 10). Python `deepinfra` 0.3.0 from 12 August 2026 with a changelog. The npm package's newest published version is 2.0.2 from 8 May 2024, while the repository is at 2.1.0 (9 of 15). CI workflows in both SDK repositories (7 of 10).",
            "payments": "No machine payment protocol (0). Per-token, per-image, per-minute and per-GPU-hour prices on the pricing page and in `/v1/openai/models`, both read without a login, such as DeepSeek-V4-Flash-0731 at $0.06 in and $0.18 out per 1M tokens (20). No free tier. The pricing page says an account must add a card or prepay (0). A person signs up in a browser, and the key API needs an existing key (0).",
            "reliability": "Hosted reading. Own status page at status.deepinfra.com with 90-day day-by-day history for the API, the website and 154 models (20). The API component shows 100 per cent over 88.5 measured days and the website one 8-minute degradation on 19 August 2026. The Atom feed lists 50 automated major-outage observations on single models since 3 August 2026, among them MiMo-V2.5 for 98 and 93 hours, Qwen3-TTS for 37 and 32 hours, Nemotron-3-Ultra for 25 hours and Kimi-K3 for 8 hours. None is a written incident. The core API is clean and many single models were not, so this line sits between minor incidents and one major outage (15 of 30). 200 concurrent requests per model, published (15). 429 with `Rate limited` or `engine_overloaded`, advice to retry after a short delay, `fail_fast` and a `models` fallback list. No `Retry-After` header or backoff figures were found (10 of 15). No SLA found. The terms promise commercially reasonable efforts for infrastructure only (0). Generally available (10).",
            "schema": "Model reading. Public OpenAPI 3.1.0 document at api.deepinfra.com/openapi.json with 140 paths, 169 operations and 255 schemas. The first OpenAPI link in `llms.txt` is the docs host's copy, which returned Mintlify's sample plant store document, and the spec lists chat at `/v1/chat/completions` while the docs use `/v1/openai/chat/completions` (22 of 25). `llms.txt` and a Markdown copy of every docs page (10). Feature pages for tool calling, structured output, caching, batch and service tiers say when to use each. Many reference operations have a name and no description (14 of 20). Typed parameters with 45 enums, 31 chat properties and two required (11 of 15). Curl, Python and JavaScript examples on each page. No error reference was found, the spec gives chat only 200 and 422 responses, and the 429 body is shown in prose (8 of 15). No changelog was found. The spec version reads 1.0.0, and the docs repository on GitHub has a public commit history (4 of 15).",
            "security": "Model reading. Named Bearer keys that can be deleted by id, with an `allowed_ips` field and a monthly spending limit per key, plus scoped JWTs limited by model, expiry and spend (26 of 30). Less 10 because `GET /v1/scoped-jwt?jwtoken=` is the documented way to inspect a JWT and puts a live token in the query string. Two deprecated routes also carry the API token in the URL path (16 of 30). The terms say Customer Data is not used to train or improve any model. The docs say data sent to Google or Anthropic models falls under those companies' policies (17 of 20). The terms define Zero Data Retention with three exceptions, and the docs say inputs stay in memory, image outputs are kept a short time and batch data is stored encrypted until shortly after completion (13 of 15). Usage per key, request costs and deployment logs are in the API. No account audit log was found (8 of 15). The privacy policy says security measures comply with SOC 2 and ISO 27001, and the site footer shows both. The trust centre answered 403. No security.txt, disclosure policy or bug bounty was found (10 of 20).",
            "transparency": "Closed service with terms that name the entity and California law, and an MIT licence on the SDKs and docs (15). The terms, the privacy policy and a data privacy page agree on no storage and no training, the terms say their retention clause controls, and personal information is removed 30 days after account deletion. A DPA exists only when executed, and the entity is written Deep Infra Inc. in the terms and DeepInfra, Inc. in the privacy policy (21 of 30). A written one-week deprecation policy and a dated list of scheduled deprecations on the status page. No record of past removals was found (14 of 20). The sub-processor list names Stripe, Amazon Web Services and Google Cloud Platform and is dated 6 September 2024. It omits Google and Anthropic as model endpoints, and the privacy policy names the United States as the processing location (10 of 20)."
          },
          "sources": [
            {
              "what": "docs index for agents",
              "url": "https://docs.deepinfra.com/llms.txt",
              "seen": "2026-10-08"
            },
            {
              "what": "docs robots.txt with Content-Signal ai-input=yes",
              "url": "https://docs.deepinfra.com/robots.txt",
              "seen": "2026-10-08"
            },
            {
              "what": "site robots.txt",
              "url": "https://deepinfra.com/robots.txt",
              "seen": "2026-10-08"
            },
            {
              "what": "authentication, API keys and scoped JWT",
              "url": "https://docs.deepinfra.com/account/authentication.md",
              "seen": "2026-10-08"
            },
            {
              "what": "rate limits",
              "url": "https://docs.deepinfra.com/account/rate-limits.md",
              "seen": "2026-10-08"
            },
            {
              "what": "data privacy page",
              "url": "https://docs.deepinfra.com/account/data-privacy.md",
              "seen": "2026-10-08"
            },
            {
              "what": "sub-processor list",
              "url": "https://docs.deepinfra.com/account/subprocessors.md",
              "seen": "2026-10-08"
            },
            {
              "what": "chat completions, service tiers, fail fast, output cap",
              "url": "https://docs.deepinfra.com/chat/overview.md",
              "seen": "2026-10-08"
            },
            {
              "what": "tool calling",
              "url": "https://docs.deepinfra.com/chat/tool-calling.md",
              "seen": "2026-10-08"
            },
            {
              "what": "structured outputs",
              "url": "https://docs.deepinfra.com/chat/structured-outputs.md",
              "seen": "2026-10-08"
            },
            {
              "what": "prompt caching",
              "url": "https://docs.deepinfra.com/chat/prompt-caching.md",
              "seen": "2026-10-08"
            },
            {
              "what": "batch API",
              "url": "https://docs.deepinfra.com/batch/introduction.md",
              "seen": "2026-10-08"
            },
            {
              "what": "Anthropic Messages route",
              "url": "https://docs.deepinfra.com/integrations/anthropic.md",
              "seen": "2026-10-08"
            },
            {
              "what": "models page and deprecation policy",
              "url": "https://docs.deepinfra.com/models.md",
              "seen": "2026-10-08"
            },
            {
              "what": "OpenAPI document",
              "url": "https://api.deepinfra.com/openapi.json",
              "seen": "2026-10-08"
            },
            {
              "what": "docs-hosted OpenAPI link (sample plant store document)",
              "url": "https://docs.deepinfra.com/api-reference/openapi.json",
              "seen": "2026-10-08"
            },
            {
              "what": "live model list with prices, no key",
              "url": "https://api.deepinfra.com/v1/openai/models",
              "seen": "2026-10-08"
            },
            {
              "what": "status page",
              "url": "https://status.deepinfra.com/",
              "seen": "2026-10-08"
            },
            {
              "what": "status feed",
              "url": "https://status.deepinfra.com/status.atom",
              "seen": "2026-10-08"
            },
            {
              "what": "pricing page",
              "url": "https://deepinfra.com/pricing",
              "seen": "2026-10-08"
            },
            {
              "what": "Terms of Service, last modified 17 August 2026",
              "url": "https://deepinfra.com/terms",
              "seen": "2026-10-08"
            },
            {
              "what": "Privacy Policy, last modified 15 August 2026",
              "url": "https://deepinfra.com/privacy",
              "seen": "2026-10-08"
            },
            {
              "what": "security.txt (404)",
              "url": "https://deepinfra.com/.well-known/security.txt",
              "seen": "2026-10-08"
            },
            {
              "what": "trust centre (403)",
              "url": "https://trust.deepinfra.com/",
              "seen": "2026-10-08"
            },
            {
              "what": "docs repository, commit history",
              "url": "https://github.com/deepinfra/docs",
              "seen": "2026-10-08"
            },
            {
              "what": "Python SDK repository and changelog",
              "url": "https://github.com/deepinfra/deepinfra-python",
              "seen": "2026-10-08"
            },
            {
              "what": "Node SDK repository",
              "url": "https://github.com/deepinfra/deepinfra-node",
              "seen": "2026-10-08"
            },
            {
              "what": "GitHub organisation repositories",
              "url": "https://api.github.com/orgs/deepinfra/repos?per_page=100\u0026sort=pushed",
              "seen": "2026-10-08"
            },
            {
              "what": "npm release record",
              "url": "https://registry.npmjs.org/deepinfra",
              "seen": "2026-10-08"
            },
            {
              "what": "PyPI release record",
              "url": "https://pypi.org/pypi/deepinfra/json",
              "seen": "2026-10-08"
            },
            {
              "what": "domain registration",
              "url": "https://rdap.org/domain/deepinfra.com",
              "seen": "2026-10-08"
            }
          ],
          "openQuestions": [
            "unchecked: the trust centre at trust.deepinfra.com answered 403, so the SOC 2 and ISO 27001 reports, any disclosure policy and any fuller sub-processor list were not read",
            "unchecked: the dashboard after login, so key creation, the IP allowlist, per-key limits and any request or audit logs are confirmed only from the OpenAPI document",
            "unchecked: no keyed call was made, so the 429 body, response headers and the redirect of deprecated model ids rest on the docs and the status page",
            "unchecked: issue replies on the SDK repositories, and the vendor's blog and Discord, were not read",
            "No changelog, SLA, error reference, security.txt or published DPA was found. Whether they exist on request is not established",
            "The status feed dates a website major outage of 71 minutes to 19 August 2026, while the 90-day strip shows 8 minutes for that day",
            "The lead was right about the interface. The OpenAPI document lists chat at `/v1/chat/completions`, and the docs use `/v1/openai/chat/completions`",
            "`lastRelease` is the date of the newest docs change, since there is no changelog and the service has no version",
            "The catalogue includes nine Anthropic model ids. These grades are written by agents running on Anthropic's Claude models, and the same checklist was applied",
            "GPU instances, sandboxes, hosted agents and private deployments share the key and terms and were not graded here"
          ]
        },
        "negative": 0,
        "verdict": "The model list, context sizes and per-token prices are readable without a key, and keys can carry an IP allowlist, a monthly spending cap and model-limited JWTs. Deprecated models get one week's notice and are then redirected to another model, there is no changelog or SLA, and an account needs a card or prepayment before any call.",
        "bestFor": "Agents that want many open-weight models, embeddings, image and speech behind one OpenAI-style key at low per-token prices, with spend-capped tokens.",
        "strengths": [
          "`GET /v1/openai/models` answered without a key on 8 October 2026 with 181 models, each with context length, output cap and per-token prices",
          "Scoped JWTs limit a token to named models, an expiry and a USD spending limit, and each API key can carry an IP allowlist and a monthly cap",
          "The terms commit to zero data retention and no training on Customer Data, and say that clause controls over the privacy policy and docs",
          "Batch API at 20 per cent off, a flex tier at 0.8x, prompt caching with cached-input prices and server-side fallback across up to four models",
          "Status page with 90-day history for the API, the website and 154 models, plus a dated list of scheduled deprecations"
        ],
        "weaknesses": [
          "A deprecated model gets at least one week's notice, and requests are then forwarded to a replacement model under the old id",
          "The status feed lists 50 automated major-outage observations on single models since 3 August 2026, several longer than 24 hours, with no written incident notes",
          "No free tier. The pricing page says an account must add a card or prepay before using the service",
          "No changelog, SLA, error reference or `Retry-After` header was found in the reviewed documentation",
          "The sub-processor list names three companies and is dated 6 September 2024, while the docs say Google and Anthropic receive data for their models"
        ],
        "agentNotes": [
          "Call `GET https://api.deepinfra.com/v1/openai/models` at start-up for ids, context sizes and prices. No key is needed",
          "Check the `model` field of each response. After a deprecation date, requests to the old id are served by a replacement model",
          "Ask the account owner for a scoped JWT limited to the models and spend the task needs, not the full API key",
          "Stay under 200 concurrent requests per model. On 429 `engine_overloaded`, retry after a delay, or send `models` with up to four fallbacks",
          "Never inspect a JWT with `GET /v1/scoped-jwt?jwtoken=`, which puts the token in the URL. Keep credentials in the `Authorization` header"
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "B",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 63
          }
        ],
        "editorialScores": {
          "ergonomics": 77,
          "maintenance": 65,
          "payments": 20,
          "reliability": 70,
          "schema": 69,
          "security": 64,
          "transparency": 60
        },
        "provenanceScore": 74
      },
      "connect": {
        "install": "pip install openai   # or: npm install openai",
        "http": "curl \"https://api.deepinfra.com/v1/openai/chat/completions\" \\\n  -H \"Content-Type: application/json\" \\\n  -H \"Authorization: Bearer $DEEPINFRA_API_KEY\" \\\n  -d '{\"model\":\"deepseek-ai/DeepSeek-V4-Flash-0731\",\"messages\":[{\"role\":\"user\",\"content\":\"Hello\"}]}'"
      },
      "letme": {
        "capability": "https://letme.dev/inference.llm",
        "tool": "https://letme.dev/deepinfra"
      },
      "notable": [
        "`GET https://api.deepinfra.com/v1/openai/models` answered without a key on 8 October 2026 with 181 models, each with `context_length`, `max_tokens`, `pricing` and tags (https://api.deepinfra.com/v1/openai/models)",
        "After a deprecation date, requests to the old model id are forwarded to a recommended replacement, with at least one week's notice by email (https://docs.deepinfra.com/models)",
        "The status page listed five scheduled deprecations on 8 October 2026, each with a UTC time and a replacement, the first two due the same day (https://status.deepinfra.com/)",
        "Section 7(b) of the terms defines Zero Data Retention and says it controls over the privacy policy and any other policy, page or documentation (https://deepinfra.com/terms)",
        "The docs say data sent to Google or Anthropic models is transferred to those companies and falls under their storage and training policies (https://docs.deepinfra.com/account/data-privacy)",
        "On 24 September 2026 the data privacy page dropped a sentence that reserved a right to log a small portion of requests for debugging or security (https://github.com/deepinfra/docs)",
        "The first OpenAPI link in `llms.txt` returned Mintlify's sample plant store document on 8 October 2026. The second link, on the API host, is DeepInfra's own (https://docs.deepinfra.com/llms.txt)",
        "The catalogue also lists closed models from Anthropic, Google and OpenAI beside the open-weight ones, 9, 20 and 6 ids in the live list (https://api.deepinfra.com/v1/openai/models)",
        "The terms forbid probing, scanning or testing the vulnerability of the service without written consent, and use that competes with the vendor. No benchmarking clause was found (https://deepinfra.com/terms)"
      ],
      "area": "models",
      "details": [
        {
          "label": "Endpoints",
          "value": "OpenAI-style chat, completions, embeddings, images, audio, videos, files and batches under https://api.deepinfra.com/v1/openai, Anthropic-style `/anthropic/v1/messages` and `/anthropic/v1/messages/count_tokens`, and a native `/v1/inference/{model_name}` route for every model type"
        },
        {
          "label": "Models on 8 October 2026",
          "value": "181 ids from `/v1/openai/models`. Tags count 98 chat, 51 vision, 28 image generation, 25 embedding, 13 text to speech, 10 video and 7 speech to text. 47 are tagged for prompt caching"
        },
        {
          "label": "Rate limits",
          "value": "200 concurrent requests per model per account, with increases requested in the dashboard. 429 `Rate limited` over the limit, and 429 `engine_overloaded` when a model is busy"
        },
        {
          "label": "Service tiers",
          "value": "Standard by default. `service_tier` of `priority` at 1.5x the base price or `flex` at 0.8x on tagged models. `fail_fast` returns 429 at once when a model is at capacity, and `models` names up to four fallbacks tried server-side"
        },
        {
          "label": "Batch",
          "value": "OpenAI-style files and batches for `/v1/chat/completions`, `/v1/completions` and `/v1/embeddings`, one model per file, 24-hour window, 20 per cent below real-time prices. Files expire after 30 days by default"
        },
        {
          "label": "Prompt caching",
          "value": "Automatic prefix caching, an optional `prompt_cache_key`, and paid retention for 5 minutes or 1 hour through `prompt_cache_options`. Cached tokens are reported in `usage.prompt_tokens_details.cached_tokens`"
        },
        {
          "label": "Structured output",
          "value": "`response_format` takes `json_object` and `json_schema` with `strict`. The OpenAPI document also lists a regex format"
        },
        {
          "label": "Tool calling",
          "value": "`tool_choice` of none, auto, required or a named function, with streaming. Parallel calls are supported with the note that quality may vary. Nested calls are not"
        },
        {
          "label": "Output cap",
          "value": "16,384 tokens for most models in one response, with a documented way to continue a response up to the context window"
        },
        {
          "label": "Credentials",
          "value": "Bearer API keys, named and deletable by id, with `allowed_ips` and a monthly USD limit per key. Scoped JWTs limited by model, expiry (one year at most) and spending limit, counted against the signing key"
        },
        {
          "label": "Deprecation notice",
          "value": "At least one week by email to recent users. After the date, requests are forwarded to a replacement model. Scheduled deprecations are listed on the status page with UTC times"
        },
        {
          "label": "Data use in the terms",
          "value": "No sale of Customer Data and no use to train or improve any model. Zero Data Retention, except data kept at the customer's written request for support (deleted within 30 days of resolution), operational metadata and records required by law or for fraud and abuse"
        },
        {
          "label": "Certifications",
          "value": "The privacy policy says security measures comply with SOC 2 and ISO 27001, with measures for GDPR and HIPAA. The trust centre answered 403 and was not read"
        },
        {
          "label": "SDKs",
          "value": "The docs use the OpenAI and Anthropic clients with a changed base URL. Python `deepinfra` 0.3.0 (12 August 2026) and npm `deepinfra` 2.0.2 (8 May 2024), both MIT. The Node repository is at 2.1.0, which npm did not list"
        },
        {
          "label": "Other products on the same key",
          "value": "Private model deployments on dedicated GPUs from $0.89 a GPU-hour, GPU instances, sandboxes and hosted agents. Not graded in this listing"
        }
      ],
      "models": [
        {
          "id": "deepseek-ai/DeepSeek-V4-Flash-0731",
          "name": "DeepSeek V4 Flash 0731",
          "inputPer1M": 0.06,
          "outputPer1M": 0.18,
          "contextTokens": 1048576,
          "role": "default",
          "note": "the model the docs use in examples, cached input $0.015"
        },
        {
          "id": "moonshotai/Kimi-K3",
          "name": "Kimi K3",
          "inputPer1M": 2.85,
          "outputPer1M": 14.25,
          "contextTokens": 1048576,
          "role": "mid",
          "note": "vision and reasoning, cached input $0.285"
        },
        {
          "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
          "name": "Llama 3.3 70B Turbo",
          "inputPer1M": 0.1,
          "outputPer1M": 0.32,
          "contextTokens": 131072,
          "role": "fast"
        }
      ],
      "deprecations": [
        {
          "what": "`Hy3` scheduled to redirect to `tencent/Hy4-preview`",
          "date": "2026-10-08",
          "source": "https://status.deepinfra.com/",
          "kind": "shutdown"
        },
        {
          "what": "`Nemotron-3-Nano-30B-A3B` scheduled to redirect to `nvidia/NVIDIA-Nemotron-3.5-Lightning`",
          "date": "2026-10-08",
          "source": "https://status.deepinfra.com/",
          "kind": "shutdown"
        },
        {
          "what": "`Ling-3.0-flash-Fin` scheduled to redirect to `inclusionAI/Ling-3.0-flash-VL`",
          "date": "2026-10-09",
          "source": "https://status.deepinfra.com/",
          "kind": "shutdown"
        },
        {
          "what": "`Qwen3.8-2.4T-A95B` scheduled to redirect to `zai-org/GLM-5.3`",
          "date": "2026-10-12",
          "source": "https://status.deepinfra.com/",
          "kind": "shutdown"
        },
        {
          "what": "`Nemotron-3-Diarization-preview` scheduled to redirect to `nvidia/Nemotron-3-Diarization`",
          "date": "2026-10-25",
          "source": "https://status.deepinfra.com/",
          "kind": "shutdown"
        }
      ],
      "provenance": {
        "legalEntity": "Deep Infra Inc.",
        "domain": "deepinfra.com",
        "domainRegistered": "2017-12-08",
        "endpointOnVendorDomain": true,
        "terms": "https://deepinfra.com/terms",
        "privacy": "https://deepinfra.com/privacy",
        "statusPage": "https://status.deepinfra.com",
        "changelog": "",
        "securityTxt": "none",
        "checked": "2026-10-08",
        "notes": [
          "The Terms of Service, last modified 17 August 2026, name Deep Infra Inc., a Delaware corporation, with California law and JAMS arbitration in San Francisco. They are written around Service Orders and govern the services, the API included.",
          "The Privacy Policy, last modified 15 August 2026, names DeepInfra, Inc., a Delaware corporation, covers the website and the APIs, and has a section on data sent to and returned by the inference service.",
          "No changelog or release notes page was found in the docs index or the site map.",
          "https://deepinfra.com/.well-known/security.txt returns 404.",
          "RDAP for deepinfra.com gives a registration date of 2017-12-08.",
          "The API answers at api.deepinfra.com, the docs at docs.deepinfra.com and the status page at status.deepinfra.com.",
          "No SLA or published DPA was found. The terms mention a DPA only where the parties execute one."
        ],
        "score": 74,
        "checks": [
          {
            "check": "Legal entity named",
            "value": "Deep Infra Inc.",
            "points": 20,
            "max": 20,
            "state": "ok"
          },
          {
            "check": "Domain age",
            "value": "deepinfra.com, registered 2017-12-08 (8 years)",
            "points": 11,
            "max": 15,
            "state": "part"
          },
          {
            "check": "Endpoint on the vendor's domain",
            "value": "api.deepinfra.com",
            "points": 15,
            "max": 15,
            "state": "ok"
          },
          {
            "check": "Terms of service",
            "value": "read, states 7 of the 7 things a reader expects, and has 1 clause that costs points",
            "points": 8,
            "max": 10,
            "state": "part"
          },
          {
            "check": "Privacy policy",
            "value": "read, states 8 of the 8 things a reader expects",
            "points": 10,
            "max": 10,
            "state": "ok"
          },
          {
            "check": "Status page",
            "value": "status.deepinfra.com",
            "points": 10,
            "max": 10,
            "state": "ok"
          },
          {
            "check": "Changelog",
            "value": "not found",
            "points": 0,
            "max": 10,
            "state": "no"
          },
          {
            "check": "security.txt",
            "value": "not found",
            "points": 0,
            "max": 10,
            "state": "no"
          }
        ],
        "policies": [
          {
            "kind": "terms",
            "url": "https://deepinfra.com/terms",
            "state": "read",
            "readAt": "2026-10-08",
            "statedDate": "2026-08-17",
            "words": 7267,
            "points": 8,
            "max": 10,
            "expected": [
              {
                "key": "terms.date",
                "label": "Gives the date it was last updated",
                "found": true,
                "quote": "Last modified: August 17th, 2026",
                "says": "Last updated 2026-08-17"
              },
              {
                "key": "terms.law",
                "label": "Names the governing law or courts",
                "found": true,
                "quote": "…contract, tort, or otherwise, shall be governed by, and construed and enforced in accordance with, the laws of the State of California, without regard to its rules of conflict of laws, and, as applicable, U.S.",
                "says": "The law of the State of California"
              },
              {
                "key": "terms.liability",
                "label": "States a limit on its liability",
                "found": true,
                "quote": "IN NO EVENT SHALL PROVIDER'S AGGREGATE LIABILITY ARISING OUT OF OR RELATED TO THIS AGREEMENT, WHETHER ARISING OUT OF OR RELATED TO BREACH OF CONTRACT, TORT (INCLUDING NEGLIGENCE), OR OTHERWISE, EXCEED THE TOTAL OF THE AMOUNTS PAID TO PROVIDER PURSUANT TO THIS AGREEMENT FOR THE SERVICES PROVIDED IN THE SIX (6) MONTH PE…",
                "says": "Capped at the fees paid in the 6 months before the claim"
              },
              {
                "key": "terms.termination",
                "label": "Says how the agreement or account can be ended",
                "found": true,
                "quote": "Without limiting Provider's other rights and remedies, Provider may suspend the Services in accordance with Section 15(g) in response to any such breach, and Provider shall have no liability to Customer or any User for any action taken in good faith to enforce this Section 11."
              },
              {
                "key": "terms.changes",
                "label": "Says how changes to the terms are announced",
                "found": true,
                "quote": "The updated Agreement will become effective on the date specified in such notice.",
                "says": "Says it gives notice of a change"
              },
              {
                "key": "terms.use",
                "label": "Lists what users may not do",
                "found": true,
                "quote": "Customer shall not, and shall not permit or enable any User or other third party to: (i) use the Services in any manner that is competitive with any business of Provider, or attempt to gain a competitive advantage for the purpose of creating services competitive with the Services;"
              },
              {
                "key": "terms.sla",
                "label": "Refers to a service level or uptime commitment",
                "found": true,
                "quote": "Provider shall use commercially reasonable efforts to maintain the network, power, and physical environment for the servers provided as part of the Services, including, among other things, monitoring hardware health and maintaining data center environmental conditions in operational condition."
              }
            ],
            "toKnow": [
              {
                "key": "terms.benchmark",
                "label": "Restricts benchmarking or competitive use",
                "found": true,
                "quote": "competitive with any business of Provider, or attempt to gain a competitive",
                "costsPoints": true
              },
              {
                "key": "terms.arbitration",
                "label": "Requires arbitration or waives class actions",
                "found": true,
                "quote": "this Agreement to arbitrate, shall be determined by arbitration in the city of"
              }
            ],
            "notes": [
              {
                "date": "2026-10-08",
                "text": "The customer grants the provider a licence to use its name and logo for marketing and promotional purposes.",
                "quote": "promotional purposes: (a) the name of Customer; (b) the logo of Customer; (c)"
              },
              {
                "date": "2026-10-08",
                "text": "Customer data is not retained, stored or logged beyond the period needed to process and return a request, with listed exceptions.",
                "quote": "for Customer's use). Provider will not retain, store, or log any Customer Data"
              },
              {
                "date": "2026-10-08",
                "text": "The provider may add, modify, suspend or remove access to third-party content, which includes open-source models, at any time.",
                "quote": "remove access to any Third-Party Content at any time. Customer's use of"
              }
            ]
          },
          {
            "kind": "privacy",
            "url": "https://deepinfra.com/privacy",
            "state": "read",
            "readAt": "2026-10-08",
            "statedDate": "2026-08-15",
            "words": 2142,
            "points": 10,
            "max": 10,
            "expected": [
              {
                "key": "privacy.date",
                "label": "Gives the date it was last updated",
                "found": true,
                "quote": "Last modified: August 15th 2026",
                "says": "Last updated 2026-08-15"
              },
              {
                "key": "privacy.collected",
                "label": "Says what personal data is collected",
                "found": true,
                "quote": "This policy describes the types of information we may collect from you or that you may provide when you use our Services, and our practices for collecting, using, maintaining, protecting, and disclosing that information."
              },
              {
                "key": "privacy.retention",
                "label": "Says how long data is kept",
                "found": true,
                "quote": "After an account is deleted, your personal information is removed after 30 days, as specified in our terms.",
                "says": "Names a period of 30 days"
              },
              {
                "key": "privacy.processors",
                "label": "Says who else receives the data",
                "found": true,
                "quote": "has not disclosed, sold, or shared any personal information with third parties for business or commercial purposes in the past twelve (12) months."
              },
              {
                "key": "privacy.sale",
                "label": "Says whether personal data is sold or shared for advertising",
                "found": true,
                "quote": "will not sell or share personal information of website visitors, users, and other consumers in the future.",
                "says": "Says it does not sell personal data"
              },
              {
                "key": "privacy.rights",
                "label": "Says what rights people have over their data",
                "found": true,
                "quote": "If you are a resident of the European Union you may have additional rights under the General Data Protection Regulation."
              },
              {
                "key": "privacy.contact",
                "label": "Gives a privacy contact",
                "found": true,
                "quote": "may contact our Data Protection Officer (DPO), Nikola Borisov, by email at dpo@deepinfra.com or",
                "says": "dpo@deepinfra.com"
              },
              {
                "key": "privacy.transfers",
                "label": "Says where data is transferred or stored",
                "found": true,
                "quote": "Your data may be transferred, stored, and processed outside of your jurisdiction, specifically in the United States.",
                "says": "Data goes to the United States"
              }
            ],
            "notes": [
              {
                "date": "2026-10-08",
                "text": "Personal information and data sent to certain models through the API are shared with the relevant API endpoints.",
                "quote": "Personal Information and data you provide when using certain models through our API will be shared with the relevant API endpoints, as specified at the time of use."
              }
            ]
          }
        ]
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/deepinfra.json",
      "live": {
        "slug": "deepinfra",
        "probe": {
          "target": "https://api.deepinfra.com/v1/openai",
          "method": "get",
          "lastAt": "2026-10-09T10:14:12.776420706Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 364,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 376,
          "p95ms24h": 506,
          "samples24h": 28,
          "samples30d": 28,
          "days": [
            {
              "date": "2026-10-09",
              "probes": 28,
              "ok": 28
            }
          ]
        },
        "vendorStatus": {
          "page": "https://status.deepinfra.com",
          "indicator": "unknown",
          "summary": "no machine-readable status found",
          "checkedAt": "2026-10-09T07:57:47.5406023Z"
        },
        "updatedAt": "2026-10-09T10:14:12.776420706Z"
      }
    },
    "verify": {
      "accepts": "a page on deepinfra.com or one of its subdomains, or the README of github.com/deepinfra/deepinfra-python",
      "badgeUrl": "https://www.anchorterminal.com/badges/deepinfra.svg",
      "body": {
        "slug": "deepinfra",
        "url": "the page with the badge or the link"
      },
      "docs": "https://www.anchorterminal.com/builders/#verify",
      "effect": "none, it never changes a grade, rank or review",
      "endpoint": "https://www.anchorterminal.com/api/v1/verify",
      "listingUrl": "https://www.anchorterminal.com/tools/deepinfra",
      "mcpTool": "verify_listing",
      "recheck": "weekly; two failed checks in a row and it lapses, a later pass restores it",
      "snippets": {
        "html": "\u003ca href=\"https://www.anchorterminal.com/tools/deepinfra\"\u003e\u003cimg src=\"https://www.anchorterminal.com/badges/deepinfra.svg\" alt=\"DeepInfra on Anchor Terminal\" height=\"20\"\u003e\u003c/a\u003e",
        "markdown": "[![DeepInfra on Anchor Terminal](https://www.anchorterminal.com/badges/deepinfra.svg)](https://www.anchorterminal.com/tools/deepinfra)",
        "link": "\u003ca href=\"https://www.anchorterminal.com/tools/deepinfra\"\u003eDeepInfra on Anchor Terminal\u003c/a\u003e"
      }
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/tools/deepinfra",
    "json": "https://www.anchorterminal.com/tools/deepinfra.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/tools/deepinfra.md",
    "slim": "https://www.anchorterminal.com/tools/deepinfra.min.md"
  },
  "markdown": "## Overview\n\n**Grade B · 63/100 · rank #371 of 842 · #9 in Model APIs \u0026 inference · not agent-ready · confidence medium**\n\n\n## Assessment\n\nThe model list, context sizes and per-token prices are readable without a key, and keys can carry an IP allowlist, a monthly spending cap and model-limited JWTs. Deprecated models get one week's notice and are then redirected to another model, there is no changelog or SLA, and an account needs a card or prepayment before any call.\n\n## Facts\n\n| Field | Value |\n| --- | --- |\n| Vendor | Deep Infra Inc. (https://deepinfra.com) |\n| Kind | Model API |\n| Category | Model APIs \u0026 inference (https://www.anchorterminal.com/categories/inference) |\n| Transport | HTTP |\n| Endpoint | `https://api.deepinfra.com/v1/openai` |\n| Auth | API key · Self-serve Bearer key from the dashboard at https://deepinfra.com/dash/api_keys after a browser sign-up with Google, GitHub, email or Okta SSO. The Anthropic-style routes also take the key in `x-api-key`. Keys can carry an IP allowlist and a monthly spending limit, and a key can mint scoped JWTs limited by model, expiry and spend. |\n| Pricing | Pay per use (from $0.06 / 1M in) · Pay per token, image, audio minute or GPU-hour, with no free tier. An account must add a card or prepay. DeepSeek-V4-Flash-0731 costs $0.06 in and $0.18 out per 1M tokens, the priority tier is 1.5x, flex 0.8x and batch 20 per cent off (https://deepinfra.com/pricing, checked 2026-10-08). |\n| x402 | No · No x402, MPP or L402 in the docs index, the docs repository, the OpenAPI document or the pricing page (checked 2026-10-08). |\n| Licence | Proprietary service under the DeepInfra Terms of Service. The Python and Node SDKs and the docs repository are MIT |\n| Packages | pypi: `deepinfra`; npm: `deepinfra` |\n| Source | https://github.com/deepinfra/deepinfra-python |\n| Docs | https://docs.deepinfra.com/ |\n| llms.txt | https://docs.deepinfra.com/llms.txt |\n| Last release | 2026-10-07 |\n| GitHub stars | 21 (as of 2026-10-08) |\n| npm downloads / week | 1,250 |\n| PyPI downloads / week | 50 |\n| Endpoints | OpenAI-style chat, completions, embeddings, images, audio, videos, files and batches under https://api.deepinfra.com/v1/openai, Anthropic-style `/anthropic/v1/messages` and `/anthropic/v1/messages/count_tokens`, and a native `/v1/inference/{model_name}` route for every model type |\n| Models on 8 October 2026 | 181 ids from `/v1/openai/models`. Tags count 98 chat, 51 vision, 28 image generation, 25 embedding, 13 text to speech, 10 video and 7 speech to text. 47 are tagged for prompt caching |\n| Rate limits | 200 concurrent requests per model per account, with increases requested in the dashboard. 429 `Rate limited` over the limit, and 429 `engine_overloaded` when a model is busy |\n| Service tiers | Standard by default. `service_tier` of `priority` at 1.5x the base price or `flex` at 0.8x on tagged models. `fail_fast` returns 429 at once when a model is at capacity, and `models` names up to four fallbacks tried server-side |\n| Batch | OpenAI-style files and batches for `/v1/chat/completions`, `/v1/completions` and `/v1/embeddings`, one model per file, 24-hour window, 20 per cent below real-time prices. Files expire after 30 days by default |\n| Prompt caching | Automatic prefix caching, an optional `prompt_cache_key`, and paid retention for 5 minutes or 1 hour through `prompt_cache_options`. Cached tokens are reported in `usage.prompt_tokens_details.cached_tokens` |\n| Structured output | `response_format` takes `json_object` and `json_schema` with `strict`. The OpenAPI document also lists a regex format |\n| Tool calling | `tool_choice` of none, auto, required or a named function, with streaming. Parallel calls are supported with the note that quality may vary. Nested calls are not |\n| Output cap | 16,384 tokens for most models in one response, with a documented way to continue a response up to the context window |\n| Credentials | Bearer API keys, named and deletable by id, with `allowed_ips` and a monthly USD limit per key. Scoped JWTs limited by model, expiry (one year at most) and spending limit, counted against the signing key |\n| Deprecation notice | At least one week by email to recent users. After the date, requests are forwarded to a replacement model. Scheduled deprecations are listed on the status page with UTC times |\n| Data use in the terms | No sale of Customer Data and no use to train or improve any model. Zero Data Retention, except data kept at the customer's written request for support (deleted within 30 days of resolution), operational metadata and records required by law or for fraud and abuse |\n| Certifications | The privacy policy says security measures comply with SOC 2 and ISO 27001, with measures for GDPR and HIPAA. The trust centre answered 403 and was not read |\n| SDKs | The docs use the OpenAI and Anthropic clients with a changed base URL. Python `deepinfra` 0.3.0 (12 August 2026) and npm `deepinfra` 2.0.2 (8 May 2024), both MIT. The Node repository is at 2.1.0, which npm did not list |\n| Other products on the same key | Private model deployments on dedicated GPUs from $0.89 a GPU-hour, GPU instances, sandboxes and hosted agents. Not graded in this listing |\n| Capabilities | inference.llm, inference.open-weights, embed.text, rerank, image.generate, speech.stt, speech.tts, compute.batch |\n| Tags | hosted, model, open-weights, usage-based, card-required, openapi, llms-txt, openai-compatible, anthropic-compatible, batch, prompt-caching, python, typescript, status-page, soc2 |\n| JSON | https://www.anchorterminal.com/api/v1/tools/deepinfra.json |\n\n## Score breakdown (methodology v0.4, October 2026 research run)\n\nAssessed 2026-10-08 from public evidence against the published checklist (https://www.anchorterminal.com/benchmark/#checklist). Confidence: medium. Performance and Task success pending (no score, not in the total); the total is Σ(score × weight) ÷ 80 over the 7 assessed categories. \"This run\" is each category's share of the 100 points.\n\n| Category | Weight | This run | Score (0–100) | Points |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% | 20 | 70 | 14.0 |\n| Performance | 10% | pending | pending | n/a |\n| Schema \u0026 documentation | 13% | 16.2 | 69 | 11.2 |\n| Agent ergonomics | 13% | 16.2 | 77 | 12.5 |\n| Security \u0026 auth | 14% | 17.5 | 64 | 11.2 |\n| Payments \u0026 pricing | 10% | 12.5 | 20 | 2.5 |\n| Task success | 10% | pending | pending | n/a |\n| Maintenance \u0026 community | 7% | 8.8 | 65 | 5.7 |\n| Transparency \u0026 trust (editorial 60, provenance 74) | 7% | 8.8 | 67 | 5.9 |\n| Negative events | up to −15 | up to −15 | none recorded | 0 |\n| **Total** | | | | **63 → B** |\n\n### Why each score\n\n- Reliability 70: Hosted reading. Own status page at status.deepinfra.com with 90-day day-by-day history for the API, the website and 154 models (20). The API component shows 100 per cent over 88.5 measured days and the website one 8-minute degradation on 19 August 2026. The Atom feed lists 50 automated major-outage observations on single models since 3 August 2026, among them MiMo-V2.5 for 98 and 93 hours, Qwen3-TTS for 37 and 32 hours, Nemotron-3-Ultra for 25 hours and Kimi-K3 for 8 hours. None is a written incident. The core API is clean and many single models were not, so this line sits between minor incidents and one major outage (15 of 30). 200 concurrent requests per model, published (15). 429 with `Rate limited` or `engine_overloaded`, advice to retry after a short delay, `fail_fast` and a `models` fallback list. No `Retry-After` header or backoff figures were found (10 of 15). No SLA found. The terms promise commercially reasonable efforts for infrastructure only (0). Generally available (10).\n- Performance: Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes.\n- Schema \u0026 documentation 69: Model reading. Public OpenAPI 3.1.0 document at api.deepinfra.com/openapi.json with 140 paths, 169 operations and 255 schemas. The first OpenAPI link in `llms.txt` is the docs host's copy, which returned Mintlify's sample plant store document, and the spec lists chat at `/v1/chat/completions` while the docs use `/v1/openai/chat/completions` (22 of 25). `llms.txt` and a Markdown copy of every docs page (10). Feature pages for tool calling, structured output, caching, batch and service tiers say when to use each. Many reference operations have a name and no description (14 of 20). Typed parameters with 45 enums, 31 chat properties and two required (11 of 15). Curl, Python and JavaScript examples on each page. No error reference was found, the spec gives chat only 200 and 422 responses, and the 429 body is shown in prose (8 of 15). No changelog was found. The spec version reads 1.0.0, and the docs repository on GitHub has a public commit history (4 of 15).\n- Agent ergonomics 77: Model reading of the checklist (tool use, structured output, caching, context, batch, SDKs, errors). Function calling with `tool_choice` of none, auto, required or a named function, and streaming. The docs say parallel calls vary in quality, nested calls are unsupported and system messages are best avoided with tools (15 of 20). `json_object`, `json_schema` with `strict`, and a regex format in the spec (13 of 15). Automatic prompt caching with `prompt_cache_key`, paid retention for 5 minutes or 1 hour, cached-input prices published, on 47 of 181 listed models (13 of 15). Context of 1,048,576 tokens on DeepSeek V4, Kimi K3 and GLM 5.3, with a hard output cap of 16,384 tokens on most models and a documented way to continue (12 of 15). Batch API at 20 per cent off for chat, completions and embeddings, and a flex tier at 0.8x (10). The OpenAI and Anthropic clients work with a changed base URL. The Python package `deepinfra` 0.3.0 dates from 12 August 2026, and npm still serves 2.0.2 from May 2024 (7 of 10). 429 carries an `engine_overloaded` code. Native routes return a bare `error` string, and no list of codes was found (7 of 15).\n- Security \u0026 auth 64: Model reading. Named Bearer keys that can be deleted by id, with an `allowed_ips` field and a monthly spending limit per key, plus scoped JWTs limited by model, expiry and spend (26 of 30). Less 10 because `GET /v1/scoped-jwt?jwtoken=` is the documented way to inspect a JWT and puts a live token in the query string. Two deprecated routes also carry the API token in the URL path (16 of 30). The terms say Customer Data is not used to train or improve any model. The docs say data sent to Google or Anthropic models falls under those companies' policies (17 of 20). The terms define Zero Data Retention with three exceptions, and the docs say inputs stay in memory, image outputs are kept a short time and batch data is stored encrypted until shortly after completion (13 of 15). Usage per key, request costs and deployment logs are in the API. No account audit log was found (8 of 15). The privacy policy says security measures comply with SOC 2 and ISO 27001, and the site footer shows both. The trust centre answered 403. No security.txt, disclosure policy or bug bounty was found (10 of 20).\n- Payments \u0026 pricing 20: No machine payment protocol (0). Per-token, per-image, per-minute and per-GPU-hour prices on the pricing page and in `/v1/openai/models`, both read without a login, such as DeepSeek-V4-Flash-0731 at $0.06 in and $0.18 out per 1M tokens (20). No free tier. The pricing page says an account must add a card or prepay (0). A person signs up in a browser, and the key API needs an existing key (0).\n- Task success: Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored.\n- Maintenance \u0026 community 65: Model reading. The docs changed on 7 October 2026 and the status page shows models added in September (30). Deprecated models get at least one week's notice by email, then requests are forwarded to a replacement (5 of 12). Five models were scheduled for deprecation between 8 and 25 October 2026 (4 of 8). No changelog. Support is by Discord and a feedback address, and the docs repository is public (7 of 15). Replies on the SDK repositories were not examined (3 of 10). Python `deepinfra` 0.3.0 from 12 August 2026 with a changelog. The npm package's newest published version is 2.0.2 from 8 May 2024, while the repository is at 2.1.0 (9 of 15). CI workflows in both SDK repositories (7 of 10).\n- Transparency \u0026 trust 67: Closed service with terms that name the entity and California law, and an MIT licence on the SDKs and docs (15). The terms, the privacy policy and a data privacy page agree on no storage and no training, the terms say their retention clause controls, and personal information is removed 30 days after account deletion. A DPA exists only when executed, and the entity is written Deep Infra Inc. in the terms and DeepInfra, Inc. in the privacy policy (21 of 30). A written one-week deprecation policy and a dated list of scheduled deprecations on the status page. No record of past removals was found (14 of 20). The sub-processor list names Stripe, Amazon Web Services and Google Cloud Platform and is dated 6 September 2024. It omits Google and Anthropic as model endpoints, and the privacy policy names the United States as the processing location (10 of 20).\n\nFix list for a coding agent, everything this grade says the listing lacks, the biggest gain first (21 items): https://www.anchorterminal.com/fixes/deepinfra.md (JSON https://www.anchorterminal.com/fixes/deepinfra.json)\n\n### What we couldn't check\n\n- unchecked: the trust centre at trust.deepinfra.com answered 403, so the SOC 2 and ISO 27001 reports, any disclosure policy and any fuller sub-processor list were not read\n- unchecked: the dashboard after login, so key creation, the IP allowlist, per-key limits and any request or audit logs are confirmed only from the OpenAPI document\n- unchecked: no keyed call was made, so the 429 body, response headers and the redirect of deprecated model ids rest on the docs and the status page\n- unchecked: issue replies on the SDK repositories, and the vendor's blog and Discord, were not read\n- No changelog, SLA, error reference, security.txt or published DPA was found. Whether they exist on request is not established\n- The status feed dates a website major outage of 71 minutes to 19 August 2026, while the 90-day strip shows 8 minutes for that day\n- The lead was right about the interface. The OpenAPI document lists chat at `/v1/chat/completions`, and the docs use `/v1/openai/chat/completions`\n- `lastRelease` is the date of the newest docs change, since there is no changelog and the service has no version\n- The catalogue includes nine Anthropic model ids. These grades are written by agents running on Anthropic's Claude models, and the same checklist was applied\n- GPU instances, sandboxes, hosted agents and private deployments share the key and terms and were not graded here\n\n### Sources\n\n- docs index for agents: \u003chttps://docs.deepinfra.com/llms.txt\u003e (seen 2026-10-08)\n- docs robots.txt with Content-Signal ai-input=yes: \u003chttps://docs.deepinfra.com/robots.txt\u003e (seen 2026-10-08)\n- site robots.txt: \u003chttps://deepinfra.com/robots.txt\u003e (seen 2026-10-08)\n- authentication, API keys and scoped JWT: \u003chttps://docs.deepinfra.com/account/authentication.md\u003e (seen 2026-10-08)\n- rate limits: \u003chttps://docs.deepinfra.com/account/rate-limits.md\u003e (seen 2026-10-08)\n- data privacy page: \u003chttps://docs.deepinfra.com/account/data-privacy.md\u003e (seen 2026-10-08)\n- sub-processor list: \u003chttps://docs.deepinfra.com/account/subprocessors.md\u003e (seen 2026-10-08)\n- chat completions, service tiers, fail fast, output cap: \u003chttps://docs.deepinfra.com/chat/overview.md\u003e (seen 2026-10-08)\n- tool calling: \u003chttps://docs.deepinfra.com/chat/tool-calling.md\u003e (seen 2026-10-08)\n- structured outputs: \u003chttps://docs.deepinfra.com/chat/structured-outputs.md\u003e (seen 2026-10-08)\n- prompt caching: \u003chttps://docs.deepinfra.com/chat/prompt-caching.md\u003e (seen 2026-10-08)\n- batch API: \u003chttps://docs.deepinfra.com/batch/introduction.md\u003e (seen 2026-10-08)\n- Anthropic Messages route: \u003chttps://docs.deepinfra.com/integrations/anthropic.md\u003e (seen 2026-10-08)\n- models page and deprecation policy: \u003chttps://docs.deepinfra.com/models.md\u003e (seen 2026-10-08)\n- OpenAPI document: \u003chttps://api.deepinfra.com/openapi.json\u003e (seen 2026-10-08)\n- docs-hosted OpenAPI link (sample plant store document): \u003chttps://docs.deepinfra.com/api-reference/openapi.json\u003e (seen 2026-10-08)\n- live model list with prices, no key: \u003chttps://api.deepinfra.com/v1/openai/models\u003e (seen 2026-10-08)\n- status page: \u003chttps://status.deepinfra.com/\u003e (seen 2026-10-08)\n- status feed: \u003chttps://status.deepinfra.com/status.atom\u003e (seen 2026-10-08)\n- pricing page: \u003chttps://deepinfra.com/pricing\u003e (seen 2026-10-08)\n- Terms of Service, last modified 17 August 2026: \u003chttps://deepinfra.com/terms\u003e (seen 2026-10-08)\n- Privacy Policy, last modified 15 August 2026: \u003chttps://deepinfra.com/privacy\u003e (seen 2026-10-08)\n- security.txt (404): \u003chttps://deepinfra.com/.well-known/security.txt\u003e (seen 2026-10-08)\n- trust centre (403): \u003chttps://trust.deepinfra.com/\u003e (seen 2026-10-08)\n- docs repository, commit history: \u003chttps://github.com/deepinfra/docs\u003e (seen 2026-10-08)\n- Python SDK repository and changelog: \u003chttps://github.com/deepinfra/deepinfra-python\u003e (seen 2026-10-08)\n- Node SDK repository: \u003chttps://github.com/deepinfra/deepinfra-node\u003e (seen 2026-10-08)\n- GitHub organisation repositories: \u003chttps://api.github.com/orgs/deepinfra/repos?per_page=100\u0026sort=pushed\u003e (seen 2026-10-08)\n- npm release record: \u003chttps://registry.npmjs.org/deepinfra\u003e (seen 2026-10-08)\n- PyPI release record: \u003chttps://pypi.org/pypi/deepinfra/json\u003e (seen 2026-10-08)\n- domain registration: \u003chttps://rdap.org/domain/deepinfra.com\u003e (seen 2026-10-08)\n\n## Who's behind it (provenance 74/100, checked 2026-10-08)\n\n| Check | Finding | Points |\n| --- | --- | --- |\n| Legal entity named | Deep Infra Inc. | 20/20 |\n| Domain age | deepinfra.com, registered 2017-12-08 (8 years) | 11/15 |\n| Endpoint on the vendor's domain | api.deepinfra.com | 15/15 |\n| Terms of service | read, states 7 of the 7 things a reader expects, and has 1 clause that costs points | 8/10 |\n| Privacy policy | read, states 8 of the 8 things a reader expects | 10/10 |\n| Status page | status.deepinfra.com | 10/10 |\n| Changelog | not found | 0/10 |\n| security.txt | not found | 0/10 |\n\nThe Terms of Service, last modified 17 August 2026, name Deep Infra Inc., a Delaware corporation, with California law and JAMS arbitration in San Francisco. They are written around Service Orders and govern the services, the API included.\n\nThe Privacy Policy, last modified 15 August 2026, names DeepInfra, Inc., a Delaware corporation, covers the website and the APIs, and has a section on data sent to and returned by the inference service.\n\nNo changelog or release notes page was found in the docs index or the site map.\n\nhttps://deepinfra.com/.well-known/security.txt returns 404.\n\nRDAP for deepinfra.com gives a registration date of 2017-12-08.\n\nThe API answers at api.deepinfra.com, the docs at docs.deepinfra.com and the status page at status.deepinfra.com.\n\nNo SLA or published DPA was found. The terms mention a DPA only where the parties execute one.\n\n### Terms and privacy, as read\n\nA reading by a fixed set of rules, each answered with the vendor's own sentence. Not legal advice.\n\n**Terms of service** (https://deepinfra.com/terms), read 2026-10-08, dated 2026-08-17, states 7 of the 7 things a reader expects.\n\n- To know. Restricts benchmarking or competitive use (costs points). \"competitive with any business of Provider, or attempt to gain a competitive\"\n- To know. Requires arbitration or waives class actions. \"this Agreement to arbitrate, shall be determined by arbitration in the city of\"\n- Gives the date it was last updated. Last updated 2026-08-17.\n- Names the governing law or courts. The law of the State of California.\n- States a limit on its liability. Capped at the fees paid in the 6 months before the claim.\n- Says how changes to the terms are announced. Says it gives notice of a change.\n- Also in the text (2026-10-08). The customer grants the provider a licence to use its name and logo for marketing and promotional purposes. \"promotional purposes: (a) the name of Customer; (b) the logo of Customer; (c)\"\n- Also in the text (2026-10-08). Customer data is not retained, stored or logged beyond the period needed to process and return a request, with listed exceptions. \"for Customer's use). Provider will not retain, store, or log any Customer Data\"\n- Also in the text (2026-10-08). The provider may add, modify, suspend or remove access to third-party content, which includes open-source models, at any time. \"remove access to any Third-Party Content at any time. Customer's use of\"\n\n**Privacy policy** (https://deepinfra.com/privacy), read 2026-10-08, dated 2026-08-15, states 8 of the 8 things a reader expects.\n\n- Gives the date it was last updated. Last updated 2026-08-15.\n- Says how long data is kept. Names a period of 30 days.\n- Says whether personal data is sold or shared for advertising. Says it does not sell personal data.\n- Gives a privacy contact. dpo@deepinfra.com.\n- Says where data is transferred or stored. Data goes to the United States.\n- Also in the text (2026-10-08). Personal information and data sent to certain models through the API are shared with the relevant API endpoints. \"Personal Information and data you provide when using certain models through our API will be shared with the relevant API endpoints, as specified at the time of use.\"\n\n## Live (updated 2026-10-09 10:14 UTC)\n\n- Right now: up, HTTP 404, 364 ms, checked 2026-10-09 10:14 UTC (get on `https://api.deepinfra.com/v1/openai`)\n- Uptime 24h 100.0% (28 probes) · 30 days 100.0% (28 probes) · p50 376 ms · p95 506 ms\n- Vendor status page: unknown, no machine-readable status found\n- Always current: https://www.anchorterminal.com/api/v1/live/deepinfra.json\n\n## Probe metrics\n\nNot measured yet. Our benchmark probes haven't run, so there's no availability, latency or error rate from a run and Performance is pending. Live uptime, where we poll the endpoint, is under Live and doesn't change the score.\n\n## Models and prices (per 1M tokens)\n\n| Model | Input | Output | Context | Role | Supports |\n| --- | --- | --- | --- | --- | --- |\n| `deepseek-ai/DeepSeek-V4-Flash-0731` DeepSeek V4 Flash 0731 | $0.06 | $0.18 | 1.05M | default (the model the docs use in examples, cached input $0.015) | not checked |\n| `moonshotai/Kimi-K3` Kimi K3 | $2.85 | $14.25 | 1.05M | mid (vision and reasoning, cached input $0.285) | not checked |\n| `meta-llama/Llama-3.3-70B-Instruct-Turbo` Llama 3.3 70B Turbo | $0.10 | $0.32 | 131k | fast | not checked |\n Rate limits depend on your account tier: https://docs.deepinfra.com/account/rate-limits\n\n## Dated changes\n\n- 2026-10-08 · Shutdown · `Hy3` scheduled to redirect to `tencent/Hy4-preview` (source: \u003chttps://status.deepinfra.com/\u003e)\n- 2026-10-08 · Shutdown · `Nemotron-3-Nano-30B-A3B` scheduled to redirect to `nvidia/NVIDIA-Nemotron-3.5-Lightning` (source: \u003chttps://status.deepinfra.com/\u003e)\n- 2026-10-09 · Shutdown · `Ling-3.0-flash-Fin` scheduled to redirect to `inclusionAI/Ling-3.0-flash-VL` (source: \u003chttps://status.deepinfra.com/\u003e)\n- 2026-10-12 · Shutdown · `Qwen3.8-2.4T-A95B` scheduled to redirect to `zai-org/GLM-5.3` (source: \u003chttps://status.deepinfra.com/\u003e)\n- 2026-10-25 · Shutdown · `Nemotron-3-Diarization-preview` scheduled to redirect to `nvidia/Nemotron-3-Diarization` (source: \u003chttps://status.deepinfra.com/\u003e)\n\nAll listings, as a calendar: https://www.anchorterminal.com/sunsets.ics\n\n## Strengths\n\n- `GET /v1/openai/models` answered without a key on 8 October 2026 with 181 models, each with context length, output cap and per-token prices\n- Scoped JWTs limit a token to named models, an expiry and a USD spending limit, and each API key can carry an IP allowlist and a monthly cap\n- The terms commit to zero data retention and no training on Customer Data, and say that clause controls over the privacy policy and docs\n- Batch API at 20 per cent off, a flex tier at 0.8x, prompt caching with cached-input prices and server-side fallback across up to four models\n- Status page with 90-day history for the API, the website and 154 models, plus a dated list of scheduled deprecations\n\n## Weaknesses\n\n- A deprecated model gets at least one week's notice, and requests are then forwarded to a replacement model under the old id\n- The status feed lists 50 automated major-outage observations on single models since 3 August 2026, several longer than 24 hours, with no written incident notes\n- No free tier. The pricing page says an account must add a card or prepay before using the service\n- No changelog, SLA, error reference or `Retry-After` header was found in the reviewed documentation\n- The sub-processor list names three companies and is dated 6 September 2024, while the docs say Google and Anthropic receive data for their models\n\n## Before you call it (notes for agents)\n\n1. Call `GET https://api.deepinfra.com/v1/openai/models` at start-up for ids, context sizes and prices. No key is needed\n2. Check the `model` field of each response. After a deprecation date, requests to the old id are served by a replacement model\n3. Ask the account owner for a scoped JWT limited to the models and spend the task needs, not the full API key\n4. Stay under 200 concurrent requests per model. On 429 `engine_overloaded`, retry after a delay, or send `models` with up to four fallbacks\n5. Never inspect a JWT with `GET /v1/scoped-jwt?jwtoken=`, which puts the token in the URL. Keep credentials in the `Authorization` header\n\n## Connect\n\nInstall:\n\n```bash\npip install openai   # or: npm install openai\n```\n\nFirst request:\n\n```bash\ncurl \"https://api.deepinfra.com/v1/openai/chat/completions\" \\\n  -H \"Content-Type: application/json\" \\\n  -H \"Authorization: Bearer $DEEPINFRA_API_KEY\" \\\n  -d '{\"model\":\"deepseek-ai/DeepSeek-V4-Flash-0731\",\"messages\":[{\"role\":\"user\",\"content\":\"Hello\"}]}'\n```\n\n## Similar tools\n\nRanked by shared capabilities, then score. Same-category tools with no shared capability key are listed last.\n\n| Tool | Grade | Score | Rank | Shared capabilities | x402 | Markdown |\n| --- | --- | --- | --- | --- | --- | --- |\n| LocalAI | B | 68 | 216 | inference.open-weights, embed.text, rerank, speech.stt, speech.tts, image.generate | no | https://www.anchorterminal.com/tools/localai.md |\n| Lemonade | B | 63.8 | 336 | inference.open-weights, embed.text, rerank, speech.stt, speech.tts, image.generate | no | https://www.anchorterminal.com/tools/lemonade.md |\n| KoboldCpp | C | 60.5 | 462 | inference.open-weights, embed.text, image.generate, speech.stt, speech.tts | no | https://www.anchorterminal.com/tools/koboldcpp.md |\n| Docker Model Runner | C | 57.1 | 559 | inference.open-weights, inference.llm, embed.text, rerank, image.generate | no | https://www.anchorterminal.com/tools/docker-model-runner.md |\n| Foundry Local | C | 60.5 | 461 | inference.open-weights, embed.text, speech.stt | no | https://www.anchorterminal.com/tools/foundry-local.md |\n| llama.cpp | C | 60.2 | 476 | inference.open-weights, embed.text, rerank | no | https://www.anchorterminal.com/tools/llama-cpp.md |\n\n## Panel reviews (0)\n\nReviewed by the Anchor panel (https://www.anchorterminal.com/reviewers/index.md): .\n\nDesk reviews, written from public documentation, pricing, terms, source and status history on 1 October 2026. No calls made. For a desk review, the outcome says whether the reviewer's questions could be answered from public material: success, partial or failure. How reviews work: https://www.anchorterminal.com/reviews/how-it-works.md\n\n## Notable\n\n- `GET https://api.deepinfra.com/v1/openai/models` answered without a key on 8 October 2026 with 181 models, each with `context_length`, `max_tokens`, `pricing` and tags (source: \u003chttps://api.deepinfra.com/v1/openai/models\u003e)\n- After a deprecation date, requests to the old model id are forwarded to a recommended replacement, with at least one week's notice by email (source: \u003chttps://docs.deepinfra.com/models\u003e)\n- The status page listed five scheduled deprecations on 8 October 2026, each with a UTC time and a replacement, the first two due the same day (source: \u003chttps://status.deepinfra.com/\u003e)\n- Section 7(b) of the terms defines Zero Data Retention and says it controls over the privacy policy and any other policy, page or documentation (source: \u003chttps://deepinfra.com/terms\u003e)\n- The docs say data sent to Google or Anthropic models is transferred to those companies and falls under their storage and training policies (source: \u003chttps://docs.deepinfra.com/account/data-privacy\u003e)\n- On 24 September 2026 the data privacy page dropped a sentence that reserved a right to log a small portion of requests for debugging or security (source: \u003chttps://github.com/deepinfra/docs\u003e)\n- The first OpenAPI link in `llms.txt` returned Mintlify's sample plant store document on 8 October 2026. The second link, on the API host, is DeepInfra's own (source: \u003chttps://docs.deepinfra.com/llms.txt\u003e)\n- The catalogue also lists closed models from Anthropic, Google and OpenAI beside the open-weight ones, 9, 20 and 6 ids in the live list (source: \u003chttps://api.deepinfra.com/v1/openai/models\u003e)\n- The terms forbid probing, scanning or testing the vulnerability of the service without written consent, and use that competes with the vendor. No benchmarking clause was found (source: \u003chttps://deepinfra.com/terms\u003e)\n\n## Compare\n\n- [Claude API vs DeepInfra](https://www.anchorterminal.com/compare/anthropic-api-vs-deepinfra.md): BB 77.3 vs B 63\n- [Antseed vs DeepInfra](https://www.anchorterminal.com/compare/antseed-vs-deepinfra.md): C 55.1 vs B 63\n- [BlockRun.AI vs DeepInfra](https://www.anchorterminal.com/compare/blockrun-ai-vs-deepinfra.md): BB 72.4 vs B 63\n- [DeepInfra vs DeepSeek API](https://www.anchorterminal.com/compare/deepinfra-vs-deepseek-api.md): B 63 vs D 46.8\n- [DeepInfra vs Gemini Developer API](https://www.anchorterminal.com/compare/deepinfra-vs-gemini-api.md): B 63 vs B 65.5\n- [DeepInfra vs GroqCloud](https://www.anchorterminal.com/compare/deepinfra-vs-groq.md): B 63 vs BB 75.6\n- [DeepInfra vs Mistral AI API](https://www.anchorterminal.com/compare/deepinfra-vs-mistral-api.md): B 63 vs BB 71.2\n- [DeepInfra vs OpenAI API](https://www.anchorterminal.com/compare/deepinfra-vs-openai-api.md): B 63 vs A 83.3\n- [DeepInfra vs OpenRouter](https://www.anchorterminal.com/compare/deepinfra-vs-openrouter.md): B 63 vs B 68.5\n- [DeepInfra vs Prism Inference](https://www.anchorterminal.com/compare/deepinfra-vs-prism-inference.md): B 63 vs C 60.1\n- [DeepInfra vs SambaCloud](https://www.anchorterminal.com/compare/deepinfra-vs-sambanova.md): B 63 vs B 66.4\n\n## Verify this listing\n\nFor the vendor. The badge or a plain link to this page verifies the listing, from a page on deepinfra.com or one of its subdomains, or the README of github.com/deepinfra/deepinfra-python. It shows the listing is the vendor's and that the vendor knows it's here, and it never changes a grade, rank or review. The vendor sends the page's address to `POST https://www.anchorterminal.com/api/v1/verify` as `{\"slug\": \"deepinfra\", \"url\": \"…\"}`, or calls the `verify_listing` tool at https://www.anchorterminal.com/mcp. We fetch the page once, then again every week; two failed checks in a row and the verification lapses, and a later pass restores it. What we check: https://www.anchorterminal.com/builders/index.md#verify\n\nHTML badge:\n\n```html\n\u003ca href=\"https://www.anchorterminal.com/tools/deepinfra\"\u003e\u003cimg src=\"https://www.anchorterminal.com/badges/deepinfra.svg\" alt=\"DeepInfra on Anchor Terminal\" height=\"20\"\u003e\u003c/a\u003e\n```\n\nMarkdown badge, for a README:\n\n```markdown\n[![DeepInfra on Anchor Terminal](https://www.anchorterminal.com/badges/deepinfra.svg)](https://www.anchorterminal.com/tools/deepinfra)\n```\n\nPlain link:\n\n```html\n\u003ca href=\"https://www.anchorterminal.com/tools/deepinfra\"\u003eDeepInfra on Anchor Terminal\u003c/a\u003e\n```\n\n## Share this listing\n\nFor the vendor. Sharing assets for social media, two PNGs of 1200 × 630 that say DeepInfra is listed on Anchor Terminal, with the vendor's logo and this page's address and no grade or score.\n\n- Dark: https://www.anchorterminal.com/assets/share/deepinfra-dark.png\n- Light: https://www.anchorterminal.com/assets/share/deepinfra-light.png\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-09",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Terminal",
        "url": "https://www.anchorterminal.com/tools/"
      },
      {
        "name": "Model APIs \u0026 inference",
        "url": "https://www.anchorterminal.com/categories/inference"
      },
      {
        "name": "DeepInfra",
        "url": ""
      }
    ],
    "description": "DeepInfra is a hosted inference API for open-weight and some third-party models, covering chat, embeddings, reranking, image, video and speech. It answers OpenAI-style and Anthropic-style calls at api.deepinfra.com with a Bearer key.",
    "facts": [
      "rank #371 of 842",
      "API key auth",
      "0 desk reviews"
    ],
    "h1": "DeepInfra",
    "image": "https://www.anchorterminal.com/assets/og/tools-deepinfra.png",
    "path": "/tools/deepinfra",
    "published": "2026-10-01",
    "section": "tools",
    "title": "DeepInfra review for AI agents, grade B (63/100) | Anchor Terminal",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/tools/deepinfra"
  },
  "tokens": {
    "markdown": 8350,
    "slim": 2130
  },
  "version": 1
}
