{
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-09",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "tool": {
    "slug": "hugging-face-inference-endpoints",
    "name": "Hugging Face Inference Endpoints",
    "vendor": "Hugging Face, Inc.",
    "vendorUrl": "https://huggingface.co",
    "kind": "http-api",
    "category": "gpu-compute",
    "summary": "Managed Hugging Face service that deploys a Hub model as a dedicated, autoscaling HTTPS endpoint on AWS, Azure or Google Cloud, using vLLM, TGI, SGLang, llama.cpp, TEI or a custom container. Managed by REST API, Python client, CLI or MCP.",
    "url": "https://www.anchorterminal.com/tools/hugging-face-inference-endpoints",
    "markdownUrl": "https://www.anchorterminal.com/tools/hugging-face-inference-endpoints.md",
    "slimMarkdownUrl": "https://www.anchorterminal.com/tools/hugging-face-inference-endpoints.min.md",
    "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/hugging-face-inference-endpoints.json",
    "repo": "https://github.com/huggingface/hf-endpoints-documentation",
    "license": "Proprietary service under the Hugging Face Terms of Service. The `huggingface_hub` Python client and CLI are Apache-2.0",
    "transports": [
      "http"
    ],
    "remoteUrl": "https://api.endpoints.huggingface.cloud",
    "packages": [
      {
        "registry": "pypi",
        "name": "huggingface_hub"
      }
    ],
    "auth": "mixed",
    "authNotes": "Hugging Face access token sent as `Authorization: Bearer $HF_TOKEN` to the management API and to each endpoint. Tokens are created in the account settings in a browser and can be fine-grained, read or write. The MCP server uses OAuth through huggingface.co (authorisation code with PKCE, device code, dynamic client registration) with `read-endpoints` and `write-endpoints` scopes. `GET /v2/provider` and the catalogue list need no token. Access is self-serve, with quota requests for larger instances.",
    "pricing": "usage",
    "pricingNotes": "Usage priced by instance hour, billed per minute while a replica is initialising or running. GPUs run from $0.50 an hour (T4) to $10 (H100 on GCP), CPUs from $0.033. No free tier or trial was found. The docs require a payment method and credits before deploying, and the pricing page says an active subscription. Paused endpoints and endpoints at zero replicas aren't billed for compute (https://huggingface.co/docs/inference-endpoints/support/pricing).",
    "priceSummary": "$0.033 / vCPU-hr",
    "where": "hosted",
    "x402": {
      "level": "no",
      "evidence": "No x402, MPP or L402 in the Inference Endpoints docs, the two OpenAPI documents or the pricing page (checked 2026-10-08).",
      "endpoints": []
    },
    "toolCount": 19,
    "popularity": {
      "githubStars": null,
      "npmWeekly": null,
      "pypiWeekly": 60014944,
      "asOf": "2026-10-08"
    },
    "docsUrl": "https://huggingface.co/docs/inference-endpoints/index",
    "llmsTxt": "https://huggingface.co/docs/inference-endpoints/llms.txt",
    "openapi": "https://api.endpoints.huggingface.cloud/openapi.json",
    "capabilities": [
      "compute.gpu",
      "compute.endpoints",
      "compute.containers"
    ],
    "tags": [
      "hosted",
      "usage-priced",
      "python",
      "cli",
      "mcp",
      "oauth",
      "openapi",
      "llms-txt",
      "status-page",
      "soc2",
      "enterprise"
    ],
    "lastRelease": "2026-10-08",
    "graded": true,
    "anchor": {
      "graded": true,
      "score": 64.5,
      "grade": "B",
      "agentReady": false,
      "rank": 314,
      "ranked": true,
      "rankOf": 842,
      "categoryRank": 3,
      "methodology": "0.4",
      "run": "2026-10-01",
      "scores": {
        "ergonomics": 62,
        "maintenance": 80,
        "payments": 20,
        "reliability": 63,
        "schema": 73,
        "security": 83,
        "transparency": 68
      },
      "pending": [
        "performance",
        "tasks"
      ],
      "breakdown": [
        {
          "key": "reliability",
          "name": "Reliability",
          "weight": 16,
          "effectiveWeight": 20,
          "score": 63,
          "points": 12.6,
          "reason": "Better Stack status page at status.huggingface.co with separate Inference Endpoints UI and API components and 90 days of history (20). The API component shows 100 per cent uptime and three scheduled maintenance windows (22 and 30 July, 7 October 2026). The UI was down for 1 hour 37 minutes on 16 July 2026 during a Hub outage, which we count as minor for the API surface (20). No rate limits were found for the management API. The Hub publishes limits per 5-minute window (1,000 API requests for a free user), without saying they cover api.endpoints.huggingface.cloud (5). The docs explain the 503 returned while a replica starts and the `X-Scale-Up-Timeout` header that holds the request, and advise 2 replicas for availability. No backoff guidance or idempotency keys for writes (8). No SLA found in the docs or on the pricing page (0). The service is generally available (10)."
        },
        {
          "key": "performance",
          "name": "Performance",
          "weight": 10,
          "effectiveWeight": 0,
          "pending": true,
          "points": 0,
          "reason": "Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes."
        },
        {
          "key": "schema",
          "name": "Schema \u0026 documentation",
          "weight": 13,
          "effectiveWeight": 16.25,
          "score": 73,
          "points": 11.86,
          "reason": "Public OpenAPI 3.1 documents for the management API (40 paths, 46 operations) and the catalogue API (3 operations) (25). llms.txt with a Markdown twin of every docs page (10). Every operation has a summary tagged READ, WRITE, PUBLIC or PRO, but only 5 of 46 have a description. The MCP tool table states each tool's purpose and says to call `get_recommended_config` before `create_endpoint` (11). 127 typed schemas with required fields and enums for state, type and accelerator. List filters are comma-separated strings (12). Parameters carry examples. The management document lists only 200 responses, plus 501 on three log routes, while the catalogue document lists 400, 401, 404, 409 and 500 (7). Paths are versioned `/v2` and `/v3` and the document is version 2.0.0. Inference Endpoints has no changelog of its own. The Hub changelog and the docs repository history are the dated record (8)."
        },
        {
          "key": "ergonomics",
          "name": "Agent ergonomics",
          "weight": 13,
          "effectiveWeight": 16.25,
          "score": 62,
          "points": 10.07,
          "reason": "List endpoints takes `limit` (default 20) and `cursor`, logs take `limit`, `tail` and `line_max_length`, and the MCP server has 19 tools (18). Cursor pagination, filters by state, type, task and tags, sorting, and a time window on logs (18). Errors are documented for the catalogue API and in prose for pause, resume and scale to zero. The management document has no error schema (8). No idempotency keys. Endpoints are addressed by name, updates are a PUT, and the MCP delete tool previews before a confirmed second call. MCP annotations weren't read (8). One-call deploy from the catalogue with tuned defaults, a Python client and the `hf endpoints` CLI. A second-language SDK for endpoint management wasn't found (10)."
        },
        {
          "key": "security",
          "name": "Security \u0026 auth",
          "weight": 14,
          "effectiveWeight": 17.5,
          "score": 83,
          "points": 14.53,
          "reason": "OAuth through huggingface.co with `read-endpoints` and `write-endpoints` scopes, PKCE and dynamic client registration for the MCP server, and revocable fine-grained access tokens for the API. No documented option to send a token in a URL (30). Read and write scopes are separate, endpoints are private by default, organisations on Team and Enterprise plans can require approval of fine-grained tokens, and the MCP delete tool needs `confirm: true`. Other write calls have no confirmation step (17). The service returns the output and logs of the owner's own model, not third-party content (10). An audit log route and MCP tool on Pro, Team or Enterprise plans, plus logs and metrics for every endpoint (12). security.txt valid to 2030 with security@huggingface.co, SOC2 Type 2 stated for the Hub and Inference Endpoints, malware and pickle scanning of repositories. No bug bounty was found (14)."
        },
        {
          "key": "payments",
          "name": "Payments \u0026 pricing",
          "weight": 10,
          "effectiveWeight": 12.5,
          "score": 20,
          "points": 2.5,
          "reason": "No machine payment protocol (0). Hourly prices for every instance published without a login and returned by the unauthenticated `/v2/provider` route (20). No free tier or trial. The docs require a payment method and credits (0). Signup, billing and token creation are browser steps (0)."
        },
        {
          "key": "tasks",
          "name": "Task success",
          "weight": 10,
          "effectiveWeight": 0,
          "pending": true,
          "points": 0,
          "reason": "Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored."
        },
        {
          "key": "maintenance",
          "name": "Maintenance \u0026 community",
          "weight": 7,
          "effectiveWeight": 8.75,
          "score": 80,
          "points": 7,
          "reason": "`huggingface_hub` 2.2.0 was published on PyPI on 8 October 2026 and the docs repository changed on 7 October (30). Releases 1.31.0 to 2.2.0 in the 30 days before, about one a week, and 8 docs commits since 10 July 2026 (20). The GitHub API refused our request, so issue response times went unread. A public changelog for the Hub, a forum and a quota contact form exist (8 of 15). Current official Python client and CLI (15). The client supports Python 3.10 and later and ships weekly. Its CI wasn't read (7)."
        },
        {
          "key": "transparency",
          "name": "Transparency \u0026 trust",
          "weight": 7,
          "effectiveWeight": 8.75,
          "score": 68,
          "points": 5.95,
          "note": "editorial 68, provenance 67",
          "reason": "Closed service under clear terms that name Inference Endpoints, with an Apache-2.0 client (20). The security page says payloads and tokens aren't stored and logs are kept 30 days. The privacy policy dates from 28 March 2023 and gives retention as long as necessary, and a data processing agreement requires an Enterprise subscription (20). Instance deprecations are dated to the month in the pricing table and flagged in the provider API, and the TGI maintenance notice is dated. No deprecation policy for the API was found (10). The privacy policy lists 11 subprocessors with countries and the docs and API list deployment regions (18)."
        }
      ],
      "assessment": {
        "date": "2026-10-08",
        "basis": "public evidence",
        "confidence": "medium",
        "notes": {
          "ergonomics": "List endpoints takes `limit` (default 20) and `cursor`, logs take `limit`, `tail` and `line_max_length`, and the MCP server has 19 tools (18). Cursor pagination, filters by state, type, task and tags, sorting, and a time window on logs (18). Errors are documented for the catalogue API and in prose for pause, resume and scale to zero. The management document has no error schema (8). No idempotency keys. Endpoints are addressed by name, updates are a PUT, and the MCP delete tool previews before a confirmed second call. MCP annotations weren't read (8). One-call deploy from the catalogue with tuned defaults, a Python client and the `hf endpoints` CLI. A second-language SDK for endpoint management wasn't found (10).",
          "maintenance": "`huggingface_hub` 2.2.0 was published on PyPI on 8 October 2026 and the docs repository changed on 7 October (30). Releases 1.31.0 to 2.2.0 in the 30 days before, about one a week, and 8 docs commits since 10 July 2026 (20). The GitHub API refused our request, so issue response times went unread. A public changelog for the Hub, a forum and a quota contact form exist (8 of 15). Current official Python client and CLI (15). The client supports Python 3.10 and later and ships weekly. Its CI wasn't read (7).",
          "payments": "No machine payment protocol (0). Hourly prices for every instance published without a login and returned by the unauthenticated `/v2/provider` route (20). No free tier or trial. The docs require a payment method and credits (0). Signup, billing and token creation are browser steps (0).",
          "reliability": "Better Stack status page at status.huggingface.co with separate Inference Endpoints UI and API components and 90 days of history (20). The API component shows 100 per cent uptime and three scheduled maintenance windows (22 and 30 July, 7 October 2026). The UI was down for 1 hour 37 minutes on 16 July 2026 during a Hub outage, which we count as minor for the API surface (20). No rate limits were found for the management API. The Hub publishes limits per 5-minute window (1,000 API requests for a free user), without saying they cover api.endpoints.huggingface.cloud (5). The docs explain the 503 returned while a replica starts and the `X-Scale-Up-Timeout` header that holds the request, and advise 2 replicas for availability. No backoff guidance or idempotency keys for writes (8). No SLA found in the docs or on the pricing page (0). The service is generally available (10).",
          "schema": "Public OpenAPI 3.1 documents for the management API (40 paths, 46 operations) and the catalogue API (3 operations) (25). llms.txt with a Markdown twin of every docs page (10). Every operation has a summary tagged READ, WRITE, PUBLIC or PRO, but only 5 of 46 have a description. The MCP tool table states each tool's purpose and says to call `get_recommended_config` before `create_endpoint` (11). 127 typed schemas with required fields and enums for state, type and accelerator. List filters are comma-separated strings (12). Parameters carry examples. The management document lists only 200 responses, plus 501 on three log routes, while the catalogue document lists 400, 401, 404, 409 and 500 (7). Paths are versioned `/v2` and `/v3` and the document is version 2.0.0. Inference Endpoints has no changelog of its own. The Hub changelog and the docs repository history are the dated record (8).",
          "security": "OAuth through huggingface.co with `read-endpoints` and `write-endpoints` scopes, PKCE and dynamic client registration for the MCP server, and revocable fine-grained access tokens for the API. No documented option to send a token in a URL (30). Read and write scopes are separate, endpoints are private by default, organisations on Team and Enterprise plans can require approval of fine-grained tokens, and the MCP delete tool needs `confirm: true`. Other write calls have no confirmation step (17). The service returns the output and logs of the owner's own model, not third-party content (10). An audit log route and MCP tool on Pro, Team or Enterprise plans, plus logs and metrics for every endpoint (12). security.txt valid to 2030 with security@huggingface.co, SOC2 Type 2 stated for the Hub and Inference Endpoints, malware and pickle scanning of repositories. No bug bounty was found (14).",
          "transparency": "Closed service under clear terms that name Inference Endpoints, with an Apache-2.0 client (20). The security page says payloads and tokens aren't stored and logs are kept 30 days. The privacy policy dates from 28 March 2023 and gives retention as long as necessary, and a data processing agreement requires an Enterprise subscription (20). Instance deprecations are dated to the month in the pricing table and flagged in the provider API, and the TGI maintenance notice is dated. No deprecation policy for the API was found (10). The privacy policy lists 11 subprocessors with countries and the docs and API list deployment regions (18)."
        },
        "sources": [
          {
            "what": "docs index (llms.txt)",
            "url": "https://huggingface.co/docs/inference-endpoints/llms.txt",
            "seen": "2026-10-08"
          },
          {
            "what": "pricing",
            "url": "https://huggingface.co/docs/inference-endpoints/support/pricing",
            "seen": "2026-10-08"
          },
          {
            "what": "FAQ",
            "url": "https://huggingface.co/docs/inference-endpoints/support/faq",
            "seen": "2026-10-08"
          },
          {
            "what": "autoscaling guide",
            "url": "https://huggingface.co/docs/inference-endpoints/guides/autoscaling",
            "seen": "2026-10-08"
          },
          {
            "what": "configuration guide",
            "url": "https://huggingface.co/docs/inference-endpoints/guides/configuration",
            "seen": "2026-10-08"
          },
          {
            "what": "security and compliance",
            "url": "https://huggingface.co/docs/inference-endpoints/guides/security",
            "seen": "2026-10-08"
          },
          {
            "what": "MCP server guide",
            "url": "https://huggingface.co/docs/inference-endpoints/guides/mcp_server",
            "seen": "2026-10-08"
          },
          {
            "what": "API reference page",
            "url": "https://huggingface.co/docs/inference-endpoints/api_reference",
            "seen": "2026-10-08"
          },
          {
            "what": "management OpenAPI document",
            "url": "https://api.endpoints.huggingface.cloud/openapi.json",
            "seen": "2026-10-08"
          },
          {
            "what": "live provider list",
            "url": "https://api.endpoints.huggingface.cloud/v2/provider",
            "seen": "2026-10-08"
          },
          {
            "what": "catalogue OpenAPI document",
            "url": "https://endpoints.huggingface.co/api/openapi.json",
            "seen": "2026-10-08"
          },
          {
            "what": "MCP OAuth resource metadata",
            "url": "https://endpoints.huggingface.co/.well-known/oauth-protected-resource/mcp",
            "seen": "2026-10-08"
          },
          {
            "what": "OAuth authorisation server metadata",
            "url": "https://huggingface.co/.well-known/oauth-authorization-server",
            "seen": "2026-10-08"
          },
          {
            "what": "status page",
            "url": "https://status.huggingface.co/",
            "seen": "2026-10-08"
          },
          {
            "what": "access tokens",
            "url": "https://huggingface.co/docs/hub/security-tokens",
            "seen": "2026-10-08"
          },
          {
            "what": "Hub rate limits",
            "url": "https://huggingface.co/docs/hub/rate-limits",
            "seen": "2026-10-08"
          },
          {
            "what": "Python client guide",
            "url": "https://huggingface.co/docs/huggingface_hub/guides/inference_endpoints",
            "seen": "2026-10-08"
          },
          {
            "what": "huggingface_hub on PyPI",
            "url": "https://pypi.org/pypi/huggingface_hub/json",
            "seen": "2026-10-08"
          },
          {
            "what": "PyPI downloads",
            "url": "https://pypistats.org/api/packages/huggingface-hub/recent",
            "seen": "2026-10-08"
          },
          {
            "what": "docs repository history",
            "url": "https://github.com/huggingface/hf-endpoints-documentation",
            "seen": "2026-10-08"
          },
          {
            "what": "terms of service",
            "url": "https://huggingface.co/terms-of-service",
            "seen": "2026-10-08"
          },
          {
            "what": "privacy policy",
            "url": "https://huggingface.co/privacy",
            "seen": "2026-10-08"
          },
          {
            "what": "security.txt",
            "url": "https://huggingface.co/.well-known/security.txt",
            "seen": "2026-10-08"
          },
          {
            "what": "platform pricing",
            "url": "https://huggingface.co/pricing",
            "seen": "2026-10-08"
          },
          {
            "what": "Hub changelog",
            "url": "https://huggingface.co/changelog",
            "seen": "2026-10-08"
          }
        ],
        "openQuestions": [
          "unchecked: GitHub stars, issues and CI for huggingface_hub, because api.github.com answered 403",
          "unchecked: the Supplemental Terms PDF beyond its first page, which is all our reader extracted",
          "unchecked: the MCP tool definitions and annotations, because the server needs an OAuth login",
          "unchecked: the permissions a fine-grained token can carry for Inference Endpoints, which are shown only in the logged-in token settings",
          "unchecked: security incidents and advisories in the last 12 months, with web search unavailable",
          "unchecked: the registration date of huggingface.co, because RDAP returned 404",
          "No SLA, rate limit or idempotency documentation was found for the management API. An Enterprise contract may carry an SLA that isn't published.",
          "The docs price table and the live `/v2/provider` list disagree on Inferentia2 x1 ($0.75 against $1.95), AWS H200 availability and the RTX PRO 6000 Blackwell, and on the default scale-to-zero period (1 hour against 15 minutes). No deduction was taken.",
          "The lead was right on the product and interfaces. It omitted the MCP server and the catalogue API. Scale to zero and prices are now read."
        ]
      },
      "negative": 0,
      "verdict": "OAuth scopes separate reading endpoints from writing them, both OpenAPI documents are public, and the unauthenticated `/v2/provider` route lists every instance with its hourly price. An account needs a payment method and credits before the first deployment, no rate limits or SLA were found for the management API, and the docs price table disagrees with the live list in places.",
      "bestFor": "Teams whose models already live on the Hugging Face Hub and who want a dedicated endpoint on a named cloud and region with standard open-source engines.",
      "strengths": [
        "Public OpenAPI 3.1 documents for the management API (46 operations) and the catalogue API (3), plus llms.txt and a Markdown twin of every docs page",
        "The MCP server at endpoints.huggingface.co/mcp uses OAuth with `read-endpoints` and `write-endpoints` scopes, PKCE and dynamic client registration",
        "`GET /v2/provider` needs no token and returns each instance type by cloud and region with status and price per hour",
        "The MCP `delete_endpoint` tool returns a preview and deletes only on a second call with `confirm: true`",
        "The Inference Endpoints API component on status.huggingface.co shows 100 per cent uptime over the 90 days to 8 October 2026"
      ],
      "weaknesses": [
        "No free tier. The docs require a payment method and credits, and replicas are billed while initialising as well as running",
        "No rate limits, SLA or idempotency keys were found for the management API, and its OpenAPI document lists only 200 responses on 43 of 46 operations",
        "The docs price table and the live provider list disagree. Inferentia2 x1 is $0.75 in the docs and $1.95 in the API, and AWS H200 is listed in the docs and marked deprecated in the API",
        "A start from zero replicas takes minutes by the docs' own account, and the proxy answers 503 until a replica is ready",
        "The Hub outage of 16 July 2026 took the Inference Endpoints UI down for 1 hour 37 minutes"
      ],
      "agentNotes": [
        "Call `GET https://api.endpoints.huggingface.cloud/v2/provider` first and pick an instance whose `status` is `available`. The docs table lists types the API marks deprecated or not available",
        "Send `X-Scale-Up-Timeout: 600` on requests to an endpoint that scales to zero, or handle 503 while the first replica starts",
        "Set `scaleToZeroTimeout` yourself. The docs give a default of 1 hour and the OpenAPI document says 15 minutes",
        "Pause or delete an endpoint when the job is done. Billing covers every minute a replica is initialising or running",
        "Give the agent a fine-grained token or the `read-endpoints` scope unless it must deploy. Endpoints are private by default and take the same Hugging Face token as a bearer"
      ],
      "metrics": {
        "kind": "remote",
        "measured": false
      },
      "reviewCount": 0,
      "avgRating": 0,
      "history": [
        {
          "basis": "public evidence",
          "confidence": "medium",
          "grade": "B",
          "methodology": "0.4",
          "pending": [
            "performance",
            "tasks"
          ],
          "run": "2026-10-01",
          "runLabel": "October 2026 research run",
          "score": 64.5
        }
      ],
      "editorialScores": {
        "ergonomics": 62,
        "maintenance": 80,
        "payments": 20,
        "reliability": 63,
        "schema": 73,
        "security": 83,
        "transparency": 68
      },
      "provenanceScore": 67
    },
    "connect": {
      "install": "pip install huggingface_hub",
      "http": "curl \"https://api.endpoints.huggingface.cloud/v2/endpoint/$NAMESPACE\" \\\n  -H \"Authorization: Bearer $HF_TOKEN\""
    },
    "letme": {
      "capability": "https://letme.dev/compute.gpu",
      "tool": "https://letme.dev/hugging-face-inference-endpoints"
    },
    "notable": [
      "The management OpenAPI 3.1 document (title HF Inference Endpoints API, version 2.0.0) has 40 paths and 46 operations, each summary tagged [READ], [WRITE], [PUBLIC] or [PRO] (https://api.endpoints.huggingface.cloud/openapi.json)",
      "A separate catalogue API under `/api/v1` on endpoints.huggingface.co lists catalogue models without a token and deploys a model or a named recipe in one POST, with 400, 401, 404, 409 and 500 responses documented (https://endpoints.huggingface.co/api/openapi.json)",
      "The MCP server documentation was added on 19 August 2026 and lists 19 tools. `get_audit_logs` needs a Pro or Enterprise plan (https://huggingface.co/docs/inference-endpoints/guides/mcp_server)",
      "The MCP resource metadata names huggingface.co as the authorisation server and the scopes openid, profile, email, read-repos, read-billing, inference-api, read-endpoints and write-endpoints (https://endpoints.huggingface.co/.well-known/oauth-protected-resource/mcp)",
      "The pricing page says accounts need an active subscription and credits added, and that replicas are charged while initialising and running, by the minute (https://huggingface.co/docs/inference-endpoints/support/pricing)",
      "The pricing table marks AWS H100 and B200 as deprecated from December 2025 and AWS Ice Lake CPUs from July 2025. GCP H200 was added to the table on 7 October 2026 (https://github.com/huggingface/hf-endpoints-documentation)",
      "The security page says the Hub and Inference Endpoints are SOC2 Type 2 certified, payloads and tokens aren't stored, and logs are kept for 30 days (https://huggingface.co/docs/inference-endpoints/guides/security)",
      "The FAQ advises at least 2 replicas for an endpoint that must stay available, after intermittent 503 errors on running endpoints (https://huggingface.co/docs/inference-endpoints/support/faq)",
      "huggingface.co/pricing describes Inference Endpoints as having no cold starts, while the autoscaling guide describes a cold start after scale to zero (https://huggingface.co/pricing, https://huggingface.co/docs/inference-endpoints/guides/autoscaling)"
    ],
    "area": "models",
    "details": [
      {
        "label": "Surfaces",
        "value": "Management REST API at https://api.endpoints.huggingface.cloud (paths under `/v2` and `/v3`), catalogue API at https://endpoints.huggingface.co/api/v1, MCP server at https://endpoints.huggingface.co/mcp, the `huggingface_hub` Python client and the `hf endpoints` CLI"
      },
      {
        "label": "Engines",
        "value": "vLLM, Text Generation Inference (in maintenance mode since 11 December 2025), SGLang, llama.cpp, Text Embeddings Inference, the Inference Toolkit, or a custom container listening on port 80"
      },
      {
        "label": "GPUs",
        "value": "T4, L4, A10G, L40S, A100, RTX PRO 6000 Blackwell, H100 (GCP), H200 (GCP), AWS Inferentia2, and CPU instances, in sizes x1 to x8"
      },
      {
        "label": "Clouds and regions",
        "value": "AWS us-east-1, us-east-2 and eu-west-1, Azure eastus, GCP us-east4 and us-south1, per the live provider list. AWS us-west-2 is listed as not available"
      },
      {
        "label": "Scale to zero",
        "value": "Optional, with `minReplica` 0. The docs give a default idle period of 1 hour and the OpenAPI document says 15 minutes. `POST /v2/endpoint/{namespace}/{name}/scale-to-zero` forces it"
      },
      {
        "label": "Cold start",
        "value": "No figure published. The docs say initialising usually takes 3 to 5 minutes and that scaling up can take a few minutes. The proxy returns 503 meanwhile unless the request carries `X-Scale-Up-Timeout`"
      },
      {
        "label": "Autoscaling",
        "value": "By hardware utilisation (default threshold 80 per cent) or pending requests (default 1.5 a replica over 20 seconds). Scale-up is evaluated every minute, scale-down every 2 minutes with a 300-second stabilisation"
      },
      {
        "label": "Billing basis",
        "value": "Hourly rate billed per minute for each replica while initialising or running. Paused endpoints stop billing. Prepaid credits, with optional automatic recharge"
      },
      {
        "label": "Endpoint access",
        "value": "Private (default, Hugging Face token of the owner or organisation members), authenticated (any Hugging Face token) or public. AWS PrivateLink is available on AWS"
      },
      {
        "label": "MCP tools",
        "value": "19, including `list_endpoints`, `create_endpoint`, `update_endpoint`, `pause_endpoint`, `scale_endpoint_to_zero`, `delete_endpoint`, `call_endpoint`, `get_recommended_config`, `get_endpoint_logs`, `get_endpoint_metric` and `get_audit_logs`"
      },
      {
        "label": "Data retention",
        "value": "The docs say request payloads and tokens passed to an endpoint aren't stored and logs are kept for 30 days. A GDPR data processing agreement comes with an Enterprise subscription"
      }
    ],
    "unitPrices": [
      {
        "item": "NVIDIA T4 16 GB x1 (AWS, GCP)",
        "unit": "gpu-hour",
        "usd": 0.5,
        "note": "Billed per minute while initialising or running"
      },
      {
        "item": "NVIDIA L4 24 GB x1 (AWS)",
        "unit": "gpu-hour",
        "usd": 0.8,
        "note": "$0.70 on GCP us-east4"
      },
      {
        "item": "NVIDIA A10G 24 GB x1 (AWS)",
        "unit": "gpu-hour",
        "usd": 1,
        "note": "us-east-1 and eu-west-1"
      },
      {
        "item": "NVIDIA L40S 48 GB x1 (AWS)",
        "unit": "gpu-hour",
        "usd": 1.8,
        "note": "us-east-1"
      },
      {
        "item": "NVIDIA A100 80 GB x1 (AWS)",
        "unit": "gpu-hour",
        "usd": 2.5,
        "note": "$3.60 on GCP us-east4"
      },
      {
        "item": "NVIDIA RTX PRO 6000 Blackwell 96 GB x1 (AWS)",
        "unit": "gpu-hour",
        "usd": 2.75,
        "note": "us-east-2, in the live provider list and absent from the docs table"
      },
      {
        "item": "NVIDIA H200 141 GB x1 (GCP)",
        "unit": "gpu-hour",
        "usd": 5,
        "note": "us-south1. The AWS H200 in us-west-2 is marked deprecated in the API"
      },
      {
        "item": "NVIDIA H100 80 GB x1 (GCP)",
        "unit": "gpu-hour",
        "usd": 10,
        "note": "us-east4. The AWS H100 at $4.50 is deprecated from December 2025"
      },
      {
        "item": "Intel Sapphire Rapids x1, 1 vCPU and 2 GB (AWS)",
        "unit": "vcpu-hour",
        "usd": 0.033,
        "note": "$0.050 on GCP and $0.060 on Azure Intel Xeon"
      }
    ],
    "provenance": {
      "legalEntity": "Hugging Face, Inc.",
      "domain": "huggingface.co",
      "domainRegistered": "",
      "endpointOnVendorDomain": false,
      "terms": "https://huggingface.co/terms-of-service",
      "privacy": "https://huggingface.co/privacy",
      "statusPage": "https://status.huggingface.co",
      "changelog": "https://huggingface.co/changelog",
      "securityTxt": "valid",
      "checked": "2026-10-08",
      "notes": [
        "The Terms of Service (effective 15 September 2022) name Hugging Face, Inc., a Delaware corporation, list Inference Endpoints among the services they cover and are governed by New York law. They link Supplemental Terms (effective 28 April 2025) as a PDF, of which our reader extracted only the first page.",
        "The privacy policy (effective 28 March 2023) names Hugging Face, Inc. and its EU establishment Hugging Face, SAS, 9 rue des Colonnes, 75002 Paris, and lists 11 subprocessors with countries. The Inference Endpoints security page points to it.",
        "The management API answers at api.endpoints.huggingface.cloud and deployed endpoints at subdomains of endpoints.huggingface.cloud, a second domain of the vendor's. The catalogue API and the MCP server are on endpoints.huggingface.co.",
        "huggingface.co/.well-known/security.txt gives security@huggingface.co and expires on 1 July 2030. endpoints.huggingface.co/.well-known/security.txt returns 404.",
        "status.huggingface.co is a Better Stack page with separate components for the Inference Endpoints UI and API.",
        "The changelog at huggingface.co/changelog covers the whole Hub. Inference Endpoints has no changelog of its own. Dated changes are in the docs repository's commit history.",
        "rdap.org returned 404 for huggingface.co, so the registration date is unrecorded."
      ],
      "score": 67,
      "checks": [
        {
          "check": "Legal entity named",
          "value": "Hugging Face, Inc.",
          "points": 20,
          "max": 20,
          "state": "ok"
        },
        {
          "check": "Domain age",
          "value": "huggingface.co, no registry record we could read",
          "points": 0,
          "max": 15,
          "state": "no"
        },
        {
          "check": "Endpoint on the vendor's domain",
          "value": "api.endpoints.huggingface.cloud is not on huggingface.co",
          "points": 0,
          "max": 15,
          "state": "no"
        },
        {
          "check": "Terms of service",
          "value": "read, states 6 of the 7 things a reader expects, and has 1 clause that costs points",
          "points": 7.1,
          "max": 10,
          "state": "part"
        },
        {
          "check": "Privacy policy",
          "value": "read, states 8 of the 8 things a reader expects",
          "points": 10,
          "max": 10,
          "state": "ok"
        },
        {
          "check": "Status page",
          "value": "status.huggingface.co",
          "points": 10,
          "max": 10,
          "state": "ok"
        },
        {
          "check": "Changelog",
          "value": "published",
          "points": 10,
          "max": 10,
          "state": "ok"
        },
        {
          "check": "security.txt",
          "value": "valid",
          "points": 10,
          "max": 10,
          "state": "ok"
        }
      ],
      "policies": [
        {
          "kind": "terms",
          "url": "https://huggingface.co/terms-of-service",
          "state": "read",
          "readAt": "2026-10-08",
          "statedDate": "2022-09-15",
          "words": 4712,
          "points": 7.1,
          "max": 10,
          "expected": [
            {
              "key": "terms.date",
              "label": "Gives the date it was last updated",
              "found": true,
              "quote": "🗓 Effective Date: September 15, 2022",
              "says": "Last updated 2022-09-15"
            },
            {
              "key": "terms.law",
              "label": "Names the governing law or courts",
              "found": true,
              "quote": "These Terms and all matters regarding their interpretation and/or enforcement are governed by the Law of the State of New York, excluding its choice of law rules.",
              "says": "The law of the State of New York"
            },
            {
              "key": "terms.liability",
              "label": "States a limit on its liability",
              "found": true,
              "quote": "Either Party’s (and each Related Party’s) aggregate liability to the other Party or any third party in any circumstance will not exceed the amount that you paid us during the 12-month period immediately preceding the last claim (or $50 if relating to a free service).",
              "says": "Capped at the fees paid in the 12 months before the claim or $50"
            },
            {
              "key": "terms.termination",
              "label": "Says how the agreement or account can be ended",
              "found": true,
              "quote": "We may at any time modify, suspend, or discontinue, temporarily or permanently, the Services (or any part thereof) with or without notice."
            },
            {
              "key": "terms.changes",
              "label": "Says how changes to the terms are announced",
              "found": true,
              "quote": "We may also post and update supplemental terms for specific Services (\"Supplemental Terms\"), and such Supplemental Terms will also apply to you.",
              "says": "Changes are posted, with no other notice named"
            },
            {
              "key": "terms.use",
              "label": "Lists what users may not do",
              "found": true,
              "quote": "You may not disclose your password to any third party, and you are solely responsible for any action taken with your Account."
            },
            {
              "key": "terms.sla",
              "label": "Refers to a service level or uptime commitment",
              "found": false
            }
          ],
          "toKnow": [
            {
              "key": "terms.nonotice",
              "label": "Says the terms or the service can change without notice",
              "found": true,
              "quote": "We may at any time modify, suspend, or discontinue, temporarily or permanently, the Services (or any part thereof) with or without notice.",
              "costsPoints": true
            },
            {
              "key": "terms.cutoff",
              "label": "Says access can be ended without notice or for any reason",
              "found": true,
              "quote": "We may do the same, and we reserve the right to suspend or terminate your access to the Services anytime with or without cause, and at our own discretion, with or without notice."
            },
            {
              "key": "old",
              "label": "Has not been updated for three years or more",
              "found": true,
              "quote": "🗓 Effective Date: September 15, 2022"
            }
          ],
          "notes": [
            {
              "date": "2026-10-08",
              "text": "Each party's total liability is capped at the amount paid in the 12 months before the last claim, or 50 US dollars where the claim relates to a free service.",
              "quote": "Either Party’s (and each Related Party’s) aggregate liability to the other Party or any third party in any circumstance will not exceed the amount that you paid us during the 12-month period immediately preceding the last claim (or $50 if relating to a free service)."
            },
            {
              "date": "2026-10-08",
              "text": "Setting a repository public grants every user a perpetual, irrevocable licence to use, reproduce, distribute and make derivative works of its content through the services.",
              "quote": "If you decide to set your Repository public, you grant each User a perpetual, irrevocable, worldwide, royalty-free, non-exclusive license to use, display, publish, reproduce, distribute, and make derivative works of your Content through our Services and functionalities;"
            },
            {
              "date": "2026-10-08",
              "text": "After an account is cancelled, the vendor says it will use commercially reasonable efforts to delete the account's information and repository content within 90 days.",
              "quote": "Upon cancellation of your Account, we will use commercially reasonable efforts to delete your information and Content of your own Repositories, whether public or private, within 90 days."
            }
          ]
        },
        {
          "kind": "privacy",
          "url": "https://huggingface.co/privacy",
          "state": "read",
          "readAt": "2026-10-08",
          "statedDate": "2023-03-28",
          "words": 2581,
          "points": 10,
          "max": 10,
          "expected": [
            {
              "key": "privacy.date",
              "label": "Gives the date it was last updated",
              "found": true,
              "quote": "🗓 Effective Date: March 28, 2023",
              "says": "Last updated 2023-03-28"
            },
            {
              "key": "privacy.collected",
              "label": "Says what personal data is collected",
              "found": true,
              "quote": "We may collect Information from third parties that help us deliver the Services or process information."
            },
            {
              "key": "privacy.retention",
              "label": "Says how long data is kept",
              "found": true,
              "quote": "We retain your Information for as long as necessary to deliver the Services, to comply with any applicable legal requirements, to maintain security and prevent incidents and, in general, to pursue our legitimate interests.",
              "says": "For as long as needed, with no period named"
            },
            {
              "key": "privacy.processors",
              "label": "Says who else receives the data",
              "found": true,
              "quote": "California Civil Code Section 1798.83 also permits customers who are California residents to request certain information regarding Our disclosure of Personal Information to third parties for direct marketing purposes."
            },
            {
              "key": "privacy.sale",
              "label": "Says whether personal data is sold or shared for advertising",
              "found": true,
              "quote": "The Company will not sell, rent or lease your Personal Information except as provided for by this Policy.",
              "says": "Says it does not sell personal data"
            },
            {
              "key": "privacy.rights",
              "label": "Says what rights people have over their data",
              "found": true,
              "quote": "The Company also reserves the right to access this information with your consent, or without your consent only for the purposes of pursuing legitimate interests such as maintaining security on its Services or complying with any legal or regulatory obligations."
            },
            {
              "key": "privacy.contact",
              "label": "Gives a privacy contact",
              "found": true,
              "quote": "To make such a request, please send an email to privacy@huggingface.co.",
              "says": "privacy@huggingface.co"
            },
            {
              "key": "privacy.transfers",
              "label": "Says where data is transferred or stored",
              "found": true,
              "quote": "By using the Services, you consent to any such transfer of information outside of your country."
            }
          ],
          "toKnow": [
            {
              "key": "old",
              "label": "Has not been updated for three years or more",
              "found": true,
              "quote": "🗓 Effective Date: March 28, 2023"
            }
          ],
          "notes": [
            {
              "date": "2026-10-08",
              "text": "Aggregated information that does not identify a user may be disclosed to advertisers and partners, with or without payment, for purposes that include targeting advertisements.",
              "quote": "The Company may disclose Anonymous Information (with or without compensation) to third parties, including advertisers and partners, for purposes including, but not limited to, targeting advertisements."
            },
            {
              "date": "2026-10-08",
              "text": "The company may access information a user keeps private without consent, for legitimate interests such as maintaining security or meeting legal and regulatory obligations.",
              "quote": "The Company also reserves the right to access this information with your consent, or without your consent only for the purposes of pursuing legitimate interests such as maintaining security on its Services or complying with any legal or regulatory obligations."
            }
          ]
        }
      ]
    },
    "pageJsonUrl": "https://www.anchorterminal.com/tools/hugging-face-inference-endpoints.json",
    "live": {
      "slug": "hugging-face-inference-endpoints",
      "probe": {
        "target": "https://api.endpoints.huggingface.cloud",
        "method": "get",
        "lastAt": "2026-10-09T10:42:45.525175216Z",
        "lastOk": true,
        "lastStatus": 200,
        "lastMs": 266,
        "authRequired": false,
        "uptime24h": 100,
        "uptime30d": 100,
        "p50ms24h": 255,
        "p95ms24h": 295,
        "samples24h": 33,
        "samples30d": 33,
        "days": [
          {
            "date": "2026-10-09",
            "probes": 33,
            "ok": 33
          }
        ]
      },
      "vendorStatus": {
        "page": "https://status.huggingface.co",
        "indicator": "unknown",
        "summary": "no machine-readable status found",
        "checkedAt": "2026-10-09T07:58:03.654181492Z"
      },
      "updatedAt": "2026-10-09T10:42:45.525175216Z"
    }
  }
}
