{
  "data": {
    "category": {
      "area": "models",
      "capabilities": [
        "compute.gpu",
        "compute.serverless",
        "compute.endpoints",
        "compute.batch",
        "compute.containers"
      ],
      "description": "Clouds that run your own models and jobs on GPUs by the second, as serverless functions, endpoints or rented machines. Compared on GPU types and price per hour, cold starts, scaling and what you have to package.",
      "indexed": [
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/3doptix-optical-design.json",
          "kind": "mcp",
          "name": "3DOptix",
          "slug": "3doptix-optical-design",
          "url": "https://www.anchorterminal.com/tools/3doptix-optical-design"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/contenta-software-ai-video-enhancer.json",
          "kind": "mcp",
          "name": "AI Video Enhancer Studio",
          "slug": "contenta-software-ai-video-enhancer",
          "url": "https://www.anchorterminal.com/tools/contenta-software-ai-video-enhancer"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/amplerun-gpu-rentals.json",
          "kind": "mcp",
          "name": "AmpleRun GPU rentals",
          "slug": "amplerun-gpu-rentals",
          "url": "https://www.anchorterminal.com/tools/amplerun-gpu-rentals"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/fitllm.json",
          "kind": "mcp",
          "name": "FitLLM",
          "slug": "fitllm",
          "url": "https://www.anchorterminal.com/tools/fitllm"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/prismnetwork-mcp.json",
          "kind": "mcp",
          "name": "prismnetwork.tech MCP server",
          "slug": "prismnetwork-mcp",
          "url": "https://www.anchorterminal.com/tools/prismnetwork-mcp"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/contenta-software-videorecompress.json",
          "kind": "mcp",
          "name": "VideoRecompress Studio",
          "slug": "contenta-software-videorecompress",
          "url": "https://www.anchorterminal.com/tools/contenta-software-videorecompress"
        }
      ],
      "indexedCount": 6,
      "json": "https://www.anchorterminal.com/categories/gpu-compute.json",
      "name": "GPU \u0026 serverless compute",
      "slug": "gpu-compute",
      "test": "The same model deployed as an endpoint on each platform, called cold and warm, then scaled to zero. We time cold starts, check the scaling and add up the cost per GPU-hour.",
      "title": "GPU and serverless compute for AI workloads",
      "toolCount": 19,
      "tools": [
        "modal-sandboxes",
        "nebius-ai-cloud",
        "baseten",
        "hugging-face-inference-endpoints",
        "modal",
        "replicate-deploy",
        "crusoe-cloud",
        "vast-ai",
        "verda",
        "coreweave",
        "northflank",
        "thunder-compute",
        "beam",
        "cerebrium",
        "runpod",
        "lambda",
        "hyperbolic",
        "koyeb",
        "massed-compute"
      ],
      "url": "https://www.anchorterminal.com/categories/gpu-compute"
    },
    "faq": [
      {
        "answer": "Modal Sandboxes has the highest benchmark score of the 19 ranked GPU and serverless compute for AI workloads, 75.5 (BB). Nebius AI Cloud is second with 67.2 (B).",
        "question": "What are the highest-rated GPU and serverless compute for AI workloads for AI agents?"
      },
      {
        "answer": "1 of the 19 ranked here grade BB or better, the bar for agent-ready on the Anchor benchmark.",
        "question": "How many GPU and serverless compute for AI workloads are agent-ready?"
      },
      {
        "answer": "None of the ranked listings here accepts x402 for its main call yet.",
        "question": "Which GPU and serverless compute for AI workloads accept x402 payments?"
      },
      {
        "answer": "By published paid prices, Northflank, at $0.0167 per vCPU hour, the lowest of the 7 listings here with a paid price in this unit (free allowances aside). Plans, volume tiers and free allowances change the sum, so check the listing's price table.",
        "question": "Which of these GPU and serverless compute for AI workloads is cheapest?"
      },
      {
        "answer": "By the Anchor benchmark score out of 100, a weighted mean of the scored categories minus deductions for negative events, from public evidence re-checked as vendors change. Listings cannot pay for a place. The latest assessment behind this page is from 9 October 2026.",
        "question": "How is this list ranked?"
      }
    ],
    "howToChoose": [
      {
        "label": "Cold start from zero",
        "detail": "Check the cold start time from zero and whether it includes loading model weights, because an agent scaled to zero waits through it before answering."
      },
      {
        "label": "Billing per second or hour",
        "detail": "Check whether billing is per second or per hour, and whether idle warm instances are charged, since that sets the real cost of bursty agent traffic."
      },
      {
        "label": "GPU types and memory",
        "detail": "Check which GPU types each region has and how much memory each type holds, because a model that does not fit in memory cannot run."
      },
      {
        "label": "Autoscaling and scale to zero",
        "detail": "Check how quickly the endpoint scales up under load and whether it can scale to zero, because a slow scale-up makes concurrent agent calls queue or time out."
      }
    ],
    "picks": [
      {
        "also": {
          "name": "Nebius AI Cloud",
          "slug": "nebius-ai-cloud",
          "why": "B, 67.2/100"
        },
        "name": "Modal Sandboxes",
        "need": "Highest score overall",
        "slug": "modal-sandboxes",
        "why": "BB, 75.5/100 on the benchmark"
      },
      {
        "name": "Replicate Deployments",
        "need": "Schema \u0026 documentation",
        "slug": "replicate-deploy",
        "why": "85/100 on schema \u0026 documentation, against 79 for the overall leader"
      },
      {
        "name": "Nebius AI Cloud",
        "need": "Agent ergonomics",
        "slug": "nebius-ai-cloud",
        "why": "77/100 on agent ergonomics, against 67 for the overall leader"
      },
      {
        "name": "Hugging Face Inference Endpoints",
        "need": "Security \u0026 auth",
        "slug": "hugging-face-inference-endpoints",
        "why": "83/100 on security \u0026 auth, against 76 for the overall leader"
      },
      {
        "name": "Replicate Deployments",
        "need": "Transparency \u0026 trust",
        "slug": "replicate-deploy",
        "why": "78/100 on transparency \u0026 trust, against 72 for the overall leader"
      },
      {
        "also": {
          "name": "Cerebrium",
          "slug": "cerebrium",
          "why": "$0.0236 per vCPU hour"
        },
        "name": "Northflank",
        "need": "Lowest paid price per vCPU hour",
        "slug": "northflank",
        "why": "$0.0167 per vCPU hour, the lowest of the 7 listings here with a paid price in this unit (free allowances aside)"
      },
      {
        "name": "Thunder Compute",
        "need": "A hosted MCP endpoint",
        "slug": "thunder-compute",
        "why": "remote MCP server, nothing to install"
      },
      {
        "name": "Beam",
        "need": "Self-hosting under an open licence",
        "slug": "beam",
        "why": "self-hosted, AGPL-3 licence"
      }
    ],
    "ranked": 19,
    "shortlist": [
      {
        "bestFor": "GPU work inside a sandbox, or agents already running on Modal.",
        "grade": "BB",
        "name": "Modal Sandboxes",
        "position": 1,
        "price": "$0.071 / vCPU-hr",
        "score": 75.5,
        "slug": "modal-sandboxes",
        "strengths": [
          "GPU sandboxes at the same per-second rates as the rest of Modal",
          "Outbound traffic blockable or limited to CIDR ranges, and no inbound connections without tunnels",
          "$30 of compute every month on Starter, no card"
        ],
        "url": "https://www.anchorterminal.com/tools/modal-sandboxes",
        "verdict": "GPU sandboxes at the same per-second rates as the rest of Modal. No REST API, and the JavaScript and Go SDKs are beta.",
        "weaknesses": [
          "No REST API, and the JavaScript and Go SDKs are beta",
          "Default lifetime of 5 minutes and a hard maximum of 24 hours",
          "gVisor rather than a VM unless you're on Team or Enterprise for the VM runtime"
        ],
        "where": "local",
        "x402": "no"
      },
      {
        "bestFor": "Teams that want whole GPU VMs or InfiniBand clusters in Europe, the UK, Israel or the US with IAM, Terraform and an SLA, and are content to manage endpoint lifecycles themselves.",
        "grade": "B",
        "name": "Nebius AI Cloud",
        "position": 2,
        "price": "Pay per use",
        "score": 67.2,
        "slug": "nebius-ai-cloud",
        "strengths": [
          "OpenAPI 3.0.3 document at `https://api.nebius.cloud/openapi.json` with 602 operations, generated from the same protobuf definitions as the gRPC API, CLI, Terraform provider and SDKs",
          "`X-Idempotency-Key` header for modifying calls, and a `retry_type` field on errors that says whether to retry the call",
          "Service accounts sign in with an uploaded RSA key and receive 12-hour tokens, with roles granted per tenant, project or resource"
        ],
        "url": "https://www.anchorterminal.com/tools/nebius-ai-cloud",
        "verdict": "One API definition generates the REST and gRPC interfaces, the CLI, Terraform provider and three SDKs, with a 602-operation OpenAPI document, `X-Idempotency-Key` and role-scoped service accounts. The status page lists 14 major incidents between 14 July and 8 October 2026, no request rate limits were found, and signup needs a browser and a card.",
        "weaknesses": [
          "Status page lists 14 incidents marked major between 14 July and 8 October 2026, including about 21 hours of partial degradation in us-central1 on 19 August",
          "No request rate limits with numbers and no Retry-After guidance found in the reviewed documentation",
          "Serverless AI endpoints run on one container VM that is started and stopped by hand; no autoscaling or scale to zero found"
        ],
        "where": "both",
        "x402": "no"
      },
      {
        "bestFor": "Teams that want one model behind a production endpoint with real autoscaling knobs, environments and scoped keys.",
        "grade": "B",
        "name": "Baseten",
        "position": 3,
        "price": "Pay per use",
        "score": 66.5,
        "slug": "baseten",
        "strengths": [
          "Team API keys scoped to inference-only, metrics-only or a single environment or model, plus a Viewer role since 1 September 2026",
          "Public OpenAPI spec for the management API at api.baseten.co/v1/spec, and llms.txt with Markdown twins",
          "Rate limits published per endpoint with a `retry_after` field on 429"
        ],
        "url": "https://www.anchorterminal.com/tools/baseten",
        "verdict": "Team API keys scoped to inference-only, metrics-only or a single environment or model, plus a Viewer role since 1 September 2026. H100 at $6.50 and A100 at $4.00 an hour, and start-up and idle replica time are billed.",
        "weaknesses": [
          "H100 at $6.50 and A100 at $4.00 an hour, and start-up and idle replica time are billed",
          "21 status-page incidents between 31 July and 29 September 2026, mostly single-cluster 5xx",
          "A bot token with admin access to the product and GitOps repositories sat exposed from March 2023 until July 2026"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Teams whose models already live on the Hugging Face Hub and who want a dedicated endpoint on a named cloud and region with standard open-source engines.",
        "grade": "B",
        "name": "Hugging Face Inference Endpoints",
        "position": 4,
        "price": "$0.033 / vCPU-hr",
        "score": 64.5,
        "slug": "hugging-face-inference-endpoints",
        "strengths": [
          "Public OpenAPI 3.1 documents for the management API (46 operations) and the catalogue API (3), plus llms.txt and a Markdown twin of every docs page",
          "The MCP server at endpoints.huggingface.co/mcp uses OAuth with `read-endpoints` and `write-endpoints` scopes, PKCE and dynamic client registration",
          "`GET /v2/provider` needs no token and returns each instance type by cloud and region with status and price per hour"
        ],
        "url": "https://www.anchorterminal.com/tools/hugging-face-inference-endpoints",
        "verdict": "OAuth scopes separate reading endpoints from writing them, both OpenAPI documents are public, and the unauthenticated `/v2/provider` route lists every instance with its hourly price. An account needs a payment method and credits before the first deployment, no rate limits or SLA were found for the management API, and the docs price table disagrees with the live list in places.",
        "weaknesses": [
          "No free tier. The docs require a payment method and credits, and replicas are billed while initialising as well as running",
          "No rate limits, SLA or idempotency keys were found for the management API, and its OpenAPI document lists only 200 responses on 43 of 46 operations",
          "The docs price table and the live provider list disagree. Inferentia2 x1 is $0.75 in the docs and $1.95 in the API, and AWS H200 is listed in the docs and marked deprecated in the API"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Python teams that want GPU functions, batch jobs and HTTP endpoints from one decorator with scale to zero.",
        "grade": "B",
        "name": "Modal",
        "position": 5,
        "price": "$250 / mo",
        "score": 63.6,
        "slug": "modal",
        "strengths": [
          "Scale to zero by default, per-second billing and about one-second container boots",
          "Retention stated per data type (inputs and outputs up to 7 days, logs 1 to 30 days)",
          "Python, JavaScript and Go SDKs, with llms.txt and dated release notes"
        ],
        "url": "https://www.anchorterminal.com/tools/modal",
        "verdict": "Scale to zero by default, per-second billing and about one-second container boots. No REST API or OpenAPI spec for deploying or invoking Functions.",
        "weaknesses": [
          "No REST API or OpenAPI spec for deploying or invoking Functions",
          "Web endpoints are open by default until proxy tokens are added",
          "RBAC, audit logs and HIPAA only on Enterprise"
        ],
        "where": "local",
        "x402": "no"
      },
      {
        "bestFor": "Teams already calling Replicate's public models who want their own model behind the same API, MCP server and webhooks.",
        "grade": "B",
        "name": "Replicate Deployments",
        "position": 6,
        "price": "Pay per use",
        "score": 63.6,
        "slug": "replicate-deploy",
        "strengths": [
          "OpenAPI file, llms.txt and an MCP server with a two-tool code mode",
          "Deployment min and max instances settable over the API, 0 allowed",
          "API prediction data deleted after one hour by default"
        ],
        "url": "https://www.anchorterminal.com/tools/replicate-deploy",
        "verdict": "OpenAPI file, llms.txt and an MCP server with a two-tool code mode. Private instances bill set-up and idle time, H100 at $5.49 an hour.",
        "weaknesses": [
          "Private instances bill set-up and idle time, H100 at $5.49 an hour",
          "API tokens have no scopes, expiry or audit log",
          "Changelog silent since 21 April 2026"
        ],
        "where": "both",
        "x402": "no"
      },
      {
        "bestFor": "Teams that want whole GPU machines or clusters with managed Kubernetes or Slurm in the US, Iceland or Norway, under an SLA, and are content to manage lifecycles themselves.",
        "grade": "B",
        "name": "Crusoe Cloud",
        "position": 7,
        "price": "$0.04 / vCPU-hr",
        "score": 63.4,
        "slug": "crusoe-cloud",
        "strengths": [
          "Requests are signed with HMAC-SHA256 over the path, query, verb and timestamp, so the secret key never travels with a request.",
          "The API description is published for `/v1` and `/v1alpha5`, with 1,231 examples, and the docs serve `llms.txt` and a Markdown twin of each page.",
          "On-demand prices are public and billed per second. H100 is $3.90 and H200 $4.29 a GPU-hour, with no charge for ingress or egress."
        ],
        "url": "https://www.anchorterminal.com/tools/crusoe-cloud",
        "verdict": "The v1 REST API has a published OpenAPI description, Markdown docs, HMAC-signed keys, reader roles and a 90-day audit log, with on-demand GPU prices public. No idempotency keys or control-plane rate limits were found, spot and B200 prices go through sales, and a storage outage in eu-norway1 lasted about four hours on 29 September 2026.",
        "weaknesses": [
          "No idempotency key or safe-retry guidance was found for create and delete calls, which return asynchronous operations to poll.",
          "No request rate limits were found for the infrastructure API. Numbers are published only for Serverless Inference.",
          "Spot prices and on-demand prices for GB200, B200 and MI355X are listed as contact sales, and provisioning compute needs a non-prepaid credit card."
        ],
        "where": "both",
        "x402": "no"
      },
      {
        "bestFor": "Cost-sensitive training, batch work and self-managed inference where the agent can search listings, set a price cap and tolerate host variance or interruption.",
        "grade": "B",
        "name": "Vast.ai",
        "position": 8,
        "price": "Pay per use",
        "score": 62.6,
        "slug": "vast-ai",
        "strengths": [
          "API keys take 11 permission categories and per-endpoint constraints on resource IDs, and can be reset or deleted at once",
          "Public OpenAPI 3.1 file with 89 operations, llms.txt and Markdown docs pages",
          "MIT CLI and Python SDK, v1.8.3 on 2 October 2026, with 12 tagged releases since 27 July 2026"
        ],
        "url": "https://www.anchorterminal.com/tools/vast-ai",
        "verdict": "API keys can be limited by permission category and by resource ID, the REST API has a public OpenAPI 3.1 file, and the CLI ships weekly with a skill file for coding agents. Machines belong to independent hosts, prices move with the market, rate-limit thresholds are unpublished, and there is no SLA or free tier.",
        "weaknesses": [
          "No SLA. The terms say availability is not guaranteed and the service can change without notice",
          "Rate-limit thresholds are unpublished and 429 responses carry no `Retry-After` header",
          "No free tier. Credit is prepaid with a $5 minimum deposit after a browser signup and email verification"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Agents that rent whole GPU machines or clusters in Finland for training or batch work, or deploy scale-to-zero container endpoints, and that can run a local CLI for MCP.",
        "grade": "B",
        "name": "Verda",
        "position": 9,
        "price": "Pay per use",
        "score": 62.3,
        "slug": "verda",
        "strengths": [
          "OpenAPI 3.1 document with 112 operations and 156 schemas, plus `llms.txt` files on three hosts and the whole docs corpus as Markdown",
          "Rate limits of 500 requests a minute per project and 60 per endpoint, with `RateLimit` headers and `Retry-After` on 429",
          "API changelog with 11 dated entries between 11 September and 7 October 2026"
        ],
        "url": "https://www.anchorterminal.com/tools/verda",
        "verdict": "The REST API has a public OpenAPI 3.1 document, a dated changelog, published rate limits with `Retry-After`, and an audit log endpoint. The CLI's MCP server refuses billed or destructive calls without `confirm: true`. Access needs a browser signup and a prepaid balance, credentials carry one scope, and instance creation has no idempotency key.",
        "weaknesses": [
          "No free tier. Accounts are prepaid, and instances are discontinued and volumes deleted when the balance reaches zero",
          "Cloud API credentials carry one scope, `cloud-api-v1`, with no read-only credential found in the reviewed documentation",
          "`POST /v1/instances` has no idempotency key, so a retried launch can start a second billed instance"
        ],
        "where": "both",
        "x402": "no"
      },
      {
        "bestFor": "Teams that already run Kubernetes and need whole multi-GPU nodes or clusters for training and dedicated inference, with a contract and IAM.",
        "grade": "C",
        "name": "CoreWeave",
        "position": 10,
        "price": "Pay per use",
        "score": 61.5,
        "slug": "coreweave",
        "strengths": [
          "On-demand and spot prices per instance-hour are public, for example HGX H100 at $49.24 on demand and $19.71 spot for 8 GPUs",
          "Docs are served as Markdown with an llms.txt index, and OpenAPI 3.0 specs for CKS, VPC, storage and inference are embedded in the reference pages",
          "IAM has Viewer and Admin roles per service, API tokens carry an expiry, and Kubernetes audit logs can be forwarded with Telemetry Relay"
        ],
        "url": "https://www.anchorterminal.com/tools/coreweave",
        "verdict": "Per-hour prices for 8-GPU H100, H200 and B200 nodes are public, the docs ship as Markdown with llms.txt and embedded OpenAPI specs, and IAM has read-only roles per service. An organisation must be approved by CoreWeave's sales team before any token exists, and no request rate limits were found in the reviewed documentation.",
        "weaknesses": [
          "No self-serve route. CoreWeave's sales team approves an organisation and emails the activation link before a token can be created",
          "No request rate limits, 429 guidance or idempotency keys found for api.coreweave.com in the reviewed documentation",
          "The CKS API is versioned v1beta1 and the Node Pool resource and Dedicated Inference API v1alpha1"
        ],
        "where": "hosted",
        "x402": "no"
      }
    ],
    "updated": "2026-10-09"
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/best/gpu-compute/",
    "json": "https://www.anchorterminal.com/best/gpu-compute/index.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/best/gpu-compute/index.md",
    "slim": "https://www.anchorterminal.com/best/gpu-compute/index.min.md"
  },
  "markdown": "The 10 highest-scoring of 19 GPU and serverless compute for AI workloads on the Anchor benchmark, with a pick for each need and where each one falls short. Scores come from public evidence, re-checked as vendors change.\n\n- Ranked: 19 · agent-ready (BB or better): 1 · accept x402: 0 · hosted endpoints: 17\n- Full ranked table: https://www.anchorterminal.com/categories/gpu-compute.md\n- Head-to-head comparisons: https://www.anchorterminal.com/compare/gpu-compute/index.md (167)\n- Methodology: https://www.anchorterminal.com/benchmark/index.md\n\n## The shortlist\n\n| # | Tool | Grade | Score | Best for | Price | Where |\n| --- | --- | --- | --- | --- | --- | --- |\n| 1 | [Modal Sandboxes](https://www.anchorterminal.com/tools/modal-sandboxes.md) | BB | 75.5 | GPU work inside a sandbox, or agents already running on Modal. | $0.071 / vCPU-hr | local |\n| 2 | [Nebius AI Cloud](https://www.anchorterminal.com/tools/nebius-ai-cloud.md) | B | 67.2 | Teams that want whole GPU VMs or InfiniBand clusters in Europe, the UK, Israel or the US with IAM, Terraform and an SLA, and are content to manage endpoint lifecycles themselves. | Pay per use | hosted and local |\n| 3 | [Baseten](https://www.anchorterminal.com/tools/baseten.md) | B | 66.5 | Teams that want one model behind a production endpoint with real autoscaling knobs, environments and scoped keys. | Pay per use | hosted |\n| 4 | [Hugging Face Inference Endpoints](https://www.anchorterminal.com/tools/hugging-face-inference-endpoints.md) | B | 64.5 | Teams whose models already live on the Hugging Face Hub and who want a dedicated endpoint on a named cloud and region with standard open-source engines. | $0.033 / vCPU-hr | hosted |\n| 5 | [Modal](https://www.anchorterminal.com/tools/modal.md) | B | 63.6 | Python teams that want GPU functions, batch jobs and HTTP endpoints from one decorator with scale to zero. | $250 / mo | local |\n| 6 | [Replicate Deployments](https://www.anchorterminal.com/tools/replicate-deploy.md) | B | 63.6 | Teams already calling Replicate's public models who want their own model behind the same API, MCP server and webhooks. | Pay per use | hosted and local |\n| 7 | [Crusoe Cloud](https://www.anchorterminal.com/tools/crusoe-cloud.md) | B | 63.4 | Teams that want whole GPU machines or clusters with managed Kubernetes or Slurm in the US, Iceland or Norway, under an SLA, and are content to manage lifecycles themselves. | $0.04 / vCPU-hr | hosted and local |\n| 8 | [Vast.ai](https://www.anchorterminal.com/tools/vast-ai.md) | B | 62.6 | Cost-sensitive training, batch work and self-managed inference where the agent can search listings, set a price cap and tolerate host variance or interruption. | Pay per use | hosted |\n| 9 | [Verda](https://www.anchorterminal.com/tools/verda.md) | B | 62.3 | Agents that rent whole GPU machines or clusters in Finland for training or batch work, or deploy scale-to-zero container endpoints, and that can run a local CLI for MCP. | Pay per use | hosted and local |\n| 10 | [CoreWeave](https://www.anchorterminal.com/tools/coreweave.md) | C | 61.5 | Teams that already run Kubernetes and need whole multi-GPU nodes or clusters for training and dedicated inference, with a contract and IAM. | Pay per use | hosted |\n\n## Picks by need\n\n- Highest score overall: [Modal Sandboxes](https://www.anchorterminal.com/tools/modal-sandboxes.md), BB, 75.5/100 on the benchmark. Also [Nebius AI Cloud](https://www.anchorterminal.com/tools/nebius-ai-cloud.md), B, 67.2/100.\n- Schema \u0026 documentation: [Replicate Deployments](https://www.anchorterminal.com/tools/replicate-deploy.md), 85/100 on schema \u0026 documentation, against 79 for the overall leader.\n- Agent ergonomics: [Nebius AI Cloud](https://www.anchorterminal.com/tools/nebius-ai-cloud.md), 77/100 on agent ergonomics, against 67 for the overall leader.\n- Security \u0026 auth: [Hugging Face Inference Endpoints](https://www.anchorterminal.com/tools/hugging-face-inference-endpoints.md), 83/100 on security \u0026 auth, against 76 for the overall leader.\n- Transparency \u0026 trust: [Replicate Deployments](https://www.anchorterminal.com/tools/replicate-deploy.md), 78/100 on transparency \u0026 trust, against 72 for the overall leader.\n- Lowest paid price per vCPU hour: [Northflank](https://www.anchorterminal.com/tools/northflank.md), $0.0167 per vCPU hour, the lowest of the 7 listings here with a paid price in this unit (free allowances aside). Also [Cerebrium](https://www.anchorterminal.com/tools/cerebrium.md), $0.0236 per vCPU hour.\n- A hosted MCP endpoint: [Thunder Compute](https://www.anchorterminal.com/tools/thunder-compute.md), remote MCP server, nothing to install.\n- Self-hosting under an open licence: [Beam](https://www.anchorterminal.com/tools/beam.md), self-hosted, AGPL-3 licence.\n\n## How to choose\n\n- Cold start from zero: Check the cold start time from zero and whether it includes loading model weights, because an agent scaled to zero waits through it before answering.\n- Billing per second or hour: Check whether billing is per second or per hour, and whether idle warm instances are charged, since that sets the real cost of bursty agent traffic.\n- GPU types and memory: Check which GPU types each region has and how much memory each type holds, because a model that does not fit in memory cannot run.\n- Autoscaling and scale to zero: Check how quickly the endpoint scales up under load and whether it can scale to zero, because a slow scale-up makes concurrent agent calls queue or time out.\n\n- How the benchmark tests this category: The same model deployed as an endpoint on each platform, called cold and warm, then scaled to zero. We time cold starts, check the scaling and add up the cost per GPU-hour.\n\n## Each one in detail\n\n### 1. Modal Sandboxes, BB 75.5/100\n\nModal's sandboxed compute environments for running code, with SDK access, GPU support and filesystem snapshots.\n\n- Verdict: GPU sandboxes at the same per-second rates as the rest of Modal. No REST API, and the JavaScript and Go SDKs are beta.\n- Choose it for: GPU work inside a sandbox, or agents already running on Modal.\n- Strength: GPU sandboxes at the same per-second rates as the rest of Modal\n- Strength: Outbound traffic blockable or limited to CIDR ranges, and no inbound connections without tunnels\n- Strength: $30 of compute every month on Starter, no card\n- Weakness: No REST API, and the JavaScript and Go SDKs are beta\n- Weakness: Default lifetime of 5 minutes and a hard maximum of 24 hours\n- Weakness: gVisor rather than a VM unless you're on Team or Enterprise for the VM runtime\n- Price: $0.071 / vCPU-hr · Auth: API key · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/modal-sandboxes.md\n\n### 2. Nebius AI Cloud, B 67.2/100\n\nNebius AI Cloud rents NVIDIA GPU virtual machines and InfiniBand clusters, with managed Kubernetes, Slurm and Serverless AI jobs and endpoints for containers. Resources are managed through REST and gRPC APIs, a CLI, a Terraform provider and SDKs.\n\n- Verdict: One API definition generates the REST and gRPC interfaces, the CLI, Terraform provider and three SDKs, with a 602-operation OpenAPI document, `X-Idempotency-Key` and role-scoped service accounts. The status page lists 14 major incidents between 14 July and 8 October 2026, no request rate limits were found, and signup needs a browser and a card.\n- Choose it for: Teams that want whole GPU VMs or InfiniBand clusters in Europe, the UK, Israel or the US with IAM, Terraform and an SLA, and are content to manage endpoint lifecycles themselves.\n- Strength: OpenAPI 3.0.3 document at `https://api.nebius.cloud/openapi.json` with 602 operations, generated from the same protobuf definitions as the gRPC API, CLI, Terraform provider and SDKs\n- Strength: `X-Idempotency-Key` header for modifying calls, and a `retry_type` field on errors that says whether to retry the call\n- Strength: Service accounts sign in with an uploaded RSA key and receive 12-hour tokens, with roles granted per tenant, project or resource\n- Weakness: Status page lists 14 incidents marked major between 14 July and 8 October 2026, including about 21 hours of partial degradation in us-central1 on 19 August\n- Weakness: No request rate limits with numbers and no Retry-After guidance found in the reviewed documentation\n- Weakness: Serverless AI endpoints run on one container VM that is started and stopped by hand; no autoscaling or scale to zero found\n- Price: Pay per use · Auth: OAuth or key · x402: no · Where: hosted and local\n- Full assessment: https://www.anchorterminal.com/tools/nebius-ai-cloud.md\n\n### 3. Baseten, B 66.5/100\n\nDedicated model deployments packaged with the open-source Truss framework and served behind a per-model HTTPS endpoint, with autoscaling from zero replicas, async inference, a management API and per-minute GPU billing from T4 to B200.\n\n- Verdict: Team API keys scoped to inference-only, metrics-only or a single environment or model, plus a Viewer role since 1 September 2026. H100 at $6.50 and A100 at $4.00 an hour, and start-up and idle replica time are billed.\n- Choose it for: Teams that want one model behind a production endpoint with real autoscaling knobs, environments and scoped keys.\n- Strength: Team API keys scoped to inference-only, metrics-only or a single environment or model, plus a Viewer role since 1 September 2026\n- Strength: Public OpenAPI spec for the management API at api.baseten.co/v1/spec, and llms.txt with Markdown twins\n- Strength: Rate limits published per endpoint with a `retry_after` field on 429\n- Weakness: H100 at $6.50 and A100 at $4.00 an hour, and start-up and idle replica time are billed\n- Weakness: 21 status-page incidents between 31 July and 29 September 2026, mostly single-cluster 5xx\n- Weakness: A bot token with admin access to the product and GitOps repositories sat exposed from March 2023 until July 2026\n- Price: Pay per use · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/baseten.md\n\n### 4. Hugging Face Inference Endpoints, B 64.5/100\n\nManaged Hugging Face service that deploys a Hub model as a dedicated, autoscaling HTTPS endpoint on AWS, Azure or Google Cloud, using vLLM, TGI, SGLang, llama.cpp, TEI or a custom container. Managed by REST API, Python client, CLI or MCP.\n\n- Verdict: OAuth scopes separate reading endpoints from writing them, both OpenAPI documents are public, and the unauthenticated `/v2/provider` route lists every instance with its hourly price. An account needs a payment method and credits before the first deployment, no rate limits or SLA were found for the management API, and the docs price table disagrees with the live list in places.\n- Choose it for: Teams whose models already live on the Hugging Face Hub and who want a dedicated endpoint on a named cloud and region with standard open-source engines.\n- Strength: Public OpenAPI 3.1 documents for the management API (46 operations) and the catalogue API (3), plus llms.txt and a Markdown twin of every docs page\n- Strength: The MCP server at endpoints.huggingface.co/mcp uses OAuth with `read-endpoints` and `write-endpoints` scopes, PKCE and dynamic client registration\n- Strength: `GET /v2/provider` needs no token and returns each instance type by cloud and region with status and price per hour\n- Weakness: No free tier. The docs require a payment method and credits, and replicas are billed while initialising as well as running\n- Weakness: No rate limits, SLA or idempotency keys were found for the management API, and its OpenAPI document lists only 200 responses on 43 of 46 operations\n- Weakness: The docs price table and the live provider list disagree. Inferentia2 x1 is $0.75 in the docs and $1.95 in the API, and AWS H200 is listed in the docs and marked deprecated in the API\n- Price: $0.033 / vCPU-hr · Auth: OAuth or key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/hugging-face-inference-endpoints.md\n\n### 5. Modal, B 63.6/100\n\nServerless functions, web endpoints, servers and GPU jobs from a Python decorator, with JavaScript and Go SDKs.\n\n- Verdict: Scale to zero by default, per-second billing and about one-second container boots. No REST API or OpenAPI spec for deploying or invoking Functions.\n- Choose it for: Python teams that want GPU functions, batch jobs and HTTP endpoints from one decorator with scale to zero.\n- Strength: Scale to zero by default, per-second billing and about one-second container boots\n- Strength: Retention stated per data type (inputs and outputs up to 7 days, logs 1 to 30 days)\n- Strength: Python, JavaScript and Go SDKs, with llms.txt and dated release notes\n- Weakness: No REST API or OpenAPI spec for deploying or invoking Functions\n- Weakness: Web endpoints are open by default until proxy tokens are added\n- Weakness: RBAC, audit logs and HIPAA only on Enterprise\n- Price: $250 / mo · Auth: API key · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/modal.md\n\n### 6. Replicate Deployments, B 63.6/100\n\nReplicate's service for deploying and running custom models.\n\n- Verdict: OpenAPI file, llms.txt and an MCP server with a two-tool code mode. Private instances bill set-up and idle time, H100 at $5.49 an hour.\n- Choose it for: Teams already calling Replicate's public models who want their own model behind the same API, MCP server and webhooks.\n- Strength: OpenAPI file, llms.txt and an MCP server with a two-tool code mode\n- Strength: Deployment min and max instances settable over the API, 0 allowed\n- Strength: API prediction data deleted after one hour by default\n- Weakness: Private instances bill set-up and idle time, H100 at $5.49 an hour\n- Weakness: API tokens have no scopes, expiry or audit log\n- Weakness: Changelog silent since 21 April 2026\n- Price: Pay per use · Auth: API key · x402: no · Where: hosted and local\n- Full assessment: https://www.anchorterminal.com/tools/replicate-deploy.md\n\n### 7. Crusoe Cloud, B 63.4/100\n\nCrusoe Cloud rents NVIDIA and AMD GPU virtual machines with managed Kubernetes and Slurm, and runs hosted model inference, dedicated deployments and LoRA fine-tuning. Resources are managed through a REST API, a CLI, a Terraform provider and a Go client.\n\n- Verdict: The v1 REST API has a published OpenAPI description, Markdown docs, HMAC-signed keys, reader roles and a 90-day audit log, with on-demand GPU prices public. No idempotency keys or control-plane rate limits were found, spot and B200 prices go through sales, and a storage outage in eu-norway1 lasted about four hours on 29 September 2026.\n- Choose it for: Teams that want whole GPU machines or clusters with managed Kubernetes or Slurm in the US, Iceland or Norway, under an SLA, and are content to manage lifecycles themselves.\n- Strength: Requests are signed with HMAC-SHA256 over the path, query, verb and timestamp, so the secret key never travels with a request.\n- Strength: The API description is published for `/v1` and `/v1alpha5`, with 1,231 examples, and the docs serve `llms.txt` and a Markdown twin of each page.\n- Strength: On-demand prices are public and billed per second. H100 is $3.90 and H200 $4.29 a GPU-hour, with no charge for ingress or egress.\n- Weakness: No idempotency key or safe-retry guidance was found for create and delete calls, which return asynchronous operations to poll.\n- Weakness: No request rate limits were found for the infrastructure API. Numbers are published only for Serverless Inference.\n- Weakness: Spot prices and on-demand prices for GB200, B200 and MI355X are listed as contact sales, and provisioning compute needs a non-prepaid credit card.\n- Price: $0.04 / vCPU-hr · Auth: API key · x402: no · Where: hosted and local\n- Full assessment: https://www.anchorterminal.com/tools/crusoe-cloud.md\n\n### 8. Vast.ai, B 62.6/100\n\nVast.ai is a marketplace for renting GPUs by the second from independent hosts and data centres, as Docker instances, virtual machines or autoscaling serverless endpoints. Agents use the `vastai` CLI, a Python SDK or a REST API.\n\n- Verdict: API keys can be limited by permission category and by resource ID, the REST API has a public OpenAPI 3.1 file, and the CLI ships weekly with a skill file for coding agents. Machines belong to independent hosts, prices move with the market, rate-limit thresholds are unpublished, and there is no SLA or free tier.\n- Choose it for: Cost-sensitive training, batch work and self-managed inference where the agent can search listings, set a price cap and tolerate host variance or interruption.\n- Strength: API keys take 11 permission categories and per-endpoint constraints on resource IDs, and can be reset or deleted at once\n- Strength: Public OpenAPI 3.1 file with 89 operations, llms.txt and Markdown docs pages\n- Strength: MIT CLI and Python SDK, v1.8.3 on 2 October 2026, with 12 tagged releases since 27 July 2026\n- Weakness: No SLA. The terms say availability is not guaranteed and the service can change without notice\n- Weakness: Rate-limit thresholds are unpublished and 429 responses carry no `Retry-After` header\n- Weakness: No free tier. Credit is prepaid with a $5 minimum deposit after a browser signup and email verification\n- Price: Pay per use · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/vast-ai.md\n\n### 9. Verda, B 62.3/100\n\nVerda, formerly DataCrunch, is a Finnish GPU cloud renting GPU and CPU instances, clusters, serverless containers and storage. Agents use its REST API, Python and Go SDKs, or a CLI with a built-in MCP server in beta.\n\n- Verdict: The REST API has a public OpenAPI 3.1 document, a dated changelog, published rate limits with `Retry-After`, and an audit log endpoint. The CLI's MCP server refuses billed or destructive calls without `confirm: true`. Access needs a browser signup and a prepaid balance, credentials carry one scope, and instance creation has no idempotency key.\n- Choose it for: Agents that rent whole GPU machines or clusters in Finland for training or batch work, or deploy scale-to-zero container endpoints, and that can run a local CLI for MCP.\n- Strength: OpenAPI 3.1 document with 112 operations and 156 schemas, plus `llms.txt` files on three hosts and the whole docs corpus as Markdown\n- Strength: Rate limits of 500 requests a minute per project and 60 per endpoint, with `RateLimit` headers and `Retry-After` on 429\n- Strength: API changelog with 11 dated entries between 11 September and 7 October 2026\n- Weakness: No free tier. Accounts are prepaid, and instances are discontinued and volumes deleted when the balance reaches zero\n- Weakness: Cloud API credentials carry one scope, `cloud-api-v1`, with no read-only credential found in the reviewed documentation\n- Weakness: `POST /v1/instances` has no idempotency key, so a retried launch can start a second billed instance\n- Price: Pay per use · Auth: OAuth · x402: no · Where: hosted and local\n- Full assessment: https://www.anchorterminal.com/tools/verda.md\n\n### 10. CoreWeave, C 61.5/100\n\nCoreWeave is a GPU cloud that rents NVIDIA GPU nodes through a managed Kubernetes service (CKS), with REST and gRPC platform APIs, a Terraform provider and a hosted MCP server for observability.\n\n- Verdict: Per-hour prices for 8-GPU H100, H200 and B200 nodes are public, the docs ship as Markdown with llms.txt and embedded OpenAPI specs, and IAM has read-only roles per service. An organisation must be approved by CoreWeave's sales team before any token exists, and no request rate limits were found in the reviewed documentation.\n- Choose it for: Teams that already run Kubernetes and need whole multi-GPU nodes or clusters for training and dedicated inference, with a contract and IAM.\n- Strength: On-demand and spot prices per instance-hour are public, for example HGX H100 at $49.24 on demand and $19.71 spot for 8 GPUs\n- Strength: Docs are served as Markdown with an llms.txt index, and OpenAPI 3.0 specs for CKS, VPC, storage and inference are embedded in the reference pages\n- Strength: IAM has Viewer and Admin roles per service, API tokens carry an expiry, and Kubernetes audit logs can be forwarded with Telemetry Relay\n- Weakness: No self-serve route. CoreWeave's sales team approves an organisation and emails the activation link before a token can be created\n- Weakness: No request rate limits, 429 guidance or idempotency keys found for api.coreweave.com in the reviewed documentation\n- Weakness: The CKS API is versioned v1beta1 and the Node Pool resource and Dedicated Inference API v1alpha1\n- Price: Pay per use · Auth: Token · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/coreweave.md\n\n9 more are ranked in the full table: https://www.anchorterminal.com/categories/gpu-compute.md\n\n## Head to head\n\n- [Baseten vs Nebius AI Cloud](https://www.anchorterminal.com/compare/baseten-vs-nebius-ai-cloud.md)\n- [Hugging Face Inference Endpoints vs Nebius AI Cloud](https://www.anchorterminal.com/compare/hugging-face-inference-endpoints-vs-nebius-ai-cloud.md)\n- [Modal vs Nebius AI Cloud](https://www.anchorterminal.com/compare/modal-vs-nebius-ai-cloud.md)\n- [Baseten vs Hugging Face Inference Endpoints](https://www.anchorterminal.com/compare/baseten-vs-hugging-face-inference-endpoints.md)\n- [Baseten vs Modal](https://www.anchorterminal.com/compare/baseten-vs-modal.md)\n- [Hugging Face Inference Endpoints vs Modal](https://www.anchorterminal.com/compare/hugging-face-inference-endpoints-vs-modal.md)\n\n## Questions\n\n### What are the highest-rated GPU and serverless compute for AI workloads for AI agents?\n\nModal Sandboxes has the highest benchmark score of the 19 ranked GPU and serverless compute for AI workloads, 75.5 (BB). Nebius AI Cloud is second with 67.2 (B).\n\n### How many GPU and serverless compute for AI workloads are agent-ready?\n\n1 of the 19 ranked here grade BB or better, the bar for agent-ready on the Anchor benchmark.\n\n### Which GPU and serverless compute for AI workloads accept x402 payments?\n\nNone of the ranked listings here accepts x402 for its main call yet.\n\n### Which of these GPU and serverless compute for AI workloads is cheapest?\n\nBy published paid prices, Northflank, at $0.0167 per vCPU hour, the lowest of the 7 listings here with a paid price in this unit (free allowances aside). Plans, volume tiers and free allowances change the sum, so check the listing's price table.\n\n### How is this list ranked?\n\nBy the Anchor benchmark score out of 100, a weighted mean of the scored categories minus deductions for negative events, from public evidence re-checked as vendors change. Listings cannot pay for a place. The latest assessment behind this page is from 9 October 2026.\n\n## How this list is made\n\nThe order is the Anchor benchmark score, the same number as on each listing. Each listing is graded from public evidence against the benchmark checklist, and the picks are worked out from those grades, prices and facts. No listing pays for its place, and paid audits or listing help never change a score.\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-10",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Best of",
        "url": "https://www.anchorterminal.com/best/"
      },
      {
        "name": "GPU \u0026 serverless compute",
        "url": ""
      }
    ],
    "description": "Modal Sandboxes (BB), Nebius AI Cloud (B) and Baseten (B) lead the 19 ranked GPU and serverless compute for AI workloads. Picks by need, strengths, weaknesses and prices from the Anchor benchmark.",
    "facts": [
      "Modal Sandboxes BB",
      "Nebius AI Cloud B",
      "Baseten B"
    ],
    "h1": "Best GPU and serverless compute for AI workloads",
    "image": "https://www.anchorterminal.com/assets/og/best-gpu-compute.png",
    "path": "/best/gpu-compute/",
    "published": "",
    "section": "tools",
    "title": "Best GPU and serverless compute for AI workloads in 2026, ranked",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/best/gpu-compute/"
  },
  "tokens": {
    "markdown": 5950,
    "slim": 1530
  },
  "version": 1
}
