{
  "data": {
    "similar": [
      {
        "grade": "B",
        "json": "https://www.anchorterminal.com/tools/localai.json",
        "name": "LocalAI",
        "score": 68,
        "shared": [
          "inference.local",
          "inference.open-weights",
          "agent.mcp-client",
          "embed.text",
          "rerank",
          "speech.stt"
        ],
        "slug": "localai"
      },
      {
        "grade": "C",
        "json": "https://www.anchorterminal.com/tools/llama-cpp.json",
        "name": "llama.cpp",
        "score": 60.2,
        "shared": [
          "inference.local",
          "inference.open-weights",
          "embed.text",
          "rerank",
          "inference.decision",
          "agent.mcp-client"
        ],
        "slug": "llama-cpp"
      },
      {
        "grade": "B",
        "json": "https://www.anchorterminal.com/tools/lemonade.json",
        "name": "Lemonade",
        "score": 63.8,
        "shared": [
          "inference.local",
          "inference.open-weights",
          "embed.text",
          "rerank",
          "speech.stt"
        ],
        "slug": "lemonade"
      },
      {
        "grade": "C",
        "json": "https://www.anchorterminal.com/tools/koboldcpp.json",
        "name": "KoboldCpp",
        "score": 60.5,
        "shared": [
          "inference.local",
          "inference.open-weights",
          "agent.mcp-client",
          "embed.text",
          "speech.stt"
        ],
        "slug": "koboldcpp"
      },
      {
        "grade": "B",
        "json": "https://www.anchorterminal.com/tools/cloudflare-workers-ai.json",
        "name": "Cloudflare Workers AI",
        "score": 68.3,
        "shared": [
          "inference.open-weights",
          "embed.text",
          "rerank",
          "speech.stt"
        ],
        "slug": "cloudflare-workers-ai"
      },
      {
        "grade": "B",
        "json": "https://www.anchorterminal.com/tools/deepinfra.json",
        "name": "DeepInfra",
        "score": 63,
        "shared": [
          "inference.open-weights",
          "embed.text",
          "rerank",
          "speech.stt"
        ],
        "slug": "deepinfra"
      }
    ],
    "tool": {
      "slug": "vllm",
      "name": "vLLM",
      "vendor": "vLLM project (PyTorch Foundation)",
      "vendorUrl": "https://vllm.ai",
      "kind": "http-api",
      "category": "local-ai",
      "summary": "vLLM is an open-source inference and serving engine for open-weight language models. `vllm serve` runs an HTTP server with OpenAI-compatible, Anthropic Messages, embedding, reranking and transcription routes on the owner's own GPUs or CPUs.",
      "url": "https://www.anchorterminal.com/tools/vllm",
      "markdownUrl": "https://www.anchorterminal.com/tools/vllm.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/vllm.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/vllm.json",
      "repo": "https://github.com/vllm-project/vllm",
      "license": "Apache-2.0",
      "transports": [
        "http"
      ],
      "packages": [
        {
          "registry": "pypi",
          "name": "vllm"
        },
        {
          "registry": "oci",
          "name": "vllm/vllm-openai"
        }
      ],
      "auth": "none",
      "authNotes": "No credential by default. `--api-key` (one or several keys) or `VLLM_API_KEY` turns on a Bearer check for paths under `/v1`, `/v2`, `/inference` and `/cohere` only, so `/invocations`, `/pooling`, `/classify`, `/score`, `/rerank` and control routes such as `/pause` stay open. Keys have no scopes and change with a restart. The key is read from the `Authorization` header, never the query string. gRPC has no authentication (https://github.com/vllm-project/vllm/blob/main/docs/usage/security.md).",
      "pricing": "free",
      "pricingNotes": "Free under Apache-2.0, with no account, key or card. Nothing is sold by the project. You pay for your own hardware and electricity.",
      "priceSummary": "Free · OSS",
      "where": "local",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the docs or the source (checked 2026-10-09).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 93444,
        "npmWeekly": null,
        "pypiWeekly": null,
        "asOf": "2026-10-09"
      },
      "docsUrl": "https://docs.vllm.ai/en/stable/",
      "capabilities": [
        "inference.local",
        "inference.open-weights",
        "embed.text",
        "rerank",
        "speech.stt",
        "inference.decision",
        "agent.mcp-client"
      ],
      "tags": [
        "open-source",
        "local",
        "self-hosted",
        "free",
        "no-card",
        "openai-compatible",
        "docker",
        "pre-1.0",
        "telemetry-default-on"
      ],
      "lastRelease": "2026-10-02",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 57.7,
        "grade": "C",
        "agentReady": false,
        "rank": 600,
        "ranked": true,
        "rankOf": 950,
        "categoryRank": 8,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 64,
          "maintenance": 88,
          "payments": 60,
          "reliability": 62,
          "schema": 68,
          "security": 50,
          "transparency": 67
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "breakdown": [
          {
            "key": "reliability",
            "name": "Reliability",
            "weight": 16,
            "effectiveWeight": 20,
            "score": 62,
            "points": 12.4,
            "reason": "Read with the local-software lines, since vLLM runs on the owner's hardware with no hosted service. Installs from PyPI (`vllm`) and Docker images, with Linux and Python 3.11 to 3.14 stated in the quickstart and `pyproject.toml` (20). A public Buildkite pipeline and GitHub workflows report 74 checks on each commit to main. On the head commit of 9 October 2026 they were still running, with the finished steps passing, and we did not read a completed run (18 of 25). 2,592 open issues under a stale bot that closes after 90 and 30 days. Of 48 bug reports filed from 22 to 28 September 2026, 42 had a comment and 13 were closed by 9 October. Many of the 81 advisories of the last 12 months are requests that stop the engine (14 of 25). Release notes carry a Breaking Changes and Deprecations section, and a written policy sets three stages for removals, but each fortnightly minor release breaks something and compatibility holds for a limited number of minor versions (10 of 15). Version 0.31.0, pre-1.0 (0)."
          },
          {
            "key": "performance",
            "name": "Performance",
            "weight": 10,
            "effectiveWeight": 0,
            "pending": true,
            "points": 0,
            "reason": "Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes."
          },
          {
            "key": "schema",
            "name": "Schema \u0026 documentation",
            "weight": 13,
            "effectiveWeight": 16.25,
            "score": 68,
            "points": 11.05,
            "reason": "The server is a FastAPI application with Pydantic request models, so a running instance serves its own OpenAPI document and Swagger page unless `--disable-fastapi-docs` is set, and the gRPC services have published proto files. No static OpenAPI file is published in the docs or the repository (15 of 25). docs.vllm.ai/llms.txt returns 404. vllm.ai/llms.txt lists site pages and blog posts, and the docs are Markdown files in the repository (5 of 10). The serving pages say which model types each route applies to, which OpenAI fields are ignored, and which routes must not be exposed (15 of 20). Typed request models, with structured outputs by JSON schema, choice, regex or grammar (12 of 15). Examples for most routes in the docs and the `examples` folder. The error shape (message, type, param, code) is defined in the source, and no error reference page was found (9 of 15). GitHub releases with full notes about every two weeks and vllm.ai/releases. The HTTP API has no version of its own beyond the `/v1` prefix (12 of 15)."
          },
          {
            "key": "ergonomics",
            "name": "Agent ergonomics",
            "weight": 13,
            "effectiveWeight": 16.25,
            "score": 64,
            "points": 10.4,
            "reason": "Read for an API. `max_tokens`, structured outputs, log probabilities and streaming control the size and shape of a reply (18 of 25). `/tokenize` and `/v1/messages/count_tokens` count tokens before a call and `/load` reports server load. `/v1/models` is not paged (14 of 20). Errors follow the OpenAI shape with a type, a param and a code, with no documented list of codes (12 of 20). Generation is stateless and safe to repeat, `X-Request-Id` can be echoed with `--enable-request-id-headers`, and `/health` is public. No retry guidance was found (11 of 20). `vllm serve \u003cmodel\u003e` starts a working server and OpenAI and Anthropic clients work against it. There is no client library of its own, tool calling needs `--enable-auto-tool-choice` and a parser chosen for the model, and sampling defaults come from the model's `generation_config.json` unless overridden (9 of 15)."
          },
          {
            "key": "security",
            "name": "Security \u0026 auth",
            "weight": 14,
            "effectiveWeight": 17.5,
            "score": 50,
            "points": 8.75,
            "reason": "Read with the tool checklist. Optional static keys from `--api-key` or `VLLM_API_KEY`, compared in constant time and read only from the `Authorization` header. They are off by default, have no scopes, and are checked only under `/v1`, `/v2`, `/inference` and `/cohere`, so `/invocations` runs the same inference without a key. With `--host` unset the server binds every interface, and gRPC has no authentication (10 of 30). Development routes, runtime LoRA loading, tool servers and endpoint plugins are off by default. `/pause`, `/abort_requests` and `/update_weights` answer without a key on generation servers, and there is no read-only mode (6 of 20). The security guide covers model-generated code in the demo code interpreter, allowed domains for media URLs, cache salting and the lack of tenant isolation (10 of 15). Access logs, `--enable-log-requests`, Prometheus `/metrics` and OpenTelemetry examples, with no per-key record (9 of 15). SECURITY.md with severity classes, a named vulnerability management team, private reporting through GitHub, CVEs and a prenotification group, and at least 100 published advisories. No security.txt and no bounty found (15 of 20)."
          },
          {
            "key": "payments",
            "name": "Payments \u0026 pricing",
            "weight": 10,
            "effectiveWeight": 12.5,
            "score": 60,
            "points": 7.5,
            "reason": "Read with the self-hosted rule. No x402, MPP or L402 in the docs or the source (0). Free under Apache-2.0 with no account, key or card, and nothing to buy, so 20, 20 and 20 on the last three lines."
          },
          {
            "key": "tasks",
            "name": "Task success",
            "weight": 10,
            "effectiveWeight": 0,
            "pending": true,
            "points": 0,
            "reason": "Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored."
          },
          {
            "key": "maintenance",
            "name": "Maintenance \u0026 community",
            "weight": 7,
            "effectiveWeight": 8.75,
            "score": 88,
            "points": 7.7,
            "reason": "v0.31.0 was tagged on 2 October 2026 and its notes published on 5 October (30). Eight stable releases from v0.25.1 on 12 July to v0.31.0, with release candidates between them (20). The v0.31.0 notes count 717 commits from 307 contributors. 2,592 open issues, and 42 of 48 bug reports from the week of 22 September had a comment by 9 October. Four bug reports filed on 9 October had none yet (19 of 25). The Python library ships in the same package as the server, and a Rust proto crate is tagged separately (proto-v0.5.0 on 9 October). The server has no client library of its own (10 of 15). Dependabot runs weekly for pip and GitHub Actions, with pre-commit and Buildkite checks on each commit (9 of 10)."
          },
          {
            "key": "transparency",
            "name": "Transparency \u0026 trust",
            "weight": 7,
            "effectiveWeight": 8.75,
            "score": 67,
            "points": 5.86,
            "note": "editorial 80, provenance 53",
            "reason": "The editorial half. Apache-2.0, all of it public (30). The usage statistics page lists what is collected, links the collecting code, keeps a local copy at `~/.config/vllm/usage_stats.json` and says a cleaned subset is published. No privacy policy, retention period or recipient list for stats.vllm.ai was found, and vllm.ai links no privacy policy (16 of 30). A written deprecation policy with three stages across minor releases, and breaking changes listed in each release's notes (17 of 20). Telemetry is disclosed with three opt-outs, `VLLM_NO_USAGE_STATS`, `DO_NOT_TRACK` and a file. It is on by default and reports every ten minutes (17 of 20)."
          }
        ],
        "assessment": {
          "date": "2026-10-09",
          "basis": "public evidence",
          "confidence": "medium",
          "notes": {
            "ergonomics": "Read for an API. `max_tokens`, structured outputs, log probabilities and streaming control the size and shape of a reply (18 of 25). `/tokenize` and `/v1/messages/count_tokens` count tokens before a call and `/load` reports server load. `/v1/models` is not paged (14 of 20). Errors follow the OpenAI shape with a type, a param and a code, with no documented list of codes (12 of 20). Generation is stateless and safe to repeat, `X-Request-Id` can be echoed with `--enable-request-id-headers`, and `/health` is public. No retry guidance was found (11 of 20). `vllm serve \u003cmodel\u003e` starts a working server and OpenAI and Anthropic clients work against it. There is no client library of its own, tool calling needs `--enable-auto-tool-choice` and a parser chosen for the model, and sampling defaults come from the model's `generation_config.json` unless overridden (9 of 15).",
            "maintenance": "v0.31.0 was tagged on 2 October 2026 and its notes published on 5 October (30). Eight stable releases from v0.25.1 on 12 July to v0.31.0, with release candidates between them (20). The v0.31.0 notes count 717 commits from 307 contributors. 2,592 open issues, and 42 of 48 bug reports from the week of 22 September had a comment by 9 October. Four bug reports filed on 9 October had none yet (19 of 25). The Python library ships in the same package as the server, and a Rust proto crate is tagged separately (proto-v0.5.0 on 9 October). The server has no client library of its own (10 of 15). Dependabot runs weekly for pip and GitHub Actions, with pre-commit and Buildkite checks on each commit (9 of 10).",
            "payments": "Read with the self-hosted rule. No x402, MPP or L402 in the docs or the source (0). Free under Apache-2.0 with no account, key or card, and nothing to buy, so 20, 20 and 20 on the last three lines.",
            "reliability": "Read with the local-software lines, since vLLM runs on the owner's hardware with no hosted service. Installs from PyPI (`vllm`) and Docker images, with Linux and Python 3.11 to 3.14 stated in the quickstart and `pyproject.toml` (20). A public Buildkite pipeline and GitHub workflows report 74 checks on each commit to main. On the head commit of 9 October 2026 they were still running, with the finished steps passing, and we did not read a completed run (18 of 25). 2,592 open issues under a stale bot that closes after 90 and 30 days. Of 48 bug reports filed from 22 to 28 September 2026, 42 had a comment and 13 were closed by 9 October. Many of the 81 advisories of the last 12 months are requests that stop the engine (14 of 25). Release notes carry a Breaking Changes and Deprecations section, and a written policy sets three stages for removals, but each fortnightly minor release breaks something and compatibility holds for a limited number of minor versions (10 of 15). Version 0.31.0, pre-1.0 (0).",
            "schema": "The server is a FastAPI application with Pydantic request models, so a running instance serves its own OpenAPI document and Swagger page unless `--disable-fastapi-docs` is set, and the gRPC services have published proto files. No static OpenAPI file is published in the docs or the repository (15 of 25). docs.vllm.ai/llms.txt returns 404. vllm.ai/llms.txt lists site pages and blog posts, and the docs are Markdown files in the repository (5 of 10). The serving pages say which model types each route applies to, which OpenAI fields are ignored, and which routes must not be exposed (15 of 20). Typed request models, with structured outputs by JSON schema, choice, regex or grammar (12 of 15). Examples for most routes in the docs and the `examples` folder. The error shape (message, type, param, code) is defined in the source, and no error reference page was found (9 of 15). GitHub releases with full notes about every two weeks and vllm.ai/releases. The HTTP API has no version of its own beyond the `/v1` prefix (12 of 15).",
            "security": "Read with the tool checklist. Optional static keys from `--api-key` or `VLLM_API_KEY`, compared in constant time and read only from the `Authorization` header. They are off by default, have no scopes, and are checked only under `/v1`, `/v2`, `/inference` and `/cohere`, so `/invocations` runs the same inference without a key. With `--host` unset the server binds every interface, and gRPC has no authentication (10 of 30). Development routes, runtime LoRA loading, tool servers and endpoint plugins are off by default. `/pause`, `/abort_requests` and `/update_weights` answer without a key on generation servers, and there is no read-only mode (6 of 20). The security guide covers model-generated code in the demo code interpreter, allowed domains for media URLs, cache salting and the lack of tenant isolation (10 of 15). Access logs, `--enable-log-requests`, Prometheus `/metrics` and OpenTelemetry examples, with no per-key record (9 of 15). SECURITY.md with severity classes, a named vulnerability management team, private reporting through GitHub, CVEs and a prenotification group, and at least 100 published advisories. No security.txt and no bounty found (15 of 20).",
            "transparency": "The editorial half. Apache-2.0, all of it public (30). The usage statistics page lists what is collected, links the collecting code, keeps a local copy at `~/.config/vllm/usage_stats.json` and says a cleaned subset is published. No privacy policy, retention period or recipient list for stats.vllm.ai was found, and vllm.ai links no privacy policy (16 of 30). A written deprecation policy with three stages across minor releases, and breaking changes listed in each release's notes (17 of 20). Telemetry is disclosed with three opt-outs, `VLLM_NO_USAGE_STATS`, `DO_NOT_TRACK` and a file. It is on by default and reports every ten minutes (17 of 20)."
          },
          "sources": [
            {
              "what": "repository README, licence and package metadata (shallow clone of main at b027ac8)",
              "url": "https://github.com/vllm-project/vllm",
              "seen": "2026-10-09"
            },
            {
              "what": "security guide, read from the repository",
              "url": "https://github.com/vllm-project/vllm/blob/main/docs/usage/security.md",
              "seen": "2026-10-09"
            },
            {
              "what": "security policy",
              "url": "https://github.com/vllm-project/vllm/blob/main/SECURITY.md",
              "seen": "2026-10-09"
            },
            {
              "what": "vulnerability management",
              "url": "https://github.com/vllm-project/vllm/blob/main/docs/contributing/vulnerability_management.md",
              "seen": "2026-10-09"
            },
            {
              "what": "published security advisories, read through the GitHub API (one page of 100)",
              "url": "https://github.com/vllm-project/vllm/security/advisories",
              "seen": "2026-10-09"
            },
            {
              "what": "advisory GHSA-94f4-hr76-p5j6",
              "url": "https://github.com/vllm-project/vllm/security/advisories/GHSA-94f4-hr76-p5j6",
              "seen": "2026-10-09"
            },
            {
              "what": "front-end arguments and defaults",
              "url": "https://github.com/vllm-project/vllm/blob/main/vllm/entrypoints/launchers/cli_args.py",
              "seen": "2026-10-09"
            },
            {
              "what": "authentication middleware",
              "url": "https://github.com/vllm-project/vllm/blob/main/vllm/entrypoints/serve/middleware/authenticate.py",
              "seen": "2026-10-09"
            },
            {
              "what": "online serving routes",
              "url": "https://github.com/vllm-project/vllm/blob/main/docs/serving/online_serving/README.md",
              "seen": "2026-10-09"
            },
            {
              "what": "OpenAI-compatible server page",
              "url": "https://github.com/vllm-project/vllm/blob/main/docs/serving/online_serving/openai_compatible_server.md",
              "seen": "2026-10-09"
            },
            {
              "what": "usage statistics",
              "url": "https://github.com/vllm-project/vllm/blob/main/docs/usage/usage_stats.md",
              "seen": "2026-10-09"
            },
            {
              "what": "usage statistics source",
              "url": "https://github.com/vllm-project/vllm/blob/main/vllm/usage/usage_lib.py",
              "seen": "2026-10-09"
            },
            {
              "what": "deprecation policy",
              "url": "https://github.com/vllm-project/vllm/blob/main/docs/contributing/deprecation_policy.md",
              "seen": "2026-10-09"
            },
            {
              "what": "release process",
              "url": "https://github.com/vllm-project/vllm/blob/main/docs/contributing/release_process.md",
              "seen": "2026-10-09"
            },
            {
              "what": "v0.31.0 release notes and release dates, read through the GitHub API",
              "url": "https://github.com/vllm-project/vllm/releases/tag/v0.31.0",
              "seen": "2026-10-09"
            },
            {
              "what": "tags and tag dates (git)",
              "url": "https://github.com/vllm-project/vllm/tags",
              "seen": "2026-10-09"
            },
            {
              "what": "open issues and a week of bug reports, read through the GitHub API",
              "url": "https://github.com/vllm-project/vllm/issues",
              "seen": "2026-10-09"
            },
            {
              "what": "quickstart",
              "url": "https://github.com/vllm-project/vllm/blob/main/docs/getting_started/quickstart.md",
              "seen": "2026-10-09"
            },
            {
              "what": "Claude Code integration",
              "url": "https://github.com/vllm-project/vllm/blob/main/docs/serving/integrations/claude_code.md",
              "seen": "2026-10-09"
            },
            {
              "what": "governance",
              "url": "https://github.com/vllm-project/vllm/blob/main/docs/governance/process.md",
              "seen": "2026-10-09"
            },
            {
              "what": "project home page",
              "url": "https://vllm.ai",
              "seen": "2026-10-09"
            },
            {
              "what": "llms.txt of the project site",
              "url": "https://vllm.ai/llms.txt",
              "seen": "2026-10-09"
            },
            {
              "what": "docs home",
              "url": "https://docs.vllm.ai/en/stable/",
              "seen": "2026-10-09"
            },
            {
              "what": "PyTorch Foundation project page",
              "url": "https://pytorch.org/projects/vllm/",
              "seen": "2026-10-09"
            }
          ],
          "openQuestions": [
            "unchecked: whether a completed CI run on main passes. The head commit's checks were still running when read",
            "unchecked: advisories beyond the first page of 100, so the 12-month count of 81 is a floor",
            "unchecked: PyPI download figures and the PyPI release date, since PyPI's robots.txt closes `/pypi/`. Release dates are tag dates",
            "unchecked: when the vllm.ai domain was registered",
            "unchecked: the first release date",
            "No terms or privacy policy is published for the software, vllm.ai or stats.vllm.ai, so both provenance fields are empty and the Apache-2.0 licence stands in",
            "Whether the Linux Foundation's privacy policy is meant to cover the usage statistics sent to stats.vllm.ai was not established",
            "The category describes hardware the owner keeps, a laptop, desktop or home server. vLLM is aimed at GPU servers and clusters, so the fit is with the server end of the category"
          ]
        },
        "negative": -6,
        "negativeNotes": [
          "2026-06-02. GHSA-94f4-hr76-p5j6 (CVE-2026-48746, 9.1), a crafted Host header bypassed the API key check on the OpenAI routes, fixed in 0.22.0. With GHSA-4r2x-xpjr-7cvv (CVE-2026-22778, 9.8) of 2 February 2026, code execution through video decoding fixed in 0.14.1, these are the two critical advisories of the last 12 months. Both were fixed and published with CVEs, so they decay, -3. https://github.com/vllm-project/vllm/security/advisories/GHSA-94f4-hr76-p5j6; https://github.com/vllm-project/vllm/security/advisories/GHSA-4r2x-xpjr-7cvv",
          "2026-10-06. GHSA-h3rc-6mm3-gc2m (8.1), a request field could select the processor code a server started with `--trust-remote-code` imports, fixed in 0.31.0, one of 50 advisories published since 11 July 2026 (10 high, 36 medium, 4 low), most of them requests that crash or exhaust the engine. All name a fixed version, and eleven were published on 9 October 2026 months after their fixes, -3. https://github.com/vllm-project/vllm/security/advisories/GHSA-h3rc-6mm3-gc2m; https://github.com/vllm-project/vllm/security/advisories"
        ],
        "verdict": "Apache-2.0 software with a release about every two weeks, each with notes that list breaking changes and security fixes. The optional API key covers only some path prefixes, so `/invocations` and control routes such as `/pause` answer without it, and at least 81 security advisories were published in the 12 months to 9 October 2026.",
        "bestFor": "An owner with a GPU server who wants many concurrent requests against one open-weight model behind OpenAI or Anthropic compatible routes.",
        "strengths": [
          "OpenAI chat, completions, responses and embeddings, Anthropic `/v1/messages`, Cohere embed and rerank, transcription and `/v1/systemone` from one server",
          "Apache-2.0, with a written three-stage deprecation policy and release notes that carry a breaking changes section",
          "Eight stable releases between 12 July and 2 October 2026, and v0.31.0 lists 717 commits from 307 contributors",
          "A 650-line security guide names every route the API key does and does not protect, and the limits of multi-tenant use",
          "Usage statistics are documented field by field, with `VLLM_NO_USAGE_STATS`, `DO_NOT_TRACK` or a file as opt-outs"
        ],
        "weaknesses": [
          "`--api-key` guards only the `/v1`, `/v2`, `/inference` and `/cohere` prefixes. `/invocations`, `/pooling`, `/classify`, `/score`, `/rerank`, `/pause` and `/update_weights` answer without it",
          "No key by default, the server binds every interface when `--host` is unset, and CORS allows any origin",
          "At least 81 GitHub security advisories in 12 months, two critical, most of them remote crashes or resource exhaustion",
          "Pre-1.0 (0.31.0), with breaking changes in each fortnightly release and compatibility kept for a limited number of minor versions",
          "Usage statistics are sent to stats.vllm.ai by default, and no privacy policy or retention period for them was found"
        ],
        "agentNotes": [
          "Put a reverse proxy that allowlists routes in front of the server. `--api-key` leaves `/invocations` and the control routes open",
          "Pass `--host 127.0.0.1` for single-machine use. With no `--host` the server listens on every interface",
          "Set `VLLM_NO_USAGE_STATS=1` or `DO_NOT_TRACK=1` before starting if nothing should be sent to stats.vllm.ai",
          "Start with `--enable-auto-tool-choice` and the `--tool-call-parser` for the model before sending tools. Tool calling is off without them",
          "Send `max_tokens` on every request, and read the breaking changes section of the release notes before upgrading a minor version"
        ],
        "metrics": {
          "kind": "local",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "C",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 57.7
          }
        ],
        "editorialScores": {
          "ergonomics": 64,
          "maintenance": 88,
          "payments": 60,
          "reliability": 62,
          "schema": 68,
          "security": 50,
          "transparency": 80
        },
        "provenanceScore": 53
      },
      "connect": {
        "install": "uv pip install vllm --torch-backend=auto\nvllm serve Qwen/Qwen2.5-1.5B-Instruct   # listens on port 8000",
        "http": "curl http://localhost:8000/v1/chat/completions \\\n    -H \"Content-Type: application/json\" \\\n    -d '{\n        \"model\": \"Qwen/Qwen2.5-1.5B-Instruct\",\n        \"messages\": [\n            {\"role\": \"system\", \"content\": \"You are a helpful assistant.\"},\n            {\"role\": \"user\", \"content\": \"Who won the world series in 2020?\"}\n        ]\n    }'",
        "claudeCode": "ANTHROPIC_BASE_URL=http://localhost:8000 \\\nANTHROPIC_API_KEY=dummy \\\nANTHROPIC_AUTH_TOKEN=dummy \\\nANTHROPIC_DEFAULT_OPUS_MODEL=my-model \\\nANTHROPIC_DEFAULT_SONNET_MODEL=my-model \\\nANTHROPIC_DEFAULT_HAIKU_MODEL=my-model \\\nclaude"
      },
      "letme": {
        "capability": "https://letme.dev/inference.local",
        "tool": "https://letme.dev/vllm"
      },
      "notable": [
        "The API key check covers only paths under `/v1`, `/v2`, `/inference` and `/cohere`. The security guide lists `/invocations`, `/generative_scoring`, `/pooling`, `/classify`, `/score`, `/rerank`, `/pause`, `/resume`, `/abort_requests`, `/update_weights`, `/tokenize` and others as answering without it (https://github.com/vllm-project/vllm/blob/main/docs/usage/security.md)",
        "With `--host` unset the launcher binds the empty address, which is every interface, on port 8000. `allowed_origins`, `allowed_methods` and `allowed_headers` default to `*`, with `allow_credentials` off (https://github.com/vllm-project/vllm/blob/main/vllm/entrypoints/launchers/cli_args.py; https://github.com/vllm-project/vllm/blob/main/vllm/entrypoints/launchers/launcher.py)",
        "The GitHub advisories list returned 100 published advisories on one page, 81 of them published since 9 October 2025 (2 critical, 19 high, 56 medium, 4 low) and 50 since 11 July 2026. More may sit on later pages (https://github.com/vllm-project/vllm/security/advisories)",
        "GHSA-94f4-hr76-p5j6 (CVE-2026-48746, 9.1), published 2 June 2026, let a crafted Host header bypass the API key check, fixed in 0.22.0. GHSA-4r2x-xpjr-7cvv (CVE-2026-22778, 9.8), published 2 February 2026, was code execution through video decoding, fixed in 0.14.1",
        "Eleven advisories were published on 9 October 2026, five of them for flaws whose fixed version is 0.24.0, tagged 28 June 2026 (https://github.com/vllm-project/vllm/security/advisories)",
        "Usage statistics are on by default and post hardware, platform, model architecture and engine settings with a random UUID to stats.vllm.ai, with a heartbeat every ten minutes (https://github.com/vllm-project/vllm/blob/main/docs/usage/usage_stats.md; https://github.com/vllm-project/vllm/blob/main/vllm/usage/usage_lib.py)",
        "The gRPC services have no authentication, authorisation or encryption and are off unless `--grpc-port` is set (https://github.com/vllm-project/vllm/blob/main/docs/usage/security.md)",
        "pytorch.org lists vLLM as a PyTorch Foundation project and says the University of California, Berkeley contributed it to the Linux Foundation in July 2024 (https://pytorch.org/projects/vllm/)",
        "vllm.ai says Python 3.10 or later is required, while `pyproject.toml` requires 3.11 to 3.14 and the quickstart says 3.11 to 3.14 (https://vllm.ai; https://github.com/vllm-project/vllm/blob/main/pyproject.toml)"
      ],
      "area": "models",
      "details": [
        {
          "label": "Interfaces",
          "value": "`vllm serve` HTTP server on port 8000, optional gRPC Inference and Control services (`--grpc-port`), the `vllm` Python library (`LLM`, `SamplingParams`), and Docker images `vllm/vllm-openai` for CUDA, ROCm, CPU and XPU"
        },
        {
          "label": "Routes",
          "value": "OpenAI `/v1/chat/completions`, `/v1/completions`, `/v1/responses`, `/v1/embeddings`, `/v1/audio/transcriptions`, `/v1/audio/translations`, `/v1/realtime` and `/v1/models`, Anthropic `/v1/messages` and `/v1/messages/count_tokens`, Cohere `/v2/embed` and `/v2/rerank`, `/v1/systemone`, `/pooling`, `/classify`, `/score`, `/tokenize`, `/detokenize`, `/health`, `/metrics`, and SageMaker `/invocations`"
        },
        {
          "label": "Credentials",
          "value": "None by default. `--api-key` (one or several) or `VLLM_API_KEY`, sent as `Authorization: Bearer`, checked only under `/v1`, `/v2`, `/inference` and `/cohere`. No scopes. gRPC has none"
        },
        {
          "label": "Network defaults",
          "value": "Binds every interface on port 8000 when `--host` is unset. CORS origins, methods and headers `*`, credentials off. TLS through `--ssl-keyfile` and `--ssl-certfile`. Inter-node traffic is unencrypted"
        },
        {
          "label": "Hardware",
          "value": "NVIDIA, AMD and Intel GPUs and x86, Arm and PowerPC CPUs per the README, with plugins for Google TPU, Intel Gaudi, IBM Spyre, Huawei Ascend, Apple Silicon (vLLM-Metal) and others. Linux, Python 3.11 to 3.14"
        },
        {
          "label": "Models",
          "value": "Hugging Face model repositories, more than 200 architectures per the README, including multimodal, embedding, reranking and speech recognition models. Downloads from Hugging Face or ModelScope"
        },
        {
          "label": "Agent tools",
          "value": "Tool calling with per-model parsers, structured outputs (JSON schema, choice, regex, grammar), reasoning parsers, and tool servers including MCP through the Responses API, off by default"
        },
        {
          "label": "Telemetry",
          "value": "Usage statistics to stats.vllm.ai, on by default. Off with `VLLM_NO_USAGE_STATS=1`, `DO_NOT_TRACK=1` or the file `~/.config/vllm/do_not_track`"
        },
        {
          "label": "Install",
          "value": "`uv pip install vllm --torch-backend=auto` or pip from PyPI, a ROCm wheel index at wheels.vllm.ai, and Docker images"
        },
        {
          "label": "Releases in 90 days",
          "value": "Eight stable releases from v0.25.1 (12 July 2026) to v0.31.0 (tagged 2 October 2026, notes published 5 October), plus release candidates"
        },
        {
          "label": "Governance",
          "value": "A PyTorch Foundation project under the Linux Foundation, contributed by UC Berkeley in July 2024. Eleven project leads form the technical steering committee"
        },
        {
          "label": "Security record",
          "value": "At least 100 published GitHub advisories, 81 in the 12 months to 9 October 2026 (2 critical, 19 high). A vulnerability management team, private reporting and CVEs"
        }
      ],
      "provenance": {
        "legalEntity": "The Linux Foundation (vLLM is a PyTorch Foundation project)",
        "domain": "vllm.ai",
        "domainRegistered": "",
        "endpointOnVendorDomain": null,
        "terms": "",
        "privacy": "",
        "statusPage": "",
        "changelog": "https://github.com/vllm-project/vllm/releases",
        "securityTxt": "none",
        "checked": "2026-10-09",
        "notes": [
          "vllm.ai links no terms and no privacy policy, and its footer reads © 2026 vLLM. The project publishes none for the software or for stats.vllm.ai, so the Apache-2.0 licence stands in for terms.",
          "pytorch.org/projects/vllm/ lists vLLM among PyTorch Foundation projects and says UC Berkeley contributed it to the Linux Foundation in July 2024. The Linux Foundation's policies are linked from that page and are not specific to vLLM.",
          "vllm.ai/.well-known/security.txt and docs.vllm.ai/.well-known/security.txt return 404. SECURITY.md asks for private reports through GitHub.",
          "There's no shared hosted endpoint. The server runs on the owner's hardware. The software posts usage statistics to stats.vllm.ai unless turned off."
        ],
        "score": 53,
        "checks": [
          {
            "check": "Legal entity named",
            "value": "The Linux Foundation (vLLM is a PyTorch Foundation project)",
            "points": 20,
            "max": 20,
            "state": "ok"
          },
          {
            "check": "Domain age",
            "value": "vllm.ai, no registry record we could read",
            "points": 0,
            "max": 15,
            "state": "no"
          },
          {
            "check": "Endpoint on the vendor's domain",
            "value": "no hosted endpoint",
            "points": 0,
            "max": 0,
            "state": "na"
          },
          {
            "check": "Terms of service",
            "value": "nothing hosted, so the Apache-2.0 licence stands in",
            "points": 10,
            "max": 10,
            "state": "ok"
          },
          {
            "check": "Privacy policy",
            "value": "nothing hosted, not scored",
            "points": 0,
            "max": 0,
            "state": "na"
          },
          {
            "check": "Status page",
            "value": "not found",
            "points": 0,
            "max": 10,
            "state": "no"
          },
          {
            "check": "Changelog",
            "value": "published",
            "points": 10,
            "max": 10,
            "state": "ok"
          },
          {
            "check": "security.txt",
            "value": "not found",
            "points": 0,
            "max": 10,
            "state": "no"
          }
        ]
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/vllm.json",
      "live": {
        "slug": "vllm",
        "versions": [
          {
            "registry": "github",
            "name": "vllm-project/vllm",
            "version": "v0.31.0",
            "released": "2026-10-05",
            "seenAt": "2026-10-09T17:27:30.444059802Z"
          },
          {
            "registry": "pypi",
            "name": "vllm",
            "version": "0.31.0",
            "released": "2026-10-05",
            "seenAt": "2026-10-09T17:27:30.259454009Z"
          }
        ],
        "githubStars": 93457,
        "pypiWeekly": 444466,
        "updatedAt": "2026-10-09T17:27:30.444059802Z"
      }
    },
    "verify": {
      "accepts": "a page on vllm.ai or one of its subdomains, or the README of github.com/vllm-project/vllm",
      "badgeUrl": "https://www.anchorterminal.com/badges/vllm.svg",
      "body": {
        "slug": "vllm",
        "url": "the page with the badge or the link"
      },
      "docs": "https://www.anchorterminal.com/builders/#verify",
      "effect": "none, it never changes a grade, rank or review",
      "endpoint": "https://www.anchorterminal.com/api/v1/verify",
      "listingUrl": "https://www.anchorterminal.com/tools/vllm",
      "mcpTool": "verify_listing",
      "recheck": "weekly; two failed checks in a row and it lapses, a later pass restores it",
      "snippets": {
        "html": "\u003ca href=\"https://www.anchorterminal.com/tools/vllm\"\u003e\u003cimg src=\"https://www.anchorterminal.com/badges/vllm.svg\" alt=\"vLLM on Anchor Terminal\" height=\"20\"\u003e\u003c/a\u003e",
        "markdown": "[![vLLM on Anchor Terminal](https://www.anchorterminal.com/badges/vllm.svg)](https://www.anchorterminal.com/tools/vllm)",
        "link": "\u003ca href=\"https://www.anchorterminal.com/tools/vllm\"\u003evLLM on Anchor Terminal\u003c/a\u003e"
      }
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/tools/vllm",
    "json": "https://www.anchorterminal.com/tools/vllm.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/tools/vllm.md",
    "slim": "https://www.anchorterminal.com/tools/vllm.min.md"
  },
  "markdown": "## Overview\n\n**Grade C · 57.7/100 · rank #600 of 950 · #8 in Local AI · not agent-ready · confidence medium**\n\n\n## Assessment\n\nApache-2.0 software with a release about every two weeks, each with notes that list breaking changes and security fixes. The optional API key covers only some path prefixes, so `/invocations` and control routes such as `/pause` answer without it, and at least 81 security advisories were published in the 12 months to 9 October 2026.\n\n## Facts\n\n| Field | Value |\n| --- | --- |\n| Vendor | vLLM project (PyTorch Foundation) (https://vllm.ai) |\n| Kind | HTTP API |\n| Category | Local AI (https://www.anchorterminal.com/categories/local-ai) |\n| Transport | HTTP |\n| Auth | None · No credential by default. `--api-key` (one or several keys) or `VLLM_API_KEY` turns on a Bearer check for paths under `/v1`, `/v2`, `/inference` and `/cohere` only, so `/invocations`, `/pooling`, `/classify`, `/score`, `/rerank` and control routes such as `/pause` stay open. Keys have no scopes and change with a restart. The key is read from the `Authorization` header, never the query string. gRPC has no authentication (https://github.com/vllm-project/vllm/blob/main/docs/usage/security.md). |\n| Pricing | Free (Free · OSS) · Free under Apache-2.0, with no account, key or card. Nothing is sold by the project. You pay for your own hardware and electricity. |\n| x402 | No · No x402, MPP or L402 in the docs or the source (checked 2026-10-09). |\n| Licence | Apache-2.0 |\n| Packages | pypi: `vllm`; oci: `vllm/vllm-openai` |\n| Source | https://github.com/vllm-project/vllm |\n| Docs | https://docs.vllm.ai/en/stable/ |\n| llms.txt | not found |\n| Last release | 2026-10-02 |\n| GitHub stars | 93,444 (as of 2026-10-09) |\n| Interfaces | `vllm serve` HTTP server on port 8000, optional gRPC Inference and Control services (`--grpc-port`), the `vllm` Python library (`LLM`, `SamplingParams`), and Docker images `vllm/vllm-openai` for CUDA, ROCm, CPU and XPU |\n| Routes | OpenAI `/v1/chat/completions`, `/v1/completions`, `/v1/responses`, `/v1/embeddings`, `/v1/audio/transcriptions`, `/v1/audio/translations`, `/v1/realtime` and `/v1/models`, Anthropic `/v1/messages` and `/v1/messages/count_tokens`, Cohere `/v2/embed` and `/v2/rerank`, `/v1/systemone`, `/pooling`, `/classify`, `/score`, `/tokenize`, `/detokenize`, `/health`, `/metrics`, and SageMaker `/invocations` |\n| Credentials | None by default. `--api-key` (one or several) or `VLLM_API_KEY`, sent as `Authorization: Bearer`, checked only under `/v1`, `/v2`, `/inference` and `/cohere`. No scopes. gRPC has none |\n| Network defaults | Binds every interface on port 8000 when `--host` is unset. CORS origins, methods and headers `*`, credentials off. TLS through `--ssl-keyfile` and `--ssl-certfile`. Inter-node traffic is unencrypted |\n| Hardware | NVIDIA, AMD and Intel GPUs and x86, Arm and PowerPC CPUs per the README, with plugins for Google TPU, Intel Gaudi, IBM Spyre, Huawei Ascend, Apple Silicon (vLLM-Metal) and others. Linux, Python 3.11 to 3.14 |\n| Models | Hugging Face model repositories, more than 200 architectures per the README, including multimodal, embedding, reranking and speech recognition models. Downloads from Hugging Face or ModelScope |\n| Agent tools | Tool calling with per-model parsers, structured outputs (JSON schema, choice, regex, grammar), reasoning parsers, and tool servers including MCP through the Responses API, off by default |\n| Telemetry | Usage statistics to stats.vllm.ai, on by default. Off with `VLLM_NO_USAGE_STATS=1`, `DO_NOT_TRACK=1` or the file `~/.config/vllm/do_not_track` |\n| Install | `uv pip install vllm --torch-backend=auto` or pip from PyPI, a ROCm wheel index at wheels.vllm.ai, and Docker images |\n| Releases in 90 days | Eight stable releases from v0.25.1 (12 July 2026) to v0.31.0 (tagged 2 October 2026, notes published 5 October), plus release candidates |\n| Governance | A PyTorch Foundation project under the Linux Foundation, contributed by UC Berkeley in July 2024. Eleven project leads form the technical steering committee |\n| Security record | At least 100 published GitHub advisories, 81 in the 12 months to 9 October 2026 (2 critical, 19 high). A vulnerability management team, private reporting and CVEs |\n| Capabilities | inference.local, inference.open-weights, embed.text, rerank, speech.stt, inference.decision, agent.mcp-client |\n| Tags | open-source, local, self-hosted, free, no-card, openai-compatible, docker, pre-1.0, telemetry-default-on |\n| JSON | https://www.anchorterminal.com/api/v1/tools/vllm.json |\n\n## Score breakdown (methodology v0.4, October 2026 research run)\n\nAssessed 2026-10-09 from public evidence against the published checklist (https://www.anchorterminal.com/benchmark/#checklist). Confidence: medium. Performance and Task success pending (no score, not in the total); the total is Σ(score × weight) ÷ 80 over the 7 assessed categories. \"This run\" is each category's share of the 100 points.\n\n| Category | Weight | This run | Score (0–100) | Points |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% | 20 | 62 | 12.4 |\n| Performance | 10% | pending | pending | n/a |\n| Schema \u0026 documentation | 13% | 16.2 | 68 | 11.1 |\n| Agent ergonomics | 13% | 16.2 | 64 | 10.4 |\n| Security \u0026 auth | 14% | 17.5 | 50 | 8.8 |\n| Payments \u0026 pricing | 10% | 12.5 | 60 | 7.5 |\n| Task success | 10% | pending | pending | n/a |\n| Maintenance \u0026 community | 7% | 8.8 | 88 | 7.7 |\n| Transparency \u0026 trust (editorial 80, provenance 53) | 7% | 8.8 | 67 | 5.9 |\n| Negative events | up to −15 | up to −15 | 2026-06-02. GHSA-94f4-hr76-p5j6 (CVE-2026-48746, 9.1), a crafted Host header bypassed the API key check on the OpenAI routes, fixed in 0.22.0. With GHSA-4r2x-xpjr-7cvv (CVE-2026-22778, 9.8) of 2 February 2026, code execution through video decoding fixed in 0.14.1, these are the two critical advisories of the last 12 months. Both were fixed and published with CVEs, so they decay, -3. https://github.com/vllm-project/vllm/security/advisories/GHSA-94f4-hr76-p5j6; https://github.com/vllm-project/vllm/security/advisories/GHSA-4r2x-xpjr-7cvv 2026-10-06. GHSA-h3rc-6mm3-gc2m (8.1), a request field could select the processor code a server started with `--trust-remote-code` imports, fixed in 0.31.0, one of 50 advisories published since 11 July 2026 (10 high, 36 medium, 4 low), most of them requests that crash or exhaust the engine. All name a fixed version, and eleven were published on 9 October 2026 months after their fixes, -3. https://github.com/vllm-project/vllm/security/advisories/GHSA-h3rc-6mm3-gc2m; https://github.com/vllm-project/vllm/security/advisories  | -6 |\n| **Total** | | | | **57.7 → C** |\n\n### Why each score\n\n- Reliability 62: Read with the local-software lines, since vLLM runs on the owner's hardware with no hosted service. Installs from PyPI (`vllm`) and Docker images, with Linux and Python 3.11 to 3.14 stated in the quickstart and `pyproject.toml` (20). A public Buildkite pipeline and GitHub workflows report 74 checks on each commit to main. On the head commit of 9 October 2026 they were still running, with the finished steps passing, and we did not read a completed run (18 of 25). 2,592 open issues under a stale bot that closes after 90 and 30 days. Of 48 bug reports filed from 22 to 28 September 2026, 42 had a comment and 13 were closed by 9 October. Many of the 81 advisories of the last 12 months are requests that stop the engine (14 of 25). Release notes carry a Breaking Changes and Deprecations section, and a written policy sets three stages for removals, but each fortnightly minor release breaks something and compatibility holds for a limited number of minor versions (10 of 15). Version 0.31.0, pre-1.0 (0).\n- Performance: Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes.\n- Schema \u0026 documentation 68: The server is a FastAPI application with Pydantic request models, so a running instance serves its own OpenAPI document and Swagger page unless `--disable-fastapi-docs` is set, and the gRPC services have published proto files. No static OpenAPI file is published in the docs or the repository (15 of 25). docs.vllm.ai/llms.txt returns 404. vllm.ai/llms.txt lists site pages and blog posts, and the docs are Markdown files in the repository (5 of 10). The serving pages say which model types each route applies to, which OpenAI fields are ignored, and which routes must not be exposed (15 of 20). Typed request models, with structured outputs by JSON schema, choice, regex or grammar (12 of 15). Examples for most routes in the docs and the `examples` folder. The error shape (message, type, param, code) is defined in the source, and no error reference page was found (9 of 15). GitHub releases with full notes about every two weeks and vllm.ai/releases. The HTTP API has no version of its own beyond the `/v1` prefix (12 of 15).\n- Agent ergonomics 64: Read for an API. `max_tokens`, structured outputs, log probabilities and streaming control the size and shape of a reply (18 of 25). `/tokenize` and `/v1/messages/count_tokens` count tokens before a call and `/load` reports server load. `/v1/models` is not paged (14 of 20). Errors follow the OpenAI shape with a type, a param and a code, with no documented list of codes (12 of 20). Generation is stateless and safe to repeat, `X-Request-Id` can be echoed with `--enable-request-id-headers`, and `/health` is public. No retry guidance was found (11 of 20). `vllm serve \u003cmodel\u003e` starts a working server and OpenAI and Anthropic clients work against it. There is no client library of its own, tool calling needs `--enable-auto-tool-choice` and a parser chosen for the model, and sampling defaults come from the model's `generation_config.json` unless overridden (9 of 15).\n- Security \u0026 auth 50: Read with the tool checklist. Optional static keys from `--api-key` or `VLLM_API_KEY`, compared in constant time and read only from the `Authorization` header. They are off by default, have no scopes, and are checked only under `/v1`, `/v2`, `/inference` and `/cohere`, so `/invocations` runs the same inference without a key. With `--host` unset the server binds every interface, and gRPC has no authentication (10 of 30). Development routes, runtime LoRA loading, tool servers and endpoint plugins are off by default. `/pause`, `/abort_requests` and `/update_weights` answer without a key on generation servers, and there is no read-only mode (6 of 20). The security guide covers model-generated code in the demo code interpreter, allowed domains for media URLs, cache salting and the lack of tenant isolation (10 of 15). Access logs, `--enable-log-requests`, Prometheus `/metrics` and OpenTelemetry examples, with no per-key record (9 of 15). SECURITY.md with severity classes, a named vulnerability management team, private reporting through GitHub, CVEs and a prenotification group, and at least 100 published advisories. No security.txt and no bounty found (15 of 20).\n- Payments \u0026 pricing 60: Read with the self-hosted rule. No x402, MPP or L402 in the docs or the source (0). Free under Apache-2.0 with no account, key or card, and nothing to buy, so 20, 20 and 20 on the last three lines.\n- Task success: Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored.\n- Maintenance \u0026 community 88: v0.31.0 was tagged on 2 October 2026 and its notes published on 5 October (30). Eight stable releases from v0.25.1 on 12 July to v0.31.0, with release candidates between them (20). The v0.31.0 notes count 717 commits from 307 contributors. 2,592 open issues, and 42 of 48 bug reports from the week of 22 September had a comment by 9 October. Four bug reports filed on 9 October had none yet (19 of 25). The Python library ships in the same package as the server, and a Rust proto crate is tagged separately (proto-v0.5.0 on 9 October). The server has no client library of its own (10 of 15). Dependabot runs weekly for pip and GitHub Actions, with pre-commit and Buildkite checks on each commit (9 of 10).\n- Transparency \u0026 trust 67: The editorial half. Apache-2.0, all of it public (30). The usage statistics page lists what is collected, links the collecting code, keeps a local copy at `~/.config/vllm/usage_stats.json` and says a cleaned subset is published. No privacy policy, retention period or recipient list for stats.vllm.ai was found, and vllm.ai links no privacy policy (16 of 30). A written deprecation policy with three stages across minor releases, and breaking changes listed in each release's notes (17 of 20). Telemetry is disclosed with three opt-outs, `VLLM_NO_USAGE_STATS`, `DO_NOT_TRACK` and a file. It is on by default and reports every ten minutes (17 of 20).\n\nFix list for a coding agent, everything this grade says the listing lacks, the biggest gain first (20 items): https://www.anchorterminal.com/fixes/vllm.md (JSON https://www.anchorterminal.com/fixes/vllm.json)\n\n### What we couldn't check\n\n- unchecked: whether a completed CI run on main passes. The head commit's checks were still running when read\n- unchecked: advisories beyond the first page of 100, so the 12-month count of 81 is a floor\n- unchecked: PyPI download figures and the PyPI release date, since PyPI's robots.txt closes `/pypi/`. Release dates are tag dates\n- unchecked: when the vllm.ai domain was registered\n- unchecked: the first release date\n- No terms or privacy policy is published for the software, vllm.ai or stats.vllm.ai, so both provenance fields are empty and the Apache-2.0 licence stands in\n- Whether the Linux Foundation's privacy policy is meant to cover the usage statistics sent to stats.vllm.ai was not established\n- The category describes hardware the owner keeps, a laptop, desktop or home server. vLLM is aimed at GPU servers and clusters, so the fit is with the server end of the category\n\n### Sources\n\n- repository README, licence and package metadata (shallow clone of main at b027ac8): \u003chttps://github.com/vllm-project/vllm\u003e (seen 2026-10-09)\n- security guide, read from the repository: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/usage/security.md\u003e (seen 2026-10-09)\n- security policy: \u003chttps://github.com/vllm-project/vllm/blob/main/SECURITY.md\u003e (seen 2026-10-09)\n- vulnerability management: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/contributing/vulnerability_management.md\u003e (seen 2026-10-09)\n- published security advisories, read through the GitHub API (one page of 100): \u003chttps://github.com/vllm-project/vllm/security/advisories\u003e (seen 2026-10-09)\n- advisory GHSA-94f4-hr76-p5j6: \u003chttps://github.com/vllm-project/vllm/security/advisories/GHSA-94f4-hr76-p5j6\u003e (seen 2026-10-09)\n- front-end arguments and defaults: \u003chttps://github.com/vllm-project/vllm/blob/main/vllm/entrypoints/launchers/cli_args.py\u003e (seen 2026-10-09)\n- authentication middleware: \u003chttps://github.com/vllm-project/vllm/blob/main/vllm/entrypoints/serve/middleware/authenticate.py\u003e (seen 2026-10-09)\n- online serving routes: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/serving/online_serving/README.md\u003e (seen 2026-10-09)\n- OpenAI-compatible server page: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/serving/online_serving/openai_compatible_server.md\u003e (seen 2026-10-09)\n- usage statistics: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/usage/usage_stats.md\u003e (seen 2026-10-09)\n- usage statistics source: \u003chttps://github.com/vllm-project/vllm/blob/main/vllm/usage/usage_lib.py\u003e (seen 2026-10-09)\n- deprecation policy: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/contributing/deprecation_policy.md\u003e (seen 2026-10-09)\n- release process: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/contributing/release_process.md\u003e (seen 2026-10-09)\n- v0.31.0 release notes and release dates, read through the GitHub API: \u003chttps://github.com/vllm-project/vllm/releases/tag/v0.31.0\u003e (seen 2026-10-09)\n- tags and tag dates (git): \u003chttps://github.com/vllm-project/vllm/tags\u003e (seen 2026-10-09)\n- open issues and a week of bug reports, read through the GitHub API: \u003chttps://github.com/vllm-project/vllm/issues\u003e (seen 2026-10-09)\n- quickstart: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/getting_started/quickstart.md\u003e (seen 2026-10-09)\n- Claude Code integration: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/serving/integrations/claude_code.md\u003e (seen 2026-10-09)\n- governance: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/governance/process.md\u003e (seen 2026-10-09)\n- project home page: \u003chttps://vllm.ai\u003e (seen 2026-10-09)\n- llms.txt of the project site: \u003chttps://vllm.ai/llms.txt\u003e (seen 2026-10-09)\n- docs home: \u003chttps://docs.vllm.ai/en/stable/\u003e (seen 2026-10-09)\n- PyTorch Foundation project page: \u003chttps://pytorch.org/projects/vllm/\u003e (seen 2026-10-09)\n\n## Who's behind it (provenance 53/100, checked 2026-10-09)\n\n| Check | Finding | Points |\n| --- | --- | --- |\n| Legal entity named | The Linux Foundation (vLLM is a PyTorch Foundation project) | 20/20 |\n| Domain age | vllm.ai, no registry record we could read | 0/15 |\n| Endpoint on the vendor's domain | no hosted endpoint | n/a |\n| Terms of service | nothing hosted, so the Apache-2.0 licence stands in | 10/10 |\n| Privacy policy | nothing hosted, not scored | n/a |\n| Status page | not found | 0/10 |\n| Changelog | published | 10/10 |\n| security.txt | not found | 0/10 |\n\nvllm.ai links no terms and no privacy policy, and its footer reads © 2026 vLLM. The project publishes none for the software or for stats.vllm.ai, so the Apache-2.0 licence stands in for terms.\n\npytorch.org/projects/vllm/ lists vLLM among PyTorch Foundation projects and says UC Berkeley contributed it to the Linux Foundation in July 2024. The Linux Foundation's policies are linked from that page and are not specific to vLLM.\n\nvllm.ai/.well-known/security.txt and docs.vllm.ai/.well-known/security.txt return 404. SECURITY.md asks for private reports through GitHub.\n\nThere's no shared hosted endpoint. The server runs on the owner's hardware. The software posts usage statistics to stats.vllm.ai unless turned off.\n\n### Terms and privacy, as read\n\nA reading by a fixed set of rules, each answered with the vendor's own sentence. Not legal advice.\n\n**Terms of service**. Nothing is hosted by the vendor, so there are no terms of service to read. The Apache-2.0 licence stands in and the check scores in full.\n\n\n**Privacy policy**. Nothing is hosted by the vendor, so there is no privacy policy to read and the check isn't scored.\n\n\n## Live (updated 2026-10-09 17:27 UTC)\n\n- github `vllm-project/vllm` v0.31.0, released 2026-10-05\n- pypi `vllm` 0.31.0, released 2026-10-05\n- Always current: https://www.anchorterminal.com/api/v1/live/vllm.json\n\n## Probe metrics\n\nNot measured yet. Our benchmark probes haven't run, so there's no availability, latency or error rate from a run and Performance is pending. Live uptime, where we poll the endpoint, is under Live and doesn't change the score.\n\n## Strengths\n\n- OpenAI chat, completions, responses and embeddings, Anthropic `/v1/messages`, Cohere embed and rerank, transcription and `/v1/systemone` from one server\n- Apache-2.0, with a written three-stage deprecation policy and release notes that carry a breaking changes section\n- Eight stable releases between 12 July and 2 October 2026, and v0.31.0 lists 717 commits from 307 contributors\n- A 650-line security guide names every route the API key does and does not protect, and the limits of multi-tenant use\n- Usage statistics are documented field by field, with `VLLM_NO_USAGE_STATS`, `DO_NOT_TRACK` or a file as opt-outs\n\n## Weaknesses\n\n- `--api-key` guards only the `/v1`, `/v2`, `/inference` and `/cohere` prefixes. `/invocations`, `/pooling`, `/classify`, `/score`, `/rerank`, `/pause` and `/update_weights` answer without it\n- No key by default, the server binds every interface when `--host` is unset, and CORS allows any origin\n- At least 81 GitHub security advisories in 12 months, two critical, most of them remote crashes or resource exhaustion\n- Pre-1.0 (0.31.0), with breaking changes in each fortnightly release and compatibility kept for a limited number of minor versions\n- Usage statistics are sent to stats.vllm.ai by default, and no privacy policy or retention period for them was found\n\n## Before you call it (notes for agents)\n\n1. Put a reverse proxy that allowlists routes in front of the server. `--api-key` leaves `/invocations` and the control routes open\n2. Pass `--host 127.0.0.1` for single-machine use. With no `--host` the server listens on every interface\n3. Set `VLLM_NO_USAGE_STATS=1` or `DO_NOT_TRACK=1` before starting if nothing should be sent to stats.vllm.ai\n4. Start with `--enable-auto-tool-choice` and the `--tool-call-parser` for the model before sending tools. Tool calling is off without them\n5. Send `max_tokens` on every request, and read the breaking changes section of the release notes before upgrading a minor version\n\n## Connect\n\nInstall:\n\n```bash\nuv pip install vllm --torch-backend=auto\nvllm serve Qwen/Qwen2.5-1.5B-Instruct   # listens on port 8000\n```\n\nFirst request:\n\n```bash\ncurl http://localhost:8000/v1/chat/completions \\\n    -H \"Content-Type: application/json\" \\\n    -d '{\n        \"model\": \"Qwen/Qwen2.5-1.5B-Instruct\",\n        \"messages\": [\n            {\"role\": \"system\", \"content\": \"You are a helpful assistant.\"},\n            {\"role\": \"user\", \"content\": \"Who won the world series in 2020?\"}\n        ]\n    }'\n```\n\nClaude Code:\n\n```bash\nANTHROPIC_BASE_URL=http://localhost:8000 \\\nANTHROPIC_API_KEY=dummy \\\nANTHROPIC_AUTH_TOKEN=dummy \\\nANTHROPIC_DEFAULT_OPUS_MODEL=my-model \\\nANTHROPIC_DEFAULT_SONNET_MODEL=my-model \\\nANTHROPIC_DEFAULT_HAIKU_MODEL=my-model \\\nclaude\n```\n\nThrough letme (picks today, calling later): https://letme.dev/vllm. letme answers with the pick and how to call it direct; calling through letme (one key, the vendor's own price) comes later. How it works: https://www.anchorterminal.com/letme/index.md\n\n## Similar tools\n\nRanked by shared capabilities, then score. Same-category tools with no shared capability key are listed last.\n\n| Tool | Grade | Score | Rank | Shared capabilities | x402 | Markdown |\n| --- | --- | --- | --- | --- | --- | --- |\n| LocalAI | B | 68 | 232 | inference.local, inference.open-weights, agent.mcp-client, embed.text, rerank, speech.stt | no | https://www.anchorterminal.com/tools/localai.md |\n| llama.cpp | C | 60.2 | 525 | inference.local, inference.open-weights, embed.text, rerank, inference.decision, agent.mcp-client | no | https://www.anchorterminal.com/tools/llama-cpp.md |\n| Lemonade | B | 63.8 | 372 | inference.local, inference.open-weights, embed.text, rerank, speech.stt | no | https://www.anchorterminal.com/tools/lemonade.md |\n| KoboldCpp | C | 60.5 | 510 | inference.local, inference.open-weights, agent.mcp-client, embed.text, speech.stt | no | https://www.anchorterminal.com/tools/koboldcpp.md |\n| Cloudflare Workers AI | B | 68.3 | 224 | inference.open-weights, embed.text, rerank, speech.stt | no | https://www.anchorterminal.com/tools/cloudflare-workers-ai.md |\n| DeepInfra | B | 63 | 411 | inference.open-weights, embed.text, rerank, speech.stt | no | https://www.anchorterminal.com/tools/deepinfra.md |\n\n## Panel reviews (0)\n\nReviewed by the Anchor panel (https://www.anchorterminal.com/reviewers/index.md): .\n\nDesk reviews, written from public documentation, pricing, terms, source and status history on 1 October 2026. No calls made. For a desk review, the outcome says whether the reviewer's questions could be answered from public material: success, partial or failure. How reviews work: https://www.anchorterminal.com/reviews/how-it-works.md\n\n## Notable\n\n- The API key check covers only paths under `/v1`, `/v2`, `/inference` and `/cohere`. The security guide lists `/invocations`, `/generative_scoring`, `/pooling`, `/classify`, `/score`, `/rerank`, `/pause`, `/resume`, `/abort_requests`, `/update_weights`, `/tokenize` and others as answering without it (source: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/usage/security.md\u003e)\n- With `--host` unset the launcher binds the empty address, which is every interface, on port 8000. `allowed_origins`, `allowed_methods` and `allowed_headers` default to `*`, with `allow_credentials` off (source: \u003chttps://github.com/vllm-project/vllm/blob/main/vllm/entrypoints/launchers/cli_args.py\u003e, \u003chttps://github.com/vllm-project/vllm/blob/main/vllm/entrypoints/launchers/launcher.py\u003e)\n- The GitHub advisories list returned 100 published advisories on one page, 81 of them published since 9 October 2025 (2 critical, 19 high, 56 medium, 4 low) and 50 since 11 July 2026. More may sit on later pages (source: \u003chttps://github.com/vllm-project/vllm/security/advisories\u003e)\n- GHSA-94f4-hr76-p5j6 (CVE-2026-48746, 9.1), published 2 June 2026, let a crafted Host header bypass the API key check, fixed in 0.22.0. GHSA-4r2x-xpjr-7cvv (CVE-2026-22778, 9.8), published 2 February 2026, was code execution through video decoding, fixed in 0.14.1\n- Eleven advisories were published on 9 October 2026, five of them for flaws whose fixed version is 0.24.0, tagged 28 June 2026 (source: \u003chttps://github.com/vllm-project/vllm/security/advisories\u003e)\n- Usage statistics are on by default and post hardware, platform, model architecture and engine settings with a random UUID to stats.vllm.ai, with a heartbeat every ten minutes (source: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/usage/usage_stats.md\u003e, \u003chttps://github.com/vllm-project/vllm/blob/main/vllm/usage/usage_lib.py\u003e)\n- The gRPC services have no authentication, authorisation or encryption and are off unless `--grpc-port` is set (source: \u003chttps://github.com/vllm-project/vllm/blob/main/docs/usage/security.md\u003e)\n- pytorch.org lists vLLM as a PyTorch Foundation project and says the University of California, Berkeley contributed it to the Linux Foundation in July 2024 (source: \u003chttps://pytorch.org/projects/vllm/\u003e)\n- vllm.ai says Python 3.10 or later is required, while `pyproject.toml` requires 3.11 to 3.14 and the quickstart says 3.11 to 3.14 (source: \u003chttps://vllm.ai\u003e, \u003chttps://github.com/vllm-project/vllm/blob/main/pyproject.toml\u003e)\n\n- #8 of 20 in Best local AI models and assistants: https://www.anchorterminal.com/best/local-ai/index.md\n- All 184 local ai comparisons: https://www.anchorterminal.com/compare/local-ai/index.md\n\n## Compare\n\n- [AnythingLLM vs vLLM](https://www.anchorterminal.com/compare/anythingllm-vs-vllm.md): D 53.3 vs C 57.7\n- [Docker Model Runner vs vLLM](https://www.anchorterminal.com/compare/docker-model-runner-vs-vllm.md): C 57.1 vs C 57.7\n- [Foundry Local vs vLLM](https://www.anchorterminal.com/compare/foundry-local-vs-vllm.md): C 60.5 vs C 57.7\n- [Core vs vLLM](https://www.anchorterminal.com/compare/ghost-core-vs-vllm.md): F 7.3 vs C 57.7\n- [GPT4All vs vLLM](https://www.anchorterminal.com/compare/gpt4all-vs-vllm.md): F 36.2 vs C 57.7\n- [Jan vs vLLM](https://www.anchorterminal.com/compare/jan-vs-vllm.md): D 51.3 vs C 57.7\n- [Khoj vs vLLM](https://www.anchorterminal.com/compare/khoj-vs-vllm.md): E 38.5 vs C 57.7\n- [KoboldCpp vs vLLM](https://www.anchorterminal.com/compare/koboldcpp-vs-vllm.md): C 60.5 vs C 57.7\n- [Lemonade vs vLLM](https://www.anchorterminal.com/compare/lemonade-vs-vllm.md): B 63.8 vs C 57.7\n- [llama.cpp vs vLLM](https://www.anchorterminal.com/compare/llama-cpp-vs-vllm.md): C 60.2 vs C 57.7\n- [LM Studio vs vLLM](https://www.anchorterminal.com/compare/lm-studio-vs-vllm.md): C 57.8 vs C 57.7\n- [LocalAI vs vLLM](https://www.anchorterminal.com/compare/localai-vs-vllm.md): B 68 vs C 57.7\n- [MLX LM vs vLLM](https://www.anchorterminal.com/compare/mlx-lm-vs-vllm.md): D 52.2 vs C 57.7\n- [Ollama vs vLLM](https://www.anchorterminal.com/compare/ollama-vs-vllm.md): C 56.3 vs C 57.7\n- [Open WebUI vs vLLM](https://www.anchorterminal.com/compare/open-webui-vs-vllm.md): D 51.8 vs C 57.7\n- [screenpipe vs vLLM](https://www.anchorterminal.com/compare/screenpipe-vs-vllm.md): C 60.8 vs C 57.7\n- [TextGen vs vLLM](https://www.anchorterminal.com/compare/text-generation-webui-vs-vllm.md): E 45.1 vs C 57.7\n- [Underdog vs vLLM](https://www.anchorterminal.com/compare/underdog-vs-vllm.md): F 29.5 vs C 57.7\n\n## Verify this listing\n\nFor the vendor. The badge or a plain link to this page verifies the listing, from a page on vllm.ai or one of its subdomains, or the README of github.com/vllm-project/vllm. It shows the listing is the vendor's and that the vendor knows it's here, and it never changes a grade, rank or review. The vendor sends the page's address to `POST https://www.anchorterminal.com/api/v1/verify` as `{\"slug\": \"vllm\", \"url\": \"…\"}`, or calls the `verify_listing` tool at https://www.anchorterminal.com/mcp. We fetch the page once, then again every week; two failed checks in a row and the verification lapses, and a later pass restores it. What we check: https://www.anchorterminal.com/builders/index.md#verify\n\nHTML badge:\n\n```html\n\u003ca href=\"https://www.anchorterminal.com/tools/vllm\"\u003e\u003cimg src=\"https://www.anchorterminal.com/badges/vllm.svg\" alt=\"vLLM on Anchor Terminal\" height=\"20\"\u003e\u003c/a\u003e\n```\n\nMarkdown badge, for a README:\n\n```markdown\n[![vLLM on Anchor Terminal](https://www.anchorterminal.com/badges/vllm.svg)](https://www.anchorterminal.com/tools/vllm)\n```\n\nPlain link:\n\n```html\n\u003ca href=\"https://www.anchorterminal.com/tools/vllm\"\u003evLLM on Anchor Terminal\u003c/a\u003e\n```\n\n## Share this listing\n\nFor the vendor. Sharing assets for social media, two PNGs of 1200 × 630 that say vLLM is listed on Anchor Terminal, with the vendor's logo and this page's address and no grade or score.\n\n- Dark: https://www.anchorterminal.com/assets/share/vllm-dark.png\n- Light: https://www.anchorterminal.com/assets/share/vllm-light.png\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-10",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Terminal",
        "url": "https://www.anchorterminal.com/tools/"
      },
      {
        "name": "Local AI",
        "url": "https://www.anchorterminal.com/categories/local-ai"
      },
      {
        "name": "vLLM",
        "url": ""
      }
    ],
    "description": "vLLM is an open-source inference and serving engine for open-weight language models. vllm serve runs an HTTP server with OpenAI-compatible, Anthropic Messages, embedding, reranking and transcription routes on the owner's own GPUs or CPUs.",
    "facts": [
      "rank #600 of 950",
      "None auth",
      "0 desk reviews"
    ],
    "h1": "vLLM",
    "image": "https://www.anchorterminal.com/assets/og/tools-vllm.png",
    "path": "/tools/vllm",
    "published": "2026-10-01",
    "section": "tools",
    "title": "vLLM review (2026): pricing, alternatives and grade C",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/tools/vllm"
  },
  "tokens": {
    "markdown": 7650,
    "slim": 1930
  },
  "version": 1
}
