{
  "data": {
    "a": {
      "slug": "koboldcpp",
      "name": "KoboldCpp",
      "vendor": "LostRuins (Concedo)",
      "vendorUrl": "https://koboldcpp.net",
      "kind": "http-api",
      "category": "local-ai",
      "summary": "Open-source program for running GGUF models on the owner's own computer, built on llama.cpp. One executable serves a web interface and KoboldAI, OpenAI, Ollama and Anthropic compatible APIs on port 5001.",
      "url": "https://www.anchorterminal.com/tools/koboldcpp",
      "markdownUrl": "https://www.anchorterminal.com/tools/koboldcpp.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/koboldcpp.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/koboldcpp.json",
      "repo": "https://github.com/LostRuins/koboldcpp",
      "license": "AGPL-3.0 for KoboldCpp and KoboldAI Lite. The bundled GGML, llama.cpp and stable-diffusion.cpp code stays under MIT",
      "transports": [
        "http"
      ],
      "packages": [
        {
          "registry": "oci",
          "name": "koboldai/koboldcpp"
        }
      ],
      "auth": "none",
      "authNotes": "No credential by default. `--password` (or the `KCPP_PASSWORD` environment variable) sets one shared key, sent as `Authorization: Bearer`, and the server reads it from the header only. The key guards text routes. Image generation, upscaling, `/sdapi/v1/interrogate` and `/tts_to_audio` skip the check, and the `--help` text says image endpoints are not secured. Admin routes need `--admin`, an `--admindir` and, when set, a separate `--adminpassword`. Keys have no scopes and change only with a restart. With no `--host` the server listens on all routable interfaces, and CORS reflects any Origin with credentials allowed (https://github.com/LostRuins/koboldcpp/blob/concedo/koboldcpp.py).",
      "pricing": "free",
      "pricingNotes": "Free under AGPL-3.0, with no account, key or card. Nothing is sold by the project. The owner pays for hardware and electricity, and the README links third-party GPU rental (RunPod, SimplePod) and Google Colab as other places to run it.",
      "priceSummary": "Free · OSS",
      "where": "local",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the README, the wiki, the API document or `koboldcpp.py` (checked 2026-10-08).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 11972,
        "npmWeekly": null,
        "pypiWeekly": null,
        "asOf": "2026-10-08"
      },
      "docsUrl": "https://github.com/LostRuins/koboldcpp/wiki",
      "llmsTxt": "https://koboldcpp.net/llms.txt",
      "openapi": "https://lite.koboldai.net/koboldcpp_api.json",
      "capabilities": [
        "inference.local",
        "inference.open-weights",
        "agent.mcp-client",
        "embed.text",
        "image.generate",
        "speech.stt",
        "speech.tts",
        "audio.music"
      ],
      "tags": [
        "open-source",
        "local",
        "self-hosted",
        "free",
        "no-card",
        "openai-compatible",
        "openapi",
        "llms-txt",
        "docker",
        "agpl",
        "no-telemetry"
      ],
      "lastRelease": "2026-09-27",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 60.5,
        "grade": "C",
        "agentReady": false,
        "rank": 510,
        "ranked": true,
        "rankOf": 950,
        "categoryRank": 5,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 63,
          "maintenance": 82,
          "payments": 60,
          "reliability": 68,
          "schema": 68,
          "security": 38,
          "transparency": 49
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-08"
        },
        "negative": 0,
        "verdict": "One file runs text, image, speech and music models behind a published OpenAPI 3.0.3 document, with eight releases in 90 days. The server listens on every interface with no password by default, and `--password` leaves the image routes open.",
        "bestFor": "An owner who wants text, image, speech and music models behind one executable with a writing and roleplay interface, and clients that speak the KoboldAI, OpenAI, Ollama or Anthropic formats.",
        "strengths": [
          "OpenAPI 3.0.3 document with 54 operations, served by the program at `/api?json=1` and published at lite.koboldai.net",
          "KoboldAI, OpenAI, Ollama, Anthropic, AUTOMATIC1111 and ComfyUI style routes from one server on port 5001",
          "Eight releases between 10 July and 27 September 2026, each with written notes, and replies on all 24 of the newest open issues",
          "AGPL-3.0, with no telemetry or update check found in `koboldcpp.py` and a wiki statement that inputs are sent nowhere",
          "Admin functions are off by default and take their own `--adminpassword`"
        ],
        "weaknesses": [
          "With no `--host` the server accepts connections on all routable interfaces, and no password is set by default",
          "`--password` covers text routes only. The `--help` text says image endpoints are not secured",
          "CORS reflects any Origin with credentials allowed and permits private-network requests",
          "No `SECURITY.md` or security.txt, and the one advisory (GHSA-qhvp-gj7g-rw26) still lists no patched version",
          "No continuous test run on pushes. Build workflows are started by hand and the only automatic test covers `AutoGuess.json`"
        ],
        "agentNotes": [
          "Start with `--host 127.0.0.1` and `--password`. The default listens on every interface with no key",
          "Send the password as `Authorization: Bearer \u003cpassword\u003e`. It is not read from the query string",
          "Treat 503 as both busy and rate limited. The server never sends 429 or `Retry-After`, and the wait in seconds is in `detail.msg`",
          "Pass `max_length` or `max_tokens`. The default is 2,048 tokens unless `--defaultgenamt` changes it",
          "Send a `genkey` with each generation so `/api/extra/generate/check` and `/api/extra/abort` act on your request and not another caller's"
        ],
        "metrics": {
          "kind": "local",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "C",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 60.5
          }
        ],
        "editorialScores": {
          "ergonomics": 63,
          "maintenance": 82,
          "payments": 60,
          "reliability": 68,
          "schema": 68,
          "security": 38,
          "transparency": 70
        },
        "provenanceScore": 27
      },
      "connect": {
        "install": "curl -fLo koboldcpp-linux-x64 https://github.com/LostRuins/koboldcpp/releases/latest/download/koboldcpp-linux-x64 \u0026\u0026 chmod +x koboldcpp-linux-x64 \u0026\u0026 ./koboldcpp-linux-x64\n./koboldcpp-linux-x64 --model /path/to/model.gguf   # listens on port 5001",
        "http": "curl --request POST \\\n    --url http://localhost:5001/api/v1/generate \\\n    --header \"Content-Type: application/json\" \\\n    --data '{\"prompt\": \"Niko the kobold stalked carefully down the alley,\", \"max_context_length\": 2048, \"max_length\": 100}'"
      },
      "letme": {
        "capability": "https://letme.dev/inference.local",
        "tool": "https://letme.dev/koboldcpp"
      },
      "area": "models",
      "provenance": {
        "legalEntity": "",
        "domain": "koboldcpp.net",
        "domainRegistered": "2026-03-18",
        "endpointOnVendorDomain": null,
        "terms": "",
        "privacy": "",
        "statusPage": "",
        "changelog": "https://github.com/LostRuins/koboldcpp/releases",
        "securityTxt": "none",
        "checked": "2026-10-08",
        "notes": [
          "No legal entity is named in the repository, the wiki or koboldcpp.net. The maintainer publishes as LostRuins on GitHub and Concedo on Discord, and the Windows binary's version resource gives KoboldAI as the company name.",
          "The project publishes no terms of service and no privacy policy, so both fields are empty. The licence is AGPL-3.0 and the only privacy statement is an FAQ entry in the wiki.",
          "The README calls koboldcpp.net the official community website. RDAP gives its registration date as 2026-03-18 and Cloudflare, Inc. as registrar. The README warns that koboldcpp.com is a fake site.",
          "koboldcpp.net/.well-known/security.txt and koboldai.org/.well-known/security.txt return 404. koboldcpp.net/llms.txt returns a short index that links llms-small.txt and llms-full.txt.",
          "There's no shared hosted endpoint. The server runs on the owner's machine. The online API reference is on lite.koboldai.net."
        ],
        "score": 27
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/koboldcpp.json",
      "live": {
        "slug": "koboldcpp",
        "versions": [
          {
            "registry": "github",
            "name": "LostRuins/koboldcpp",
            "version": "v1.122.1",
            "released": "2026-09-26",
            "seenAt": "2026-10-09T17:00:45.262567963Z"
          }
        ],
        "githubStars": 11978,
        "securityTxt": {
          "url": "https://koboldcpp.net/.well-known/security.txt",
          "state": "none",
          "checkedAt": "2026-10-09T15:39:49.61001824Z"
        },
        "llmsTxt": {
          "url": "https://koboldcpp.net/llms.txt",
          "ok": true,
          "status": 200,
          "checkedAt": "2026-10-09T14:02:12.647597076Z"
        },
        "updatedAt": "2026-10-09T17:00:45.262567963Z"
      }
    },
    "answer": "KoboldCpp scores 60.5 (C) on agent readiness against vLLM's 57.7 (C), and leads in 1 of 7 scored categories. vLLM leads on security \u0026 auth, maintenance \u0026 community and transparency \u0026 trust.",
    "b": {
      "slug": "vllm",
      "name": "vLLM",
      "vendor": "vLLM project (PyTorch Foundation)",
      "vendorUrl": "https://vllm.ai",
      "kind": "http-api",
      "category": "local-ai",
      "summary": "vLLM is an open-source inference and serving engine for open-weight language models. `vllm serve` runs an HTTP server with OpenAI-compatible, Anthropic Messages, embedding, reranking and transcription routes on the owner's own GPUs or CPUs.",
      "url": "https://www.anchorterminal.com/tools/vllm",
      "markdownUrl": "https://www.anchorterminal.com/tools/vllm.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/vllm.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/vllm.json",
      "repo": "https://github.com/vllm-project/vllm",
      "license": "Apache-2.0",
      "transports": [
        "http"
      ],
      "packages": [
        {
          "registry": "pypi",
          "name": "vllm"
        },
        {
          "registry": "oci",
          "name": "vllm/vllm-openai"
        }
      ],
      "auth": "none",
      "authNotes": "No credential by default. `--api-key` (one or several keys) or `VLLM_API_KEY` turns on a Bearer check for paths under `/v1`, `/v2`, `/inference` and `/cohere` only, so `/invocations`, `/pooling`, `/classify`, `/score`, `/rerank` and control routes such as `/pause` stay open. Keys have no scopes and change with a restart. The key is read from the `Authorization` header, never the query string. gRPC has no authentication (https://github.com/vllm-project/vllm/blob/main/docs/usage/security.md).",
      "pricing": "free",
      "pricingNotes": "Free under Apache-2.0, with no account, key or card. Nothing is sold by the project. You pay for your own hardware and electricity.",
      "priceSummary": "Free · OSS",
      "where": "local",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the docs or the source (checked 2026-10-09).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 93444,
        "npmWeekly": null,
        "pypiWeekly": null,
        "asOf": "2026-10-09"
      },
      "docsUrl": "https://docs.vllm.ai/en/stable/",
      "capabilities": [
        "inference.local",
        "inference.open-weights",
        "embed.text",
        "rerank",
        "speech.stt",
        "inference.decision",
        "agent.mcp-client"
      ],
      "tags": [
        "open-source",
        "local",
        "self-hosted",
        "free",
        "no-card",
        "openai-compatible",
        "docker",
        "pre-1.0",
        "telemetry-default-on"
      ],
      "lastRelease": "2026-10-02",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 57.7,
        "grade": "C",
        "agentReady": false,
        "rank": 600,
        "ranked": true,
        "rankOf": 950,
        "categoryRank": 8,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 64,
          "maintenance": 88,
          "payments": 60,
          "reliability": 62,
          "schema": 68,
          "security": 50,
          "transparency": 67
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-09"
        },
        "negative": -6,
        "negativeNotes": [
          "2026-06-02. GHSA-94f4-hr76-p5j6 (CVE-2026-48746, 9.1), a crafted Host header bypassed the API key check on the OpenAI routes, fixed in 0.22.0. With GHSA-4r2x-xpjr-7cvv (CVE-2026-22778, 9.8) of 2 February 2026, code execution through video decoding fixed in 0.14.1, these are the two critical advisories of the last 12 months. Both were fixed and published with CVEs, so they decay, -3. https://github.com/vllm-project/vllm/security/advisories/GHSA-94f4-hr76-p5j6; https://github.com/vllm-project/vllm/security/advisories/GHSA-4r2x-xpjr-7cvv",
          "2026-10-06. GHSA-h3rc-6mm3-gc2m (8.1), a request field could select the processor code a server started with `--trust-remote-code` imports, fixed in 0.31.0, one of 50 advisories published since 11 July 2026 (10 high, 36 medium, 4 low), most of them requests that crash or exhaust the engine. All name a fixed version, and eleven were published on 9 October 2026 months after their fixes, -3. https://github.com/vllm-project/vllm/security/advisories/GHSA-h3rc-6mm3-gc2m; https://github.com/vllm-project/vllm/security/advisories"
        ],
        "verdict": "Apache-2.0 software with a release about every two weeks, each with notes that list breaking changes and security fixes. The optional API key covers only some path prefixes, so `/invocations` and control routes such as `/pause` answer without it, and at least 81 security advisories were published in the 12 months to 9 October 2026.",
        "bestFor": "An owner with a GPU server who wants many concurrent requests against one open-weight model behind OpenAI or Anthropic compatible routes.",
        "strengths": [
          "OpenAI chat, completions, responses and embeddings, Anthropic `/v1/messages`, Cohere embed and rerank, transcription and `/v1/systemone` from one server",
          "Apache-2.0, with a written three-stage deprecation policy and release notes that carry a breaking changes section",
          "Eight stable releases between 12 July and 2 October 2026, and v0.31.0 lists 717 commits from 307 contributors",
          "A 650-line security guide names every route the API key does and does not protect, and the limits of multi-tenant use",
          "Usage statistics are documented field by field, with `VLLM_NO_USAGE_STATS`, `DO_NOT_TRACK` or a file as opt-outs"
        ],
        "weaknesses": [
          "`--api-key` guards only the `/v1`, `/v2`, `/inference` and `/cohere` prefixes. `/invocations`, `/pooling`, `/classify`, `/score`, `/rerank`, `/pause` and `/update_weights` answer without it",
          "No key by default, the server binds every interface when `--host` is unset, and CORS allows any origin",
          "At least 81 GitHub security advisories in 12 months, two critical, most of them remote crashes or resource exhaustion",
          "Pre-1.0 (0.31.0), with breaking changes in each fortnightly release and compatibility kept for a limited number of minor versions",
          "Usage statistics are sent to stats.vllm.ai by default, and no privacy policy or retention period for them was found"
        ],
        "agentNotes": [
          "Put a reverse proxy that allowlists routes in front of the server. `--api-key` leaves `/invocations` and the control routes open",
          "Pass `--host 127.0.0.1` for single-machine use. With no `--host` the server listens on every interface",
          "Set `VLLM_NO_USAGE_STATS=1` or `DO_NOT_TRACK=1` before starting if nothing should be sent to stats.vllm.ai",
          "Start with `--enable-auto-tool-choice` and the `--tool-call-parser` for the model before sending tools. Tool calling is off without them",
          "Send `max_tokens` on every request, and read the breaking changes section of the release notes before upgrading a minor version"
        ],
        "metrics": {
          "kind": "local",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "C",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 57.7
          }
        ],
        "editorialScores": {
          "ergonomics": 64,
          "maintenance": 88,
          "payments": 60,
          "reliability": 62,
          "schema": 68,
          "security": 50,
          "transparency": 80
        },
        "provenanceScore": 53
      },
      "connect": {
        "install": "uv pip install vllm --torch-backend=auto\nvllm serve Qwen/Qwen2.5-1.5B-Instruct   # listens on port 8000",
        "http": "curl http://localhost:8000/v1/chat/completions \\\n    -H \"Content-Type: application/json\" \\\n    -d '{\n        \"model\": \"Qwen/Qwen2.5-1.5B-Instruct\",\n        \"messages\": [\n            {\"role\": \"system\", \"content\": \"You are a helpful assistant.\"},\n            {\"role\": \"user\", \"content\": \"Who won the world series in 2020?\"}\n        ]\n    }'",
        "claudeCode": "ANTHROPIC_BASE_URL=http://localhost:8000 \\\nANTHROPIC_API_KEY=dummy \\\nANTHROPIC_AUTH_TOKEN=dummy \\\nANTHROPIC_DEFAULT_OPUS_MODEL=my-model \\\nANTHROPIC_DEFAULT_SONNET_MODEL=my-model \\\nANTHROPIC_DEFAULT_HAIKU_MODEL=my-model \\\nclaude"
      },
      "letme": {
        "capability": "https://letme.dev/inference.local",
        "tool": "https://letme.dev/vllm"
      },
      "area": "models",
      "provenance": {
        "legalEntity": "The Linux Foundation (vLLM is a PyTorch Foundation project)",
        "domain": "vllm.ai",
        "domainRegistered": "",
        "endpointOnVendorDomain": null,
        "terms": "",
        "privacy": "",
        "statusPage": "",
        "changelog": "https://github.com/vllm-project/vllm/releases",
        "securityTxt": "none",
        "checked": "2026-10-09",
        "notes": [
          "vllm.ai links no terms and no privacy policy, and its footer reads © 2026 vLLM. The project publishes none for the software or for stats.vllm.ai, so the Apache-2.0 licence stands in for terms.",
          "pytorch.org/projects/vllm/ lists vLLM among PyTorch Foundation projects and says UC Berkeley contributed it to the Linux Foundation in July 2024. The Linux Foundation's policies are linked from that page and are not specific to vLLM.",
          "vllm.ai/.well-known/security.txt and docs.vllm.ai/.well-known/security.txt return 404. SECURITY.md asks for private reports through GitHub.",
          "There's no shared hosted endpoint. The server runs on the owner's hardware. The software posts usage statistics to stats.vllm.ai unless turned off."
        ],
        "score": 53
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/vllm.json",
      "live": {
        "slug": "vllm",
        "versions": [
          {
            "registry": "github",
            "name": "vllm-project/vllm",
            "version": "v0.31.0",
            "released": "2026-10-05",
            "seenAt": "2026-10-09T17:27:30.444059802Z"
          },
          {
            "registry": "pypi",
            "name": "vllm",
            "version": "0.31.0",
            "released": "2026-10-05",
            "seenAt": "2026-10-09T17:27:30.259454009Z"
          }
        ],
        "githubStars": 93457,
        "pypiWeekly": 444466,
        "updatedAt": "2026-10-09T17:27:30.444059802Z"
      }
    },
    "facts": [
      {
        "a": "HTTP API",
        "b": "HTTP API",
        "name": "Kind"
      },
      {
        "a": "LostRuins (Concedo)",
        "b": "vLLM project (PyTorch Foundation)",
        "name": "Vendor"
      },
      {
        "a": "no (local only)",
        "b": "no (local only)",
        "name": "Hosted endpoint"
      },
      {
        "a": "HTTP",
        "b": "HTTP",
        "name": "Transports"
      },
      {
        "a": "None",
        "b": "None",
        "name": "Auth"
      },
      {
        "a": "Free",
        "b": "Free",
        "name": "Pricing"
      },
      {
        "a": "no",
        "b": "no",
        "name": "x402"
      },
      {
        "a": "AGPL-3.0 for KoboldCpp and KoboldAI Lite. The bundled GGML, llama.cpp and stable-diffusion.cpp code stays under MIT",
        "b": "Apache-2.0",
        "name": "Licence"
      },
      {
        "a": "no",
        "b": "no",
        "name": "Read-only variant documented"
      },
      {
        "a": "yes",
        "b": "no",
        "name": "llms.txt"
      },
      {
        "a": "2026-09-27",
        "b": "2026-10-02",
        "name": "Last release"
      },
      {
        "a": "no document linked",
        "b": "no document linked",
        "name": "Terms last updated"
      },
      {
        "a": "no document linked",
        "b": "no document linked",
        "name": "Privacy policy last updated"
      },
      {
        "a": "",
        "b": "",
        "name": "Customer content may train models"
      },
      {
        "a": "",
        "b": "",
        "name": "Terms restrict automated access"
      },
      {
        "a": "",
        "b": "",
        "name": "Terms restrict benchmarking"
      },
      {
        "a": "",
        "b": "",
        "name": "Terms or service can change without notice"
      },
      {
        "a": "",
        "b": "",
        "name": "Arbitration or class-action waiver"
      },
      {
        "a": "12k stars",
        "b": "93k stars",
        "name": "Popularity"
      }
    ],
    "faq": [
      {
        "answer": "KoboldCpp scores 60.5 (C) on agent readiness against vLLM's 57.7 (C), and leads in 1 of 7 scored categories. vLLM leads on security \u0026 auth, maintenance \u0026 community and transparency \u0026 trust.",
        "question": "Which is better for AI agents, KoboldCpp or vLLM?"
      },
      {
        "answer": "Neither needs a key.",
        "question": "Do KoboldCpp and vLLM need an API key?"
      },
      {
        "answer": "No hosted endpoint is listed for KoboldCpp. No hosted endpoint is listed for vLLM.",
        "question": "Can an agent call KoboldCpp and vLLM without installing anything?"
      },
      {
        "answer": "Yes. KoboldCpp is open source (AGPL-3.0 for KoboldCpp and KoboldAI Lite. The bundled GGML, llama.cpp and stable-diffusion.cpp code stays under MIT). vLLM is open source (Apache-2.0).",
        "question": "Are KoboldCpp and vLLM open source?"
      }
    ],
    "goodFor": [
      {
        "aheadOn": [
          "Reliability, 68 against 62"
        ],
        "also": [
          "No incidents deducted, where vLLM loses 6 points for them"
        ],
        "goodFor": "An owner who wants text, image, speech and music models behind one executable with a writing and roleplay interface, and clients that speak the KoboldAI, OpenAI, Ollama or Anthropic formats.",
        "slug": "koboldcpp",
        "watchFor": "With no `--host` the server accepts connections on all routable interfaces, and no password is set by default"
      },
      {
        "aheadOn": [
          "Security \u0026 auth, 50 against 38",
          "Maintenance \u0026 community, 88 against 82",
          "Transparency \u0026 trust, 67 against 49"
        ],
        "also": null,
        "goodFor": "An owner with a GPU server who wants many concurrent requests against one open-weight model behind OpenAI or Anthropic compatible routes.",
        "slug": "vllm",
        "watchFor": "`--api-key` guards only the `/v1`, `/v2`, `/inference` and `/cohere` prefixes. `/invocations`, `/pooling`, `/classify`, `/score`, `/rerank`, `/pause` and `/update_weights` answer without it"
      }
    ],
    "job": {
      "capability": "inference.local",
      "name": "Local inference"
    },
    "others": [
      {
        "json": "https://www.anchorterminal.com/compare/anythingllm-vs-koboldcpp.json",
        "title": "AnythingLLM vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/anythingllm-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/anythingllm-vs-vllm.json",
        "title": "AnythingLLM vs vLLM",
        "url": "https://www.anchorterminal.com/compare/anythingllm-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/docker-model-runner-vs-koboldcpp.json",
        "title": "Docker Model Runner vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/docker-model-runner-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/docker-model-runner-vs-vllm.json",
        "title": "Docker Model Runner vs vLLM",
        "url": "https://www.anchorterminal.com/compare/docker-model-runner-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-koboldcpp.json",
        "title": "Foundry Local vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-vllm.json",
        "title": "Foundry Local vs vLLM",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/ghost-core-vs-koboldcpp.json",
        "title": "Core vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/ghost-core-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/ghost-core-vs-vllm.json",
        "title": "Core vs vLLM",
        "url": "https://www.anchorterminal.com/compare/ghost-core-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/gpt4all-vs-koboldcpp.json",
        "title": "GPT4All vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/gpt4all-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/gpt4all-vs-vllm.json",
        "title": "GPT4All vs vLLM",
        "url": "https://www.anchorterminal.com/compare/gpt4all-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/jan-vs-koboldcpp.json",
        "title": "Jan vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/jan-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/jan-vs-vllm.json",
        "title": "Jan vs vLLM",
        "url": "https://www.anchorterminal.com/compare/jan-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/khoj-vs-koboldcpp.json",
        "title": "Khoj vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/khoj-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/khoj-vs-vllm.json",
        "title": "Khoj vs vLLM",
        "url": "https://www.anchorterminal.com/compare/khoj-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-lemonade.json",
        "title": "KoboldCpp vs Lemonade",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-lemonade"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp.json",
        "title": "KoboldCpp vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-lm-studio.json",
        "title": "KoboldCpp vs LM Studio",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-lm-studio"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-localai.json",
        "title": "KoboldCpp vs LocalAI",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-localai"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-mlx-lm.json",
        "title": "KoboldCpp vs MLX LM",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-mlx-lm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-ollama.json",
        "title": "KoboldCpp vs Ollama",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-ollama"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-open-webui.json",
        "title": "KoboldCpp vs Open WebUI",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-open-webui"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-screenpipe.json",
        "title": "KoboldCpp vs screenpipe",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-screenpipe"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-text-generation-webui.json",
        "title": "KoboldCpp vs TextGen",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-text-generation-webui"
      },
      {
        "json": "https://www.anchorterminal.com/compare/lemonade-vs-vllm.json",
        "title": "Lemonade vs vLLM",
        "url": "https://www.anchorterminal.com/compare/lemonade-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-vllm.json",
        "title": "llama.cpp vs vLLM",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/lm-studio-vs-vllm.json",
        "title": "LM Studio vs vLLM",
        "url": "https://www.anchorterminal.com/compare/lm-studio-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/localai-vs-vllm.json",
        "title": "LocalAI vs vLLM",
        "url": "https://www.anchorterminal.com/compare/localai-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/mlx-lm-vs-vllm.json",
        "title": "MLX LM vs vLLM",
        "url": "https://www.anchorterminal.com/compare/mlx-lm-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/ollama-vs-vllm.json",
        "title": "Ollama vs vLLM",
        "url": "https://www.anchorterminal.com/compare/ollama-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/open-webui-vs-vllm.json",
        "title": "Open WebUI vs vLLM",
        "url": "https://www.anchorterminal.com/compare/open-webui-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/screenpipe-vs-vllm.json",
        "title": "screenpipe vs vLLM",
        "url": "https://www.anchorterminal.com/compare/screenpipe-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/text-generation-webui-vs-vllm.json",
        "title": "TextGen vs vLLM",
        "url": "https://www.anchorterminal.com/compare/text-generation-webui-vs-vllm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-underdog.json",
        "title": "KoboldCpp vs Underdog",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-underdog"
      },
      {
        "json": "https://www.anchorterminal.com/compare/underdog-vs-vllm.json",
        "title": "Underdog vs vLLM",
        "url": "https://www.anchorterminal.com/compare/underdog-vs-vllm"
      }
    ],
    "scores": [
      {
        "by": 6,
        "edge": "koboldcpp",
        "key": "reliability",
        "koboldcpp": 68,
        "name": "Reliability",
        "vllm": 62,
        "weight": 16
      },
      {
        "key": "performance",
        "name": "Performance",
        "pending": true,
        "weight": 10
      },
      {
        "by": 0,
        "edge": "",
        "key": "schema",
        "koboldcpp": 68,
        "name": "Schema \u0026 documentation",
        "vllm": 68,
        "weight": 13
      },
      {
        "by": 1,
        "edge": "vllm",
        "key": "ergonomics",
        "koboldcpp": 63,
        "name": "Agent ergonomics",
        "vllm": 64,
        "weight": 13
      },
      {
        "by": 12,
        "edge": "vllm",
        "key": "security",
        "koboldcpp": 38,
        "name": "Security \u0026 auth",
        "vllm": 50,
        "weight": 14
      },
      {
        "by": 0,
        "edge": "",
        "key": "payments",
        "koboldcpp": 60,
        "name": "Payments \u0026 pricing",
        "vllm": 60,
        "weight": 10
      },
      {
        "key": "tasks",
        "name": "Task success",
        "pending": true,
        "weight": 10
      },
      {
        "by": 6,
        "edge": "vllm",
        "key": "maintenance",
        "koboldcpp": 82,
        "name": "Maintenance \u0026 community",
        "vllm": 88,
        "weight": 7
      },
      {
        "by": 18,
        "edge": "vllm",
        "key": "transparency",
        "koboldcpp": 49,
        "name": "Transparency \u0026 trust",
        "vllm": 67,
        "weight": 7
      }
    ],
    "summary": "KoboldCpp scores 60.5 (C) on agent readiness against vLLM's 57.7 (C), and leads in 1 of 7 scored categories. vLLM leads on security \u0026 auth, maintenance \u0026 community and transparency \u0026 trust. Both do local inference.",
    "verdicts": {
      "koboldcpp": "One file runs text, image, speech and music models behind a published OpenAPI 3.0.3 document, with eight releases in 90 days. The server listens on every interface with no password by default, and `--password` leaves the image routes open.",
      "vllm": "Apache-2.0 software with a release about every two weeks, each with notes that list breaking changes and security fixes. The optional API key covers only some path prefixes, so `/invocations` and control routes such as `/pause` answer without it, and at least 81 security advisories were published in the 12 months to 9 October 2026."
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/compare/koboldcpp-vs-vllm",
    "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-vllm.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/compare/koboldcpp-vs-vllm.md",
    "slim": "https://www.anchorterminal.com/compare/koboldcpp-vs-vllm.min.md"
  },
  "markdown": "KoboldCpp scores 60.5 (C) on agent readiness against vLLM's 57.7 (C), and leads in 1 of 7 scored categories. vLLM leads on security \u0026 auth, maintenance \u0026 community and transparency \u0026 trust. Both do local inference.\n\n- KoboldCpp: grade C, 60.5/100, rank #510 of 950. Markdown https://www.anchorterminal.com/tools/koboldcpp.md · JSON https://www.anchorterminal.com/api/v1/tools/koboldcpp.json\n- vLLM: grade C, 57.7/100, rank #600 of 950. Markdown https://www.anchorterminal.com/tools/vllm.md · JSON https://www.anchorterminal.com/api/v1/tools/vllm.json\n- Best local AI models and assistants: https://www.anchorterminal.com/best/local-ai/index.md\n- All 184 local ai comparisons: https://www.anchorterminal.com/compare/local-ai/index.md\n\n## Which one, for what\n\n### KoboldCpp (C)\n\nGood for: An owner who wants text, image, speech and music models behind one executable with a writing and roleplay interface, and clients that speak the KoboldAI, OpenAI, Ollama or Anthropic formats.\n\nAhead on:\n- Reliability, 68 against 62\n\nAlso in its favour:\n- No incidents deducted, where vLLM loses 6 points for them\n\nWatch for: With no `--host` the server accepts connections on all routable interfaces, and no password is set by default\n\n### vLLM (C)\n\nGood for: An owner with a GPU server who wants many concurrent requests against one open-weight model behind OpenAI or Anthropic compatible routes.\n\nAhead on:\n- Security \u0026 auth, 50 against 38\n- Maintenance \u0026 community, 88 against 82\n- Transparency \u0026 trust, 67 against 49\n\nWatch for: `--api-key` guards only the `/v1`, `/v2`, `/inference` and `/cohere` prefixes. `/invocations`, `/pooling`, `/classify`, `/score`, `/rerank`, `/pause` and `/update_weights` answer without it\n\n\n## Score by category\n\n| Category | Weight | KoboldCpp | vLLM | Edge |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% (20 this run) | 68 | 62 | KoboldCpp +6 |\n| Performance | 10%, pending | pending | pending | not scored in this run |\n| Schema \u0026 documentation | 13% (16.2 this run) | 68 | 68 | even |\n| Agent ergonomics | 13% (16.2 this run) | 63 | 64 | vLLM +1 |\n| Security \u0026 auth | 14% (17.5 this run) | 38 | 50 | vLLM +12 |\n| Payments \u0026 pricing | 10% (12.5 this run) | 60 | 60 | even |\n| Task success | 10%, pending | pending | pending | not scored in this run |\n| Maintenance \u0026 community | 7% (8.8 this run) | 82 | 88 | vLLM +6 |\n| Transparency \u0026 trust | 7% (8.8 this run) | 49 | 67 | vLLM +18 |\n| Negative events | ≤15 | 0 | -6 | |\n| **Total** | | **60.5 · C** | **57.7 · C** | |\n\n## Facts side by side\n\n| Fact | KoboldCpp | vLLM |\n| --- | --- | --- |\n| Kind | HTTP API | HTTP API |\n| Vendor | LostRuins (Concedo) | vLLM project (PyTorch Foundation) |\n| Hosted endpoint | no (local only) | no (local only) |\n| Transports | HTTP | HTTP |\n| Auth | None | None |\n| Pricing | Free | Free |\n| x402 | no | no |\n| Licence | AGPL-3.0 for KoboldCpp and KoboldAI Lite. The bundled GGML, llama.cpp and stable-diffusion.cpp code stays under MIT | Apache-2.0 |\n| Read-only variant documented | no | no |\n| llms.txt | yes | no |\n| Last release | 2026-09-27 | 2026-10-02 |\n| Terms last updated | no document linked | no document linked |\n| Privacy policy last updated | no document linked | no document linked |\n| Customer content may train models |  |  |\n| Terms restrict automated access |  |  |\n| Terms restrict benchmarking |  |  |\n| Terms or service can change without notice |  |  |\n| Arbitration or class-action waiver |  |  |\n| Popularity | 12k stars | 93k stars |\n\n## Verdicts\n\n**KoboldCpp.** One file runs text, image, speech and music models behind a published OpenAPI 3.0.3 document, with eight releases in 90 days. The server listens on every interface with no password by default, and `--password` leaves the image routes open.\n\n**vLLM.** Apache-2.0 software with a release about every two weeks, each with notes that list breaking changes and security fixes. The optional API key covers only some path prefixes, so `/invocations` and control routes such as `/pause` answer without it, and at least 81 security advisories were published in the 12 months to 9 October 2026.\n\n## Before you call either\n\n### KoboldCpp\n\n1. Start with `--host 127.0.0.1` and `--password`. The default listens on every interface with no key\n2. Send the password as `Authorization: Bearer \u003cpassword\u003e`. It is not read from the query string\n3. Treat 503 as both busy and rate limited. The server never sends 429 or `Retry-After`, and the wait in seconds is in `detail.msg`\n4. Pass `max_length` or `max_tokens`. The default is 2,048 tokens unless `--defaultgenamt` changes it\n5. Send a `genkey` with each generation so `/api/extra/generate/check` and `/api/extra/abort` act on your request and not another caller's\n\n### vLLM\n\n1. Put a reverse proxy that allowlists routes in front of the server. `--api-key` leaves `/invocations` and the control routes open\n2. Pass `--host 127.0.0.1` for single-machine use. With no `--host` the server listens on every interface\n3. Set `VLLM_NO_USAGE_STATS=1` or `DO_NOT_TRACK=1` before starting if nothing should be sent to stats.vllm.ai\n4. Start with `--enable-auto-tool-choice` and the `--tool-call-parser` for the model before sending tools. Tool calling is off without them\n5. Send `max_tokens` on every request, and read the breaking changes section of the release notes before upgrading a minor version\n\n## Questions\n\n### Which is better for AI agents, KoboldCpp or vLLM?\n\nKoboldCpp scores 60.5 (C) on agent readiness against vLLM's 57.7 (C), and leads in 1 of 7 scored categories. vLLM leads on security \u0026 auth, maintenance \u0026 community and transparency \u0026 trust.\n\n### Do KoboldCpp and vLLM need an API key?\n\nNeither needs a key.\n\n### Can an agent call KoboldCpp and vLLM without installing anything?\n\nNo hosted endpoint is listed for KoboldCpp. No hosted endpoint is listed for vLLM.\n\n### Are KoboldCpp and vLLM open source?\n\nYes. KoboldCpp is open source (AGPL-3.0 for KoboldCpp and KoboldAI Lite. The bundled GGML, llama.cpp and stable-diffusion.cpp code stays under MIT). vLLM is open source (Apache-2.0).\n\n\n## For agents\n\n- This comparison as JSON: https://www.anchorterminal.com/compare/koboldcpp-vs-vllm.json, and with the fewest tokens: https://www.anchorterminal.com/compare/koboldcpp-vs-vllm.min.md\n- Over MCP at https://www.anchorterminal.com/mcp (no key): `compare_tools {\"a\": \"koboldcpp\", \"b\": \"vllm\"}`. From a terminal: `anchor compare koboldcpp vllm`\n- Each listing in full: https://www.anchorterminal.com/api/v1/tools/koboldcpp.json and https://www.anchorterminal.com/api/v1/tools/vllm.json\n\n## Other comparisons with KoboldCpp or vLLM\n\n- [AnythingLLM vs KoboldCpp](https://www.anchorterminal.com/compare/anythingllm-vs-koboldcpp.md)\n- [AnythingLLM vs vLLM](https://www.anchorterminal.com/compare/anythingllm-vs-vllm.md)\n- [Docker Model Runner vs KoboldCpp](https://www.anchorterminal.com/compare/docker-model-runner-vs-koboldcpp.md)\n- [Docker Model Runner vs vLLM](https://www.anchorterminal.com/compare/docker-model-runner-vs-vllm.md)\n- [Foundry Local vs KoboldCpp](https://www.anchorterminal.com/compare/foundry-local-vs-koboldcpp.md)\n- [Foundry Local vs vLLM](https://www.anchorterminal.com/compare/foundry-local-vs-vllm.md)\n- [Core vs KoboldCpp](https://www.anchorterminal.com/compare/ghost-core-vs-koboldcpp.md)\n- [Core vs vLLM](https://www.anchorterminal.com/compare/ghost-core-vs-vllm.md)\n- [GPT4All vs KoboldCpp](https://www.anchorterminal.com/compare/gpt4all-vs-koboldcpp.md)\n- [GPT4All vs vLLM](https://www.anchorterminal.com/compare/gpt4all-vs-vllm.md)\n- [Jan vs KoboldCpp](https://www.anchorterminal.com/compare/jan-vs-koboldcpp.md)\n- [Jan vs vLLM](https://www.anchorterminal.com/compare/jan-vs-vllm.md)\n- [Khoj vs KoboldCpp](https://www.anchorterminal.com/compare/khoj-vs-koboldcpp.md)\n- [Khoj vs vLLM](https://www.anchorterminal.com/compare/khoj-vs-vllm.md)\n- [KoboldCpp vs Lemonade](https://www.anchorterminal.com/compare/koboldcpp-vs-lemonade.md)\n- [KoboldCpp vs llama.cpp](https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp.md)\n- [KoboldCpp vs LM Studio](https://www.anchorterminal.com/compare/koboldcpp-vs-lm-studio.md)\n- [KoboldCpp vs LocalAI](https://www.anchorterminal.com/compare/koboldcpp-vs-localai.md)\n- [KoboldCpp vs MLX LM](https://www.anchorterminal.com/compare/koboldcpp-vs-mlx-lm.md)\n- [KoboldCpp vs Ollama](https://www.anchorterminal.com/compare/koboldcpp-vs-ollama.md)\n- [KoboldCpp vs Open WebUI](https://www.anchorterminal.com/compare/koboldcpp-vs-open-webui.md)\n- [KoboldCpp vs screenpipe](https://www.anchorterminal.com/compare/koboldcpp-vs-screenpipe.md)\n- [KoboldCpp vs TextGen](https://www.anchorterminal.com/compare/koboldcpp-vs-text-generation-webui.md)\n- [Lemonade vs vLLM](https://www.anchorterminal.com/compare/lemonade-vs-vllm.md)\n- [llama.cpp vs vLLM](https://www.anchorterminal.com/compare/llama-cpp-vs-vllm.md)\n- [LM Studio vs vLLM](https://www.anchorterminal.com/compare/lm-studio-vs-vllm.md)\n- [LocalAI vs vLLM](https://www.anchorterminal.com/compare/localai-vs-vllm.md)\n- [MLX LM vs vLLM](https://www.anchorterminal.com/compare/mlx-lm-vs-vllm.md)\n- [Ollama vs vLLM](https://www.anchorterminal.com/compare/ollama-vs-vllm.md)\n- [Open WebUI vs vLLM](https://www.anchorterminal.com/compare/open-webui-vs-vllm.md)\n- [screenpipe vs vLLM](https://www.anchorterminal.com/compare/screenpipe-vs-vllm.md)\n- [TextGen vs vLLM](https://www.anchorterminal.com/compare/text-generation-webui-vs-vllm.md)\n- [KoboldCpp vs Underdog](https://www.anchorterminal.com/compare/koboldcpp-vs-underdog.md)\n- [Underdog vs vLLM](https://www.anchorterminal.com/compare/underdog-vs-vllm.md)\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-10",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Compare",
        "url": "https://www.anchorterminal.com/compare/"
      },
      {
        "name": "KoboldCpp vs vLLM",
        "url": ""
      }
    ],
    "description": "KoboldCpp scores 60.5 (C) to vLLM's 57.7 (C) for local inference. Prices, MCP, x402, uptime and agent notes side by side.",
    "facts": [
      "KoboldCpp C 60.5",
      "vLLM C 57.7",
      "scores"
    ],
    "h1": "KoboldCpp vs vLLM",
    "image": "https://www.anchorterminal.com/assets/og/compare-koboldcpp-vs-vllm.png",
    "path": "/compare/koboldcpp-vs-vllm",
    "published": "2026-10-01",
    "section": "tools",
    "title": "KoboldCpp vs vLLM for AI agents in 2026: scores and prices",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-vllm"
  },
  "tokens": {
    "markdown": 2550,
    "slim": 530
  },
  "version": 1
}
