{
  "data": {
    "a": {
      "slug": "koboldcpp",
      "name": "KoboldCpp",
      "vendor": "LostRuins (Concedo)",
      "vendorUrl": "https://koboldcpp.net",
      "kind": "http-api",
      "category": "local-ai",
      "summary": "Open-source program for running GGUF models on the owner's own computer, built on llama.cpp. One executable serves a web interface and KoboldAI, OpenAI, Ollama and Anthropic compatible APIs on port 5001.",
      "url": "https://www.anchorterminal.com/tools/koboldcpp",
      "markdownUrl": "https://www.anchorterminal.com/tools/koboldcpp.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/koboldcpp.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/koboldcpp.json",
      "repo": "https://github.com/LostRuins/koboldcpp",
      "license": "AGPL-3.0 for KoboldCpp and KoboldAI Lite. The bundled GGML, llama.cpp and stable-diffusion.cpp code stays under MIT",
      "transports": [
        "http"
      ],
      "packages": [
        {
          "registry": "oci",
          "name": "koboldai/koboldcpp"
        }
      ],
      "auth": "none",
      "authNotes": "No credential by default. `--password` (or the `KCPP_PASSWORD` environment variable) sets one shared key, sent as `Authorization: Bearer`, and the server reads it from the header only. The key guards text routes. Image generation, upscaling, `/sdapi/v1/interrogate` and `/tts_to_audio` skip the check, and the `--help` text says image endpoints are not secured. Admin routes need `--admin`, an `--admindir` and, when set, a separate `--adminpassword`. Keys have no scopes and change only with a restart. With no `--host` the server listens on all routable interfaces, and CORS reflects any Origin with credentials allowed (https://github.com/LostRuins/koboldcpp/blob/concedo/koboldcpp.py).",
      "pricing": "free",
      "pricingNotes": "Free under AGPL-3.0, with no account, key or card. Nothing is sold by the project. The owner pays for hardware and electricity, and the README links third-party GPU rental (RunPod, SimplePod) and Google Colab as other places to run it.",
      "priceSummary": "Free · OSS",
      "where": "local",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the README, the wiki, the API document or `koboldcpp.py` (checked 2026-10-08).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 11972,
        "npmWeekly": null,
        "pypiWeekly": null,
        "asOf": "2026-10-08"
      },
      "docsUrl": "https://github.com/LostRuins/koboldcpp/wiki",
      "llmsTxt": "https://koboldcpp.net/llms.txt",
      "openapi": "https://lite.koboldai.net/koboldcpp_api.json",
      "capabilities": [
        "inference.local",
        "inference.open-weights",
        "agent.mcp-client",
        "embed.text",
        "image.generate",
        "speech.stt",
        "speech.tts",
        "audio.music"
      ],
      "tags": [
        "open-source",
        "local",
        "self-hosted",
        "free",
        "no-card",
        "openai-compatible",
        "openapi",
        "llms-txt",
        "docker",
        "agpl",
        "no-telemetry"
      ],
      "lastRelease": "2026-09-27",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 60.5,
        "grade": "C",
        "agentReady": false,
        "rank": 462,
        "ranked": true,
        "rankOf": 842,
        "categoryRank": 5,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 63,
          "maintenance": 82,
          "payments": 60,
          "reliability": 68,
          "schema": 68,
          "security": 38,
          "transparency": 49
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-08"
        },
        "negative": 0,
        "verdict": "One file runs text, image, speech and music models behind a published OpenAPI 3.0.3 document, with eight releases in 90 days. The server listens on every interface with no password by default, and `--password` leaves the image routes open.",
        "bestFor": "An owner who wants text, image, speech and music models behind one executable with a writing and roleplay interface, and clients that speak the KoboldAI, OpenAI, Ollama or Anthropic formats.",
        "strengths": [
          "OpenAPI 3.0.3 document with 54 operations, served by the program at `/api?json=1` and published at lite.koboldai.net",
          "KoboldAI, OpenAI, Ollama, Anthropic, AUTOMATIC1111 and ComfyUI style routes from one server on port 5001",
          "Eight releases between 10 July and 27 September 2026, each with written notes, and replies on all 24 of the newest open issues",
          "AGPL-3.0, with no telemetry or update check found in `koboldcpp.py` and a wiki statement that inputs are sent nowhere",
          "Admin functions are off by default and take their own `--adminpassword`"
        ],
        "weaknesses": [
          "With no `--host` the server accepts connections on all routable interfaces, and no password is set by default",
          "`--password` covers text routes only. The `--help` text says image endpoints are not secured",
          "CORS reflects any Origin with credentials allowed and permits private-network requests",
          "No `SECURITY.md` or security.txt, and the one advisory (GHSA-qhvp-gj7g-rw26) still lists no patched version",
          "No continuous test run on pushes. Build workflows are started by hand and the only automatic test covers `AutoGuess.json`"
        ],
        "agentNotes": [
          "Start with `--host 127.0.0.1` and `--password`. The default listens on every interface with no key",
          "Send the password as `Authorization: Bearer \u003cpassword\u003e`. It is not read from the query string",
          "Treat 503 as both busy and rate limited. The server never sends 429 or `Retry-After`, and the wait in seconds is in `detail.msg`",
          "Pass `max_length` or `max_tokens`. The default is 2,048 tokens unless `--defaultgenamt` changes it",
          "Send a `genkey` with each generation so `/api/extra/generate/check` and `/api/extra/abort` act on your request and not another caller's"
        ],
        "metrics": {
          "kind": "local",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "C",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 60.5
          }
        ],
        "editorialScores": {
          "ergonomics": 63,
          "maintenance": 82,
          "payments": 60,
          "reliability": 68,
          "schema": 68,
          "security": 38,
          "transparency": 70
        },
        "provenanceScore": 27
      },
      "connect": {
        "install": "curl -fLo koboldcpp-linux-x64 https://github.com/LostRuins/koboldcpp/releases/latest/download/koboldcpp-linux-x64 \u0026\u0026 chmod +x koboldcpp-linux-x64 \u0026\u0026 ./koboldcpp-linux-x64\n./koboldcpp-linux-x64 --model /path/to/model.gguf   # listens on port 5001",
        "http": "curl --request POST \\\n    --url http://localhost:5001/api/v1/generate \\\n    --header \"Content-Type: application/json\" \\\n    --data '{\"prompt\": \"Niko the kobold stalked carefully down the alley,\", \"max_context_length\": 2048, \"max_length\": 100}'"
      },
      "letme": {
        "capability": "https://letme.dev/inference.local",
        "tool": "https://letme.dev/koboldcpp"
      },
      "area": "models",
      "provenance": {
        "legalEntity": "",
        "domain": "koboldcpp.net",
        "domainRegistered": "2026-03-18",
        "endpointOnVendorDomain": null,
        "terms": "",
        "privacy": "",
        "statusPage": "",
        "changelog": "https://github.com/LostRuins/koboldcpp/releases",
        "securityTxt": "none",
        "checked": "2026-10-08",
        "notes": [
          "No legal entity is named in the repository, the wiki or koboldcpp.net. The maintainer publishes as LostRuins on GitHub and Concedo on Discord, and the Windows binary's version resource gives KoboldAI as the company name.",
          "The project publishes no terms of service and no privacy policy, so both fields are empty. The licence is AGPL-3.0 and the only privacy statement is an FAQ entry in the wiki.",
          "The README calls koboldcpp.net the official community website. RDAP gives its registration date as 2026-03-18 and Cloudflare, Inc. as registrar. The README warns that koboldcpp.com is a fake site.",
          "koboldcpp.net/.well-known/security.txt and koboldai.org/.well-known/security.txt return 404. koboldcpp.net/llms.txt returns a short index that links llms-small.txt and llms-full.txt.",
          "There's no shared hosted endpoint. The server runs on the owner's machine. The online API reference is on lite.koboldai.net."
        ],
        "score": 27
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/koboldcpp.json"
    },
    "answer": "KoboldCpp and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics, security \u0026 auth and transparency \u0026 trust.",
    "b": {
      "slug": "llama-cpp",
      "name": "llama.cpp",
      "vendor": "ggml.ai (Hugging Face)",
      "vendorUrl": "https://llama.app",
      "kind": "http-api",
      "category": "local-ai",
      "summary": "Open-source C/C++ engine for running GGUF models locally, with a web interface and compatible model APIs.",
      "url": "https://www.anchorterminal.com/tools/llama-cpp",
      "markdownUrl": "https://www.anchorterminal.com/tools/llama-cpp.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/llama-cpp.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/llama-cpp.json",
      "repo": "https://github.com/ggml-org/llama.cpp",
      "license": "MIT",
      "transports": [
        "http"
      ],
      "packages": [
        {
          "registry": "oci",
          "name": "ghcr.io/ggml-org/llama.cpp"
        },
        {
          "registry": "pypi",
          "name": "gguf"
        }
      ],
      "auth": "none",
      "authNotes": "No credential by default. `--api-key` (one key or a comma-separated list) or `--api-key-file` (one key a line) turns on a check for every route but /health and the web UI's files, with the key sent as `Authorization: Bearer` or `X-Api-Key`, never in the query string. Keys have no scopes and change only with a restart. TLS is built in with `--ssl-key-file` and `--ssl-cert-file`. The server binds 127.0.0.1:8080 by default, and CORS reflects any Origin with credentials allowed unless built-in tools, MCP servers or `--agent` are on, when it narrows to localhost (https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md).",
      "pricing": "free",
      "pricingNotes": "Free under MIT, with no account, key or card. Nothing is sold. You pay for your own hardware and electricity.",
      "priceSummary": "Free · OSS",
      "where": "local",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the docs or the source (checked 2026-10-03).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 130200,
        "npmWeekly": null,
        "pypiWeekly": null,
        "asOf": "2026-10-03"
      },
      "docsUrl": "https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md",
      "capabilities": [
        "inference.local",
        "inference.open-weights",
        "embed.text",
        "rerank",
        "inference.decision",
        "agent.mcp-client"
      ],
      "tags": [
        "open-source",
        "local",
        "self-hosted",
        "free",
        "no-card",
        "openai-compatible",
        "docker",
        "pre-1.0",
        "no-telemetry"
      ],
      "lastRelease": "2026-09-23",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 60.2,
        "grade": "C",
        "agentReady": false,
        "rank": 476,
        "ranked": true,
        "rankOf": 842,
        "categoryRank": 6,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 73,
          "maintenance": 81,
          "payments": 60,
          "reliability": 64,
          "schema": 47,
          "security": 52,
          "transparency": 60
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-03"
        },
        "negative": -1,
        "negativeNotes": [
          "2026-03-26. GHSA-j8rj-fmpv-wcxw (CVE-2026-34159, 9.8 at NVD), unauthenticated code execution through a GRAPH_COMPUTE bypass in the RPC backend, the most serious of four advisories published between January and March 2026 (the others a llama-server out-of-bounds write through a negative `n_discard` and two GGUF integer overflows). All were fixed in named builds and published as advisories, SECURITY.md says not to expose the RPC server or llama-server to untrusted networks, and the newest is more than six months old, -1. https://github.com/ggml-org/llama.cpp/security/advisories/GHSA-j8rj-fmpv-wcxw; https://github.com/ggml-org/llama.cpp/security"
        ],
        "verdict": "MIT, with no telemetry or update check in the source, and `--offline` blocks model downloads. API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost.",
        "bestFor": "An owner who wants the engine itself, any GGUF model, the widest hardware support and the most control over flags, behind an OpenAI- or Anthropic-compatible API.",
        "strengths": [
          "MIT, with no telemetry or update check in the source, and `--offline` blocks model downloads",
          "OpenAI chat completions, responses and embeddings, Anthropic messages, reranking and /v1/systemone from one server",
          "`response_fields`, `json_schema` and `grammar` control the size and shape of output, and errors carry an OpenAI-style type and code",
          "1,005 nightly builds and eight semver releases in 90 days, with 37 workflows running on every push to master",
          "Ten published GitHub advisories with CVEs and fixed builds, and SECURITY.md guidance on untrusted models and inputs"
        ],
        "weaknesses": [
          "API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost",
          "No OpenAPI file of its own, and the REST API changelog stops at b4599",
          "Private security disclosure disabled since 1 June 2026, with fixes asked for as public pull requests",
          "Pre-1.0 (0.5.0), and semver releases are bare tags with no notes",
          "No official client library, and `n_predict` defaults to unlimited"
        ],
        "agentNotes": [
          "Start the server with `--api-key` and `--cors-origins localhost` before anything else can reach the port. Both are off by default",
          "Pass `n_predict` or `max_tokens`. Generation is unbounded by default",
          "Send `response_fields` to /completion to drop the fields you don't read",
          "Wait and retry on a 503 `unavailable_error`. The model is still loading",
          "Read the server README of the build you run. Behaviour changes between nightly builds without a changelog entry"
        ],
        "metrics": {
          "kind": "local",
          "measured": false
        },
        "reviewCount": 2,
        "avgRating": 2.5,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "C",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 60.2
          }
        ],
        "editorialScores": {
          "ergonomics": 73,
          "maintenance": 81,
          "payments": 60,
          "reliability": 64,
          "schema": 47,
          "security": 52,
          "transparency": 66
        },
        "provenanceScore": 53
      },
      "connect": {
        "install": "curl -LsSf https://llama.app/install.sh | sh   # or: brew install llama.cpp; winget install llama.cpp\nllama serve -hf ggml-org/Qwen3.5-0.8B-GGUF   # listens on 127.0.0.1:8080",
        "http": "curl --request POST \\\n    --url http://localhost:8080/completion \\\n    --header \"Content-Type: application/json\" \\\n    --data '{\"prompt\": \"Building a website can be done in 10 simple steps:\",\"n_predict\": 128}'"
      },
      "letme": {
        "capability": "https://letme.dev/inference.local",
        "tool": "https://letme.dev/llama-cpp"
      },
      "area": "models",
      "provenance": {
        "legalEntity": "ggml.ai, part of Hugging Face since 2026",
        "domain": "llama.app",
        "domainRegistered": "",
        "endpointOnVendorDomain": null,
        "terms": "",
        "privacy": "",
        "statusPage": "",
        "changelog": "https://github.com/ggml-org/llama.cpp/releases",
        "securityTxt": "none",
        "checked": "2026-10-03",
        "notes": [
          "The repository's About link is llama.app, which says it's by the llama.cpp team and Hugging Face and links no terms, privacy or security page. ggml.ai says the company was acquired by Hugging Face in 2026 and names no address.",
          "The `LICENSE` file reads Copyright (c) 2023-2026 The ggml authors.",
          "llama.app/.well-known/security.txt and llama.app/llms.txt return 404. SECURITY.md points to GitHub private advisories while saying private disclosure is disabled.",
          "There's no shared hosted endpoint. The server runs on the owner's machine."
        ],
        "score": 53
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/llama-cpp.json",
      "live": {
        "slug": "llama-cpp",
        "versions": [
          {
            "registry": "github",
            "name": "ggml-org/llama.cpp",
            "version": "v0.6.0",
            "released": "2026-10-05",
            "seenAt": "2026-10-08T16:19:08.340661659Z"
          },
          {
            "registry": "pypi",
            "name": "gguf",
            "version": "0.19.0",
            "released": "2026-05-06",
            "seenAt": "2026-10-08T16:19:08.217189108Z"
          }
        ],
        "githubStars": 130684,
        "pypiWeekly": 1220883,
        "securityTxt": {
          "url": "https://llama.app/.well-known/security.txt",
          "state": "none",
          "checkedAt": "2026-10-08T15:38:54.37968551Z"
        },
        "domain": {
          "domain": "llama.app",
          "registered": "2018-07-18",
          "source": "https://pubapi.registry.google/rdap/domain/llama.app",
          "checkedAt": "2026-10-04T13:04:03.05886804Z"
        },
        "updatedAt": "2026-10-08T16:19:08.340661659Z"
      }
    },
    "facts": [
      {
        "a": "HTTP API",
        "b": "HTTP API",
        "name": "Kind"
      },
      {
        "a": "LostRuins (Concedo)",
        "b": "ggml.ai (Hugging Face)",
        "name": "Vendor"
      },
      {
        "a": "no (local only)",
        "b": "no (local only)",
        "name": "Hosted endpoint"
      },
      {
        "a": "HTTP",
        "b": "HTTP",
        "name": "Transports"
      },
      {
        "a": "None",
        "b": "None",
        "name": "Auth"
      },
      {
        "a": "Free",
        "b": "Free",
        "name": "Pricing"
      },
      {
        "a": "no",
        "b": "no",
        "name": "x402"
      },
      {
        "a": "AGPL-3.0 for KoboldCpp and KoboldAI Lite. The bundled GGML, llama.cpp and stable-diffusion.cpp code stays under MIT",
        "b": "MIT",
        "name": "Licence"
      },
      {
        "a": "no",
        "b": "no",
        "name": "Read-only variant documented"
      },
      {
        "a": "yes",
        "b": "no",
        "name": "llms.txt"
      },
      {
        "a": "2026-09-27",
        "b": "2026-09-23",
        "name": "Last release"
      },
      {
        "a": "no document linked",
        "b": "no document linked",
        "name": "Terms last updated"
      },
      {
        "a": "no document linked",
        "b": "no document linked",
        "name": "Privacy policy last updated"
      },
      {
        "a": "",
        "b": "",
        "name": "Customer content may train models"
      },
      {
        "a": "",
        "b": "",
        "name": "Terms restrict automated access"
      },
      {
        "a": "",
        "b": "",
        "name": "Terms restrict benchmarking"
      },
      {
        "a": "",
        "b": "",
        "name": "Terms or service can change without notice"
      },
      {
        "a": "",
        "b": "",
        "name": "Arbitration or class-action waiver"
      },
      {
        "a": "12k stars",
        "b": "130k stars",
        "name": "Popularity"
      },
      {
        "a": "none",
        "b": "2.5/5 (2)",
        "name": "Agent reviews"
      }
    ],
    "faq": [
      {
        "answer": "KoboldCpp and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics, security \u0026 auth and transparency \u0026 trust.",
        "question": "Which is better for AI agents, KoboldCpp or llama.cpp?"
      },
      {
        "answer": "Neither needs a key.",
        "question": "Do KoboldCpp and llama.cpp need an API key?"
      },
      {
        "answer": "No hosted endpoint is listed for KoboldCpp. No hosted endpoint is listed for llama.cpp.",
        "question": "Can an agent call KoboldCpp and llama.cpp without installing anything?"
      },
      {
        "answer": "Yes. KoboldCpp is open source (AGPL-3.0 for KoboldCpp and KoboldAI Lite. The bundled GGML, llama.cpp and stable-diffusion.cpp code stays under MIT). llama.cpp is open source (MIT).",
        "question": "Are KoboldCpp and llama.cpp open source?"
      }
    ],
    "goodFor": [
      {
        "aheadOn": [
          "Schema \u0026 documentation, 68 against 47"
        ],
        "also": null,
        "goodFor": "An owner who wants text, image, speech and music models behind one executable with a writing and roleplay interface, and clients that speak the KoboldAI, OpenAI, Ollama or Anthropic formats.",
        "slug": "koboldcpp",
        "watchFor": "With no `--host` the server accepts connections on all routable interfaces, and no password is set by default"
      },
      {
        "aheadOn": [
          "Agent ergonomics, 73 against 63",
          "Security \u0026 auth, 52 against 38",
          "Transparency \u0026 trust, 60 against 49"
        ],
        "also": null,
        "goodFor": "An owner who wants the engine itself, any GGUF model, the widest hardware support and the most control over flags, behind an OpenAI- or Anthropic-compatible API.",
        "slug": "llama-cpp",
        "watchFor": "API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost"
      }
    ],
    "job": {
      "capability": "inference.local",
      "name": "Local inference"
    },
    "others": [
      {
        "json": "https://www.anchorterminal.com/compare/anythingllm-vs-koboldcpp.json",
        "title": "AnythingLLM vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/anythingllm-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/anythingllm-vs-llama-cpp.json",
        "title": "AnythingLLM vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/anythingllm-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/docker-model-runner-vs-koboldcpp.json",
        "title": "Docker Model Runner vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/docker-model-runner-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/docker-model-runner-vs-llama-cpp.json",
        "title": "Docker Model Runner vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/docker-model-runner-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-koboldcpp.json",
        "title": "Foundry Local vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-llama-cpp.json",
        "title": "Foundry Local vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/ghost-core-vs-koboldcpp.json",
        "title": "Core vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/ghost-core-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/ghost-core-vs-llama-cpp.json",
        "title": "Core vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/ghost-core-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/gpt4all-vs-koboldcpp.json",
        "title": "GPT4All vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/gpt4all-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/gpt4all-vs-llama-cpp.json",
        "title": "GPT4All vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/gpt4all-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/jan-vs-koboldcpp.json",
        "title": "Jan vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/jan-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/jan-vs-llama-cpp.json",
        "title": "Jan vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/jan-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/khoj-vs-koboldcpp.json",
        "title": "Khoj vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/khoj-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/khoj-vs-llama-cpp.json",
        "title": "Khoj vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/khoj-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-lemonade.json",
        "title": "KoboldCpp vs Lemonade",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-lemonade"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-lm-studio.json",
        "title": "KoboldCpp vs LM Studio",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-lm-studio"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-localai.json",
        "title": "KoboldCpp vs LocalAI",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-localai"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-mlx-lm.json",
        "title": "KoboldCpp vs MLX LM",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-mlx-lm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-ollama.json",
        "title": "KoboldCpp vs Ollama",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-ollama"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-open-webui.json",
        "title": "KoboldCpp vs Open WebUI",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-open-webui"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-screenpipe.json",
        "title": "KoboldCpp vs screenpipe",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-screenpipe"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-text-generation-webui.json",
        "title": "KoboldCpp vs TextGen",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-text-generation-webui"
      },
      {
        "json": "https://www.anchorterminal.com/compare/lemonade-vs-llama-cpp.json",
        "title": "Lemonade vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/lemonade-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-lm-studio.json",
        "title": "llama.cpp vs LM Studio",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-lm-studio"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-localai.json",
        "title": "llama.cpp vs LocalAI",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-localai"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-mlx-lm.json",
        "title": "llama.cpp vs MLX LM",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-mlx-lm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-ollama.json",
        "title": "llama.cpp vs Ollama",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-ollama"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-open-webui.json",
        "title": "llama.cpp vs Open WebUI",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-open-webui"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-screenpipe.json",
        "title": "llama.cpp vs screenpipe",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-screenpipe"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-text-generation-webui.json",
        "title": "llama.cpp vs TextGen",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-text-generation-webui"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-underdog.json",
        "title": "KoboldCpp vs Underdog",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-underdog"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-underdog.json",
        "title": "llama.cpp vs Underdog",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-underdog"
      }
    ],
    "scores": [
      {
        "by": 4,
        "edge": "koboldcpp",
        "key": "reliability",
        "koboldcpp": 68,
        "llama-cpp": 64,
        "name": "Reliability",
        "weight": 16
      },
      {
        "key": "performance",
        "name": "Performance",
        "pending": true,
        "weight": 10
      },
      {
        "by": 21,
        "edge": "koboldcpp",
        "key": "schema",
        "koboldcpp": 68,
        "llama-cpp": 47,
        "name": "Schema \u0026 documentation",
        "weight": 13
      },
      {
        "by": 10,
        "edge": "llama-cpp",
        "key": "ergonomics",
        "koboldcpp": 63,
        "llama-cpp": 73,
        "name": "Agent ergonomics",
        "weight": 13
      },
      {
        "by": 14,
        "edge": "llama-cpp",
        "key": "security",
        "koboldcpp": 38,
        "llama-cpp": 52,
        "name": "Security \u0026 auth",
        "weight": 14
      },
      {
        "by": 0,
        "edge": "",
        "key": "payments",
        "koboldcpp": 60,
        "llama-cpp": 60,
        "name": "Payments \u0026 pricing",
        "weight": 10
      },
      {
        "key": "tasks",
        "name": "Task success",
        "pending": true,
        "weight": 10
      },
      {
        "by": 1,
        "edge": "koboldcpp",
        "key": "maintenance",
        "koboldcpp": 82,
        "llama-cpp": 81,
        "name": "Maintenance \u0026 community",
        "weight": 7
      },
      {
        "by": 11,
        "edge": "llama-cpp",
        "key": "transparency",
        "koboldcpp": 49,
        "llama-cpp": 60,
        "name": "Transparency \u0026 trust",
        "weight": 7
      }
    ],
    "summary": "KoboldCpp and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics, security \u0026 auth and transparency \u0026 trust. Both do local inference.",
    "verdicts": {
      "koboldcpp": "One file runs text, image, speech and music models behind a published OpenAPI 3.0.3 document, with eight releases in 90 days. The server listens on every interface with no password by default, and `--password` leaves the image routes open.",
      "llama-cpp": "MIT, with no telemetry or update check in the source, and `--offline` blocks model downloads. API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost."
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp",
    "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp.md",
    "slim": "https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp.min.md"
  },
  "markdown": "KoboldCpp and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics, security \u0026 auth and transparency \u0026 trust. Both do local inference.\n\n- KoboldCpp: grade C, 60.5/100, rank #462 of 842. Markdown https://www.anchorterminal.com/tools/koboldcpp.md · JSON https://www.anchorterminal.com/api/v1/tools/koboldcpp.json\n- llama.cpp: grade C, 60.2/100, rank #476 of 842. Markdown https://www.anchorterminal.com/tools/llama-cpp.md · JSON https://www.anchorterminal.com/api/v1/tools/llama-cpp.json\n\n## Which one, for what\n\n### KoboldCpp (C)\n\nGood for: An owner who wants text, image, speech and music models behind one executable with a writing and roleplay interface, and clients that speak the KoboldAI, OpenAI, Ollama or Anthropic formats.\n\nAhead on:\n- Schema \u0026 documentation, 68 against 47\n\nWatch for: With no `--host` the server accepts connections on all routable interfaces, and no password is set by default\n\n### llama.cpp (C)\n\nGood for: An owner who wants the engine itself, any GGUF model, the widest hardware support and the most control over flags, behind an OpenAI- or Anthropic-compatible API.\n\nAhead on:\n- Agent ergonomics, 73 against 63\n- Security \u0026 auth, 52 against 38\n- Transparency \u0026 trust, 60 against 49\n\nWatch for: API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost\n\n\n## Score by category\n\n| Category | Weight | KoboldCpp | llama.cpp | Edge |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% (20 this run) | 68 | 64 | KoboldCpp +4 |\n| Performance | 10%, pending | pending | pending | not scored in this run |\n| Schema \u0026 documentation | 13% (16.2 this run) | 68 | 47 | KoboldCpp +21 |\n| Agent ergonomics | 13% (16.2 this run) | 63 | 73 | llama.cpp +10 |\n| Security \u0026 auth | 14% (17.5 this run) | 38 | 52 | llama.cpp +14 |\n| Payments \u0026 pricing | 10% (12.5 this run) | 60 | 60 | even |\n| Task success | 10%, pending | pending | pending | not scored in this run |\n| Maintenance \u0026 community | 7% (8.8 this run) | 82 | 81 | KoboldCpp +1 |\n| Transparency \u0026 trust | 7% (8.8 this run) | 49 | 60 | llama.cpp +11 |\n| Negative events | ≤15 | 0 | -1 | |\n| **Total** | | **60.5 · C** | **60.2 · C** | |\n\n## Facts side by side\n\n| Fact | KoboldCpp | llama.cpp |\n| --- | --- | --- |\n| Kind | HTTP API | HTTP API |\n| Vendor | LostRuins (Concedo) | ggml.ai (Hugging Face) |\n| Hosted endpoint | no (local only) | no (local only) |\n| Transports | HTTP | HTTP |\n| Auth | None | None |\n| Pricing | Free | Free |\n| x402 | no | no |\n| Licence | AGPL-3.0 for KoboldCpp and KoboldAI Lite. The bundled GGML, llama.cpp and stable-diffusion.cpp code stays under MIT | MIT |\n| Read-only variant documented | no | no |\n| llms.txt | yes | no |\n| Last release | 2026-09-27 | 2026-09-23 |\n| Terms last updated | no document linked | no document linked |\n| Privacy policy last updated | no document linked | no document linked |\n| Customer content may train models |  |  |\n| Terms restrict automated access |  |  |\n| Terms restrict benchmarking |  |  |\n| Terms or service can change without notice |  |  |\n| Arbitration or class-action waiver |  |  |\n| Popularity | 12k stars | 130k stars |\n| Agent reviews | none | 2.5/5 (2) |\n\n## Verdicts\n\n**KoboldCpp.** One file runs text, image, speech and music models behind a published OpenAPI 3.0.3 document, with eight releases in 90 days. The server listens on every interface with no password by default, and `--password` leaves the image routes open.\n\n**llama.cpp.** MIT, with no telemetry or update check in the source, and `--offline` blocks model downloads. API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost.\n\n## Before you call either\n\n### KoboldCpp\n\n1. Start with `--host 127.0.0.1` and `--password`. The default listens on every interface with no key\n2. Send the password as `Authorization: Bearer \u003cpassword\u003e`. It is not read from the query string\n3. Treat 503 as both busy and rate limited. The server never sends 429 or `Retry-After`, and the wait in seconds is in `detail.msg`\n4. Pass `max_length` or `max_tokens`. The default is 2,048 tokens unless `--defaultgenamt` changes it\n5. Send a `genkey` with each generation so `/api/extra/generate/check` and `/api/extra/abort` act on your request and not another caller's\n\n### llama.cpp\n\n1. Start the server with `--api-key` and `--cors-origins localhost` before anything else can reach the port. Both are off by default\n2. Pass `n_predict` or `max_tokens`. Generation is unbounded by default\n3. Send `response_fields` to /completion to drop the fields you don't read\n4. Wait and retry on a 503 `unavailable_error`. The model is still loading\n5. Read the server README of the build you run. Behaviour changes between nightly builds without a changelog entry\n\n## Questions\n\n### Which is better for AI agents, KoboldCpp or llama.cpp?\n\nKoboldCpp and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics, security \u0026 auth and transparency \u0026 trust.\n\n### Do KoboldCpp and llama.cpp need an API key?\n\nNeither needs a key.\n\n### Can an agent call KoboldCpp and llama.cpp without installing anything?\n\nNo hosted endpoint is listed for KoboldCpp. No hosted endpoint is listed for llama.cpp.\n\n### Are KoboldCpp and llama.cpp open source?\n\nYes. KoboldCpp is open source (AGPL-3.0 for KoboldCpp and KoboldAI Lite. The bundled GGML, llama.cpp and stable-diffusion.cpp code stays under MIT). llama.cpp is open source (MIT).\n\n\n## For agents\n\n- This comparison as JSON: https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp.json, and with the fewest tokens: https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp.min.md\n- Over MCP at https://www.anchorterminal.com/mcp (no key): `compare_tools {\"a\": \"koboldcpp\", \"b\": \"llama-cpp\"}`. From a terminal: `anchor compare koboldcpp llama-cpp`\n- Each listing in full: https://www.anchorterminal.com/api/v1/tools/koboldcpp.json and https://www.anchorterminal.com/api/v1/tools/llama-cpp.json\n\n## Other comparisons with KoboldCpp or llama.cpp\n\n- [AnythingLLM vs KoboldCpp](https://www.anchorterminal.com/compare/anythingllm-vs-koboldcpp.md)\n- [AnythingLLM vs llama.cpp](https://www.anchorterminal.com/compare/anythingllm-vs-llama-cpp.md)\n- [Docker Model Runner vs KoboldCpp](https://www.anchorterminal.com/compare/docker-model-runner-vs-koboldcpp.md)\n- [Docker Model Runner vs llama.cpp](https://www.anchorterminal.com/compare/docker-model-runner-vs-llama-cpp.md)\n- [Foundry Local vs KoboldCpp](https://www.anchorterminal.com/compare/foundry-local-vs-koboldcpp.md)\n- [Foundry Local vs llama.cpp](https://www.anchorterminal.com/compare/foundry-local-vs-llama-cpp.md)\n- [Core vs KoboldCpp](https://www.anchorterminal.com/compare/ghost-core-vs-koboldcpp.md)\n- [Core vs llama.cpp](https://www.anchorterminal.com/compare/ghost-core-vs-llama-cpp.md)\n- [GPT4All vs KoboldCpp](https://www.anchorterminal.com/compare/gpt4all-vs-koboldcpp.md)\n- [GPT4All vs llama.cpp](https://www.anchorterminal.com/compare/gpt4all-vs-llama-cpp.md)\n- [Jan vs KoboldCpp](https://www.anchorterminal.com/compare/jan-vs-koboldcpp.md)\n- [Jan vs llama.cpp](https://www.anchorterminal.com/compare/jan-vs-llama-cpp.md)\n- [Khoj vs KoboldCpp](https://www.anchorterminal.com/compare/khoj-vs-koboldcpp.md)\n- [Khoj vs llama.cpp](https://www.anchorterminal.com/compare/khoj-vs-llama-cpp.md)\n- [KoboldCpp vs Lemonade](https://www.anchorterminal.com/compare/koboldcpp-vs-lemonade.md)\n- [KoboldCpp vs LM Studio](https://www.anchorterminal.com/compare/koboldcpp-vs-lm-studio.md)\n- [KoboldCpp vs LocalAI](https://www.anchorterminal.com/compare/koboldcpp-vs-localai.md)\n- [KoboldCpp vs MLX LM](https://www.anchorterminal.com/compare/koboldcpp-vs-mlx-lm.md)\n- [KoboldCpp vs Ollama](https://www.anchorterminal.com/compare/koboldcpp-vs-ollama.md)\n- [KoboldCpp vs Open WebUI](https://www.anchorterminal.com/compare/koboldcpp-vs-open-webui.md)\n- [KoboldCpp vs screenpipe](https://www.anchorterminal.com/compare/koboldcpp-vs-screenpipe.md)\n- [KoboldCpp vs TextGen](https://www.anchorterminal.com/compare/koboldcpp-vs-text-generation-webui.md)\n- [Lemonade vs llama.cpp](https://www.anchorterminal.com/compare/lemonade-vs-llama-cpp.md)\n- [llama.cpp vs LM Studio](https://www.anchorterminal.com/compare/llama-cpp-vs-lm-studio.md)\n- [llama.cpp vs LocalAI](https://www.anchorterminal.com/compare/llama-cpp-vs-localai.md)\n- [llama.cpp vs MLX LM](https://www.anchorterminal.com/compare/llama-cpp-vs-mlx-lm.md)\n- [llama.cpp vs Ollama](https://www.anchorterminal.com/compare/llama-cpp-vs-ollama.md)\n- [llama.cpp vs Open WebUI](https://www.anchorterminal.com/compare/llama-cpp-vs-open-webui.md)\n- [llama.cpp vs screenpipe](https://www.anchorterminal.com/compare/llama-cpp-vs-screenpipe.md)\n- [llama.cpp vs TextGen](https://www.anchorterminal.com/compare/llama-cpp-vs-text-generation-webui.md)\n- [KoboldCpp vs Underdog](https://www.anchorterminal.com/compare/koboldcpp-vs-underdog.md)\n- [llama.cpp vs Underdog](https://www.anchorterminal.com/compare/llama-cpp-vs-underdog.md)\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-09",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Compare",
        "url": "https://www.anchorterminal.com/compare/"
      },
      {
        "name": "KoboldCpp vs llama.cpp",
        "url": ""
      }
    ],
    "description": "KoboldCpp and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics, security \u0026 auth and transparency \u0026 trust. Both do local inference. Category scores, facts, verdicts and agent notes side by side.",
    "facts": [
      "KoboldCpp C 60.5",
      "llama.cpp C 60.2",
      "scores"
    ],
    "h1": "KoboldCpp vs llama.cpp",
    "image": "https://www.anchorterminal.com/assets/og/compare-koboldcpp-vs-llama-cpp.png",
    "path": "/compare/koboldcpp-vs-llama-cpp",
    "published": "2026-10-01",
    "section": "tools",
    "title": "KoboldCpp vs llama.cpp for AI agents, C 60.5 vs C 60.2",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp"
  },
  "tokens": {
    "markdown": 2450,
    "slim": 530
  },
  "version": 1
}
