{
  "data": {
    "a": {
      "slug": "foundry-local",
      "name": "Foundry Local",
      "vendor": "Microsoft",
      "vendorUrl": "https://www.foundrylocal.ai",
      "kind": "sdk",
      "category": "local-ai",
      "summary": "Microsoft's on-device model runtime, built on ONNX Runtime. Applications embed it through SDKs for C#, JavaScript, Python and Rust, and it can start an optional OpenAI-compatible server on localhost. A preview CLI is also available.",
      "url": "https://www.anchorterminal.com/tools/foundry-local",
      "markdownUrl": "https://www.anchorterminal.com/tools/foundry-local.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/foundry-local.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/foundry-local.json",
      "repo": "https://github.com/microsoft/Foundry-Local",
      "license": "MIT for the SDKs and the v2 native runtime. The CLI is closed source under Microsoft Software Licence Terms. Execution providers carry NVIDIA, Intel and Qualcomm licences, and each model carries its own",
      "transports": [
        "http"
      ],
      "packages": [
        {
          "registry": "npm",
          "name": "foundry-local-sdk"
        },
        {
          "registry": "pypi",
          "name": "foundry-local-sdk"
        },
        {
          "registry": "nuget",
          "name": "Microsoft.AI.Foundry.Local"
        },
        {
          "registry": "cargo",
          "name": "foundry-local-sdk"
        }
      ],
      "auth": "none",
      "authNotes": "No authentication. The SDK runs in the application's own process, and the optional local server takes no key or token. It binds to 127.0.0.1 on a dynamic port unless the owner configures `web.urls` or starts the CLI daemon with `--port`. No account or Azure subscription is needed.",
      "pricing": "free",
      "pricingNotes": "Free, with no account, card or Azure subscription. The SDK is MIT and the CLI is a free download under Microsoft's licence terms. Microsoft states there are no per-token costs. Foundry Local on Azure Local is a separate product for servers and is not covered here (checked 2026-10-08).",
      "priceSummary": "Free · OSS",
      "where": "local",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the docs, the README or the SDK source (checked 2026-10-08).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 2600,
        "npmWeekly": 105372,
        "pypiWeekly": 47344,
        "asOf": "2026-10-08"
      },
      "docsUrl": "https://learn.microsoft.com/en-us/azure/foundry-local/",
      "capabilities": [
        "inference.local",
        "inference.open-weights",
        "embed.text",
        "speech.stt"
      ],
      "tags": [
        "local",
        "open-source",
        "free",
        "no-card",
        "account-free",
        "no-auth",
        "openai-compatible",
        "csharp",
        "typescript",
        "python",
        "rust",
        "npu",
        "telemetry-on-by-default"
      ],
      "lastRelease": "2026-09-29",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 60.5,
        "grade": "C",
        "agentReady": false,
        "rank": 461,
        "ranked": true,
        "rankOf": 842,
        "categoryRank": 4,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 61,
          "maintenance": 88,
          "payments": 60,
          "reliability": 68,
          "schema": 53,
          "security": 39,
          "transparency": 72
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-08"
        },
        "negative": 0,
        "verdict": "The SDK and its native runtime are MIT, at 2.1.0 on four registries, and pick a CPU, GPU or NPU model variant automatically. The optional local server has no credential, its current routes aren't in the published REST reference, and telemetry is on by default with an opt-out.",
        "bestFor": "An application that ships a model to end users' Windows, macOS or Linux devices and wants NPU and GPU variants chosen automatically, especially on Windows.",
        "strengths": [
          "SDK 2.1.0 published on npm, PyPI, NuGet and crates.io on 29 September 2026, with the SDK and the v2 native runtime under MIT in a public repository",
          "A model alias selects the best variant for the machine's CPU, GPU or NPU, with CUDA, WebGPU, OpenVINO, QNN and Vitis AI execution providers",
          "The v2 server answers `/v1/chat/completions`, `/v1/responses`, `/v1/embeddings`, `/v1/audio/transcriptions` and `/v1/models` in OpenAI's shapes",
          "Six releases in the 90 days to 8 October 2026, and the v2.0.1 notes carry a breaking-changes section",
          "No account, key, card or Azure subscription is needed, and prompts and outputs are processed on the device per the docs"
        ],
        "weaknesses": [
          "The local server takes no credential, and its routes include model load and unload and `POST /shutdown`",
          "The REST reference on Microsoft Learn lists `/openai/*` and `/foundry/list` routes that the v2 runtime source doesn't register, and doesn't cover `/v1/responses`",
          "Telemetry is on by default through Microsoft's 1DS SDK. The opt-out covers non-essential telemetry only, and the Learn FAQ doesn't mention it",
          "The CLI is a closed-source public preview, and its REST reference warns of breaking changes without notice",
          "76 open issues on 8 October 2026, among them a Linux ARM64 segmentation fault (#1182) and several unanswered GPU and NPU detection reports"
        ],
        "agentNotes": [
          "Read the server URL from `manager.urls[0]` or `foundry server status`. The port is dynamic unless the owner sets `web.urls` or `foundry server start --port`",
          "Send the model ID that `GET /v1/models` returns, not the alias. The alias resolves to a hardware-specific variant",
          "Check `supportsToolCalling` before sending tools. Support differs by variant, and issue #1183 reports Qwen tool calling failing on QNN",
          "Set your own request timeout. Inference has no built-in one, and cancellation takes effect only after the current generation step",
          "Ask the owner to set `ORT_TELEMETRY_DISABLED=1` or `disableNonessentialTelemetry` before the manager is created if telemetry must be off"
        ],
        "metrics": {
          "kind": "local",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "C",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 60.5
          }
        ],
        "editorialScores": {
          "ergonomics": 61,
          "maintenance": 88,
          "payments": 60,
          "reliability": 68,
          "schema": 53,
          "security": 39,
          "transparency": 63
        },
        "provenanceScore": 80
      },
      "connect": {
        "install": "pip install foundry-local-sdk   # or: npm install foundry-local-sdk\n# CLI (preview): winget install Microsoft.FoundryLocal   # macOS: brew tap microsoft/foundrylocal \u0026\u0026 brew install foundrylocal",
        "http": "foundry server start --port 39839 --idle-timeout 0\ncurl http://localhost:39839/v1/chat/completions \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"\u003cMODEL_ID\u003e\", \"messages\": [{\"role\": \"user\", \"content\": \"What is the golden ratio?\"}]}'"
      },
      "letme": {
        "capability": "https://letme.dev/inference.local",
        "tool": "https://letme.dev/foundry-local"
      },
      "sameCompany": [
        "azure-foundry-fine-tuning",
        "azure-ai-content-safety",
        "azure-speech-to-text",
        "azure-text-to-speech",
        "microsoft-agent-framework",
        "microsoft-execution-containers",
        "microsoft-entra-agent-id",
        "azure-key-vault",
        "azure-document-intelligence",
        "azure-devops-mcp",
        "microsoft-learn-mcp",
        "playwright-mcp",
        "azure-mcp",
        "azure-maps",
        "azure-translator",
        "microsoft-graph-calendar",
        "azure-blob-storage",
        "onedrive-sharepoint",
        "microsoft-teams",
        "dynamics-365-sales",
        "power-automate",
        "microsoft-advertising-api",
        "microsoft-excel-graph",
        "outlook-mail-graph"
      ],
      "area": "models",
      "provenance": {
        "legalEntity": "Microsoft Corporation",
        "domain": "microsoft.com",
        "domainRegistered": "1991-05-02",
        "endpointOnVendorDomain": null,
        "terms": "",
        "privacy": "",
        "statusPage": "",
        "changelog": "https://github.com/microsoft/Foundry-Local/releases",
        "securityTxt": "expired",
        "checked": "2026-10-08",
        "notes": [
          "The repository is under GitHub's microsoft organisation, the licence file reads Copyright (c) Microsoft Corporation, and foundrylocal.ai names Microsoft Corporation as publisher.",
          "The terms link is the repository's LICENSE file, which holds the MIT licence for the SDK and the Microsoft Software Licence Terms for the CLI. Microsoft publishes no other agreement for Foundry Local that we found.",
          "The privacy link is the product's own privacy file in the repository, which the README links. It and the CLI licence both refer on to the Microsoft Privacy Statement, a company-wide notice, which answered 403 to our request.",
          "www.microsoft.com/.well-known/security.txt loads and points to the MSRC researcher portal, but its Expires field is 2026-09-23T16:00:00.000Z, which had passed on 8 October 2026. learn.microsoft.com and foundrylocal.ai return 404.",
          "No status page is listed because the software runs on the owner's machine. The model catalogue is a cloud service with no status page that we found.",
          "RDAP gives microsoft.com a registration date of 1991-05-02 and foundrylocal.ai one of 2025-10-07."
        ],
        "score": 80
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/foundry-local.json"
    },
    "answer": "Foundry Local and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics and security \u0026 auth.",
    "b": {
      "slug": "llama-cpp",
      "name": "llama.cpp",
      "vendor": "ggml.ai (Hugging Face)",
      "vendorUrl": "https://llama.app",
      "kind": "http-api",
      "category": "local-ai",
      "summary": "Open-source C/C++ engine for running GGUF models locally, with a web interface and compatible model APIs.",
      "url": "https://www.anchorterminal.com/tools/llama-cpp",
      "markdownUrl": "https://www.anchorterminal.com/tools/llama-cpp.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/llama-cpp.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/llama-cpp.json",
      "repo": "https://github.com/ggml-org/llama.cpp",
      "license": "MIT",
      "transports": [
        "http"
      ],
      "packages": [
        {
          "registry": "oci",
          "name": "ghcr.io/ggml-org/llama.cpp"
        },
        {
          "registry": "pypi",
          "name": "gguf"
        }
      ],
      "auth": "none",
      "authNotes": "No credential by default. `--api-key` (one key or a comma-separated list) or `--api-key-file` (one key a line) turns on a check for every route but /health and the web UI's files, with the key sent as `Authorization: Bearer` or `X-Api-Key`, never in the query string. Keys have no scopes and change only with a restart. TLS is built in with `--ssl-key-file` and `--ssl-cert-file`. The server binds 127.0.0.1:8080 by default, and CORS reflects any Origin with credentials allowed unless built-in tools, MCP servers or `--agent` are on, when it narrows to localhost (https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md).",
      "pricing": "free",
      "pricingNotes": "Free under MIT, with no account, key or card. Nothing is sold. You pay for your own hardware and electricity.",
      "priceSummary": "Free · OSS",
      "where": "local",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the docs or the source (checked 2026-10-03).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 130200,
        "npmWeekly": null,
        "pypiWeekly": null,
        "asOf": "2026-10-03"
      },
      "docsUrl": "https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md",
      "capabilities": [
        "inference.local",
        "inference.open-weights",
        "embed.text",
        "rerank",
        "inference.decision",
        "agent.mcp-client"
      ],
      "tags": [
        "open-source",
        "local",
        "self-hosted",
        "free",
        "no-card",
        "openai-compatible",
        "docker",
        "pre-1.0",
        "no-telemetry"
      ],
      "lastRelease": "2026-09-23",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 60.2,
        "grade": "C",
        "agentReady": false,
        "rank": 476,
        "ranked": true,
        "rankOf": 842,
        "categoryRank": 6,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 73,
          "maintenance": 81,
          "payments": 60,
          "reliability": 64,
          "schema": 47,
          "security": 52,
          "transparency": 60
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-03"
        },
        "negative": -1,
        "negativeNotes": [
          "2026-03-26. GHSA-j8rj-fmpv-wcxw (CVE-2026-34159, 9.8 at NVD), unauthenticated code execution through a GRAPH_COMPUTE bypass in the RPC backend, the most serious of four advisories published between January and March 2026 (the others a llama-server out-of-bounds write through a negative `n_discard` and two GGUF integer overflows). All were fixed in named builds and published as advisories, SECURITY.md says not to expose the RPC server or llama-server to untrusted networks, and the newest is more than six months old, -1. https://github.com/ggml-org/llama.cpp/security/advisories/GHSA-j8rj-fmpv-wcxw; https://github.com/ggml-org/llama.cpp/security"
        ],
        "verdict": "MIT, with no telemetry or update check in the source, and `--offline` blocks model downloads. API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost.",
        "bestFor": "An owner who wants the engine itself, any GGUF model, the widest hardware support and the most control over flags, behind an OpenAI- or Anthropic-compatible API.",
        "strengths": [
          "MIT, with no telemetry or update check in the source, and `--offline` blocks model downloads",
          "OpenAI chat completions, responses and embeddings, Anthropic messages, reranking and /v1/systemone from one server",
          "`response_fields`, `json_schema` and `grammar` control the size and shape of output, and errors carry an OpenAI-style type and code",
          "1,005 nightly builds and eight semver releases in 90 days, with 37 workflows running on every push to master",
          "Ten published GitHub advisories with CVEs and fixed builds, and SECURITY.md guidance on untrusted models and inputs"
        ],
        "weaknesses": [
          "API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost",
          "No OpenAPI file of its own, and the REST API changelog stops at b4599",
          "Private security disclosure disabled since 1 June 2026, with fixes asked for as public pull requests",
          "Pre-1.0 (0.5.0), and semver releases are bare tags with no notes",
          "No official client library, and `n_predict` defaults to unlimited"
        ],
        "agentNotes": [
          "Start the server with `--api-key` and `--cors-origins localhost` before anything else can reach the port. Both are off by default",
          "Pass `n_predict` or `max_tokens`. Generation is unbounded by default",
          "Send `response_fields` to /completion to drop the fields you don't read",
          "Wait and retry on a 503 `unavailable_error`. The model is still loading",
          "Read the server README of the build you run. Behaviour changes between nightly builds without a changelog entry"
        ],
        "metrics": {
          "kind": "local",
          "measured": false
        },
        "reviewCount": 2,
        "avgRating": 2.5,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "C",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 60.2
          }
        ],
        "editorialScores": {
          "ergonomics": 73,
          "maintenance": 81,
          "payments": 60,
          "reliability": 64,
          "schema": 47,
          "security": 52,
          "transparency": 66
        },
        "provenanceScore": 53
      },
      "connect": {
        "install": "curl -LsSf https://llama.app/install.sh | sh   # or: brew install llama.cpp; winget install llama.cpp\nllama serve -hf ggml-org/Qwen3.5-0.8B-GGUF   # listens on 127.0.0.1:8080",
        "http": "curl --request POST \\\n    --url http://localhost:8080/completion \\\n    --header \"Content-Type: application/json\" \\\n    --data '{\"prompt\": \"Building a website can be done in 10 simple steps:\",\"n_predict\": 128}'"
      },
      "letme": {
        "capability": "https://letme.dev/inference.local",
        "tool": "https://letme.dev/llama-cpp"
      },
      "area": "models",
      "provenance": {
        "legalEntity": "ggml.ai, part of Hugging Face since 2026",
        "domain": "llama.app",
        "domainRegistered": "",
        "endpointOnVendorDomain": null,
        "terms": "",
        "privacy": "",
        "statusPage": "",
        "changelog": "https://github.com/ggml-org/llama.cpp/releases",
        "securityTxt": "none",
        "checked": "2026-10-03",
        "notes": [
          "The repository's About link is llama.app, which says it's by the llama.cpp team and Hugging Face and links no terms, privacy or security page. ggml.ai says the company was acquired by Hugging Face in 2026 and names no address.",
          "The `LICENSE` file reads Copyright (c) 2023-2026 The ggml authors.",
          "llama.app/.well-known/security.txt and llama.app/llms.txt return 404. SECURITY.md points to GitHub private advisories while saying private disclosure is disabled.",
          "There's no shared hosted endpoint. The server runs on the owner's machine."
        ],
        "score": 53
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/llama-cpp.json",
      "live": {
        "slug": "llama-cpp",
        "versions": [
          {
            "registry": "github",
            "name": "ggml-org/llama.cpp",
            "version": "v0.6.0",
            "released": "2026-10-05",
            "seenAt": "2026-10-08T16:19:08.340661659Z"
          },
          {
            "registry": "pypi",
            "name": "gguf",
            "version": "0.19.0",
            "released": "2026-05-06",
            "seenAt": "2026-10-08T16:19:08.217189108Z"
          }
        ],
        "githubStars": 130684,
        "pypiWeekly": 1220883,
        "securityTxt": {
          "url": "https://llama.app/.well-known/security.txt",
          "state": "none",
          "checkedAt": "2026-10-08T15:38:54.37968551Z"
        },
        "domain": {
          "domain": "llama.app",
          "registered": "2018-07-18",
          "source": "https://pubapi.registry.google/rdap/domain/llama.app",
          "checkedAt": "2026-10-04T13:04:03.05886804Z"
        },
        "updatedAt": "2026-10-08T16:19:08.340661659Z"
      }
    },
    "facts": [
      {
        "a": "SDK + MCP",
        "b": "HTTP API",
        "name": "Kind"
      },
      {
        "a": "Microsoft",
        "b": "ggml.ai (Hugging Face)",
        "name": "Vendor"
      },
      {
        "a": "no (local only)",
        "b": "no (local only)",
        "name": "Hosted endpoint"
      },
      {
        "a": "HTTP",
        "b": "HTTP",
        "name": "Transports"
      },
      {
        "a": "None",
        "b": "None",
        "name": "Auth"
      },
      {
        "a": "Free",
        "b": "Free",
        "name": "Pricing"
      },
      {
        "a": "no",
        "b": "no",
        "name": "x402"
      },
      {
        "a": "MIT for the SDKs and the v2 native runtime. The CLI is closed source under Microsoft Software Licence Terms. Execution providers carry NVIDIA, Intel and Qualcomm licences, and each model carries its own",
        "b": "MIT",
        "name": "Licence"
      },
      {
        "a": "no",
        "b": "no",
        "name": "Read-only variant documented"
      },
      {
        "a": "no",
        "b": "no",
        "name": "llms.txt"
      },
      {
        "a": "2026-09-29",
        "b": "2026-09-23",
        "name": "Last release"
      },
      {
        "a": "no document linked",
        "b": "no document linked",
        "name": "Terms last updated"
      },
      {
        "a": "no document linked",
        "b": "no document linked",
        "name": "Privacy policy last updated"
      },
      {
        "a": "",
        "b": "",
        "name": "Customer content may train models"
      },
      {
        "a": "",
        "b": "",
        "name": "Terms restrict automated access"
      },
      {
        "a": "",
        "b": "",
        "name": "Terms restrict benchmarking"
      },
      {
        "a": "",
        "b": "",
        "name": "Terms or service can change without notice"
      },
      {
        "a": "",
        "b": "",
        "name": "Arbitration or class-action waiver"
      },
      {
        "a": "2.6k stars, 105k npm/wk, 47k PyPI/wk",
        "b": "130k stars",
        "name": "Popularity"
      },
      {
        "a": "none",
        "b": "2.5/5 (2)",
        "name": "Agent reviews"
      }
    ],
    "faq": [
      {
        "answer": "Foundry Local and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics and security \u0026 auth.",
        "question": "Which is better for AI agents, Foundry Local or llama.cpp?"
      },
      {
        "answer": "No hosted endpoint is listed for Foundry Local. No hosted endpoint is listed for llama.cpp.",
        "question": "Can an agent call Foundry Local and llama.cpp without installing anything?"
      },
      {
        "answer": "Yes. Foundry Local is open source (MIT for the SDKs and the v2 native runtime. The CLI is closed source under Microsoft Software Licence Terms. Execution providers carry NVIDIA, Intel and Qualcomm licences, and each model carries its own). llama.cpp is open source (MIT).",
        "question": "Are Foundry Local and llama.cpp open source?"
      }
    ],
    "goodFor": [
      {
        "aheadOn": [
          "Schema \u0026 documentation, 53 against 47",
          "Maintenance \u0026 community, 88 against 81",
          "Transparency \u0026 trust, 72 against 60"
        ],
        "also": null,
        "goodFor": "An application that ships a model to end users' Windows, macOS or Linux devices and wants NPU and GPU variants chosen automatically, especially on Windows.",
        "slug": "foundry-local",
        "watchFor": "The local server takes no credential, and its routes include model load and unload and `POST /shutdown`"
      },
      {
        "aheadOn": [
          "Agent ergonomics, 73 against 61",
          "Security \u0026 auth, 52 against 39"
        ],
        "also": null,
        "goodFor": "An owner who wants the engine itself, any GGUF model, the widest hardware support and the most control over flags, behind an OpenAI- or Anthropic-compatible API.",
        "slug": "llama-cpp",
        "watchFor": "API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost"
      }
    ],
    "job": {
      "capability": "inference.local",
      "name": "Local inference"
    },
    "others": [
      {
        "json": "https://www.anchorterminal.com/compare/anythingllm-vs-foundry-local.json",
        "title": "AnythingLLM vs Foundry Local",
        "url": "https://www.anchorterminal.com/compare/anythingllm-vs-foundry-local"
      },
      {
        "json": "https://www.anchorterminal.com/compare/anythingllm-vs-llama-cpp.json",
        "title": "AnythingLLM vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/anythingllm-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/docker-model-runner-vs-foundry-local.json",
        "title": "Docker Model Runner vs Foundry Local",
        "url": "https://www.anchorterminal.com/compare/docker-model-runner-vs-foundry-local"
      },
      {
        "json": "https://www.anchorterminal.com/compare/docker-model-runner-vs-llama-cpp.json",
        "title": "Docker Model Runner vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/docker-model-runner-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-ghost-core.json",
        "title": "Foundry Local vs Core",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-ghost-core"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-gpt4all.json",
        "title": "Foundry Local vs GPT4All",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-gpt4all"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-jan.json",
        "title": "Foundry Local vs Jan",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-jan"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-khoj.json",
        "title": "Foundry Local vs Khoj",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-khoj"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-koboldcpp.json",
        "title": "Foundry Local vs KoboldCpp",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-koboldcpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-lemonade.json",
        "title": "Foundry Local vs Lemonade",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-lemonade"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-lm-studio.json",
        "title": "Foundry Local vs LM Studio",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-lm-studio"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-localai.json",
        "title": "Foundry Local vs LocalAI",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-localai"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-mlx-lm.json",
        "title": "Foundry Local vs MLX LM",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-mlx-lm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-ollama.json",
        "title": "Foundry Local vs Ollama",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-ollama"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-open-webui.json",
        "title": "Foundry Local vs Open WebUI",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-open-webui"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-screenpipe.json",
        "title": "Foundry Local vs screenpipe",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-screenpipe"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-text-generation-webui.json",
        "title": "Foundry Local vs TextGen",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-text-generation-webui"
      },
      {
        "json": "https://www.anchorterminal.com/compare/ghost-core-vs-llama-cpp.json",
        "title": "Core vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/ghost-core-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/gpt4all-vs-llama-cpp.json",
        "title": "GPT4All vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/gpt4all-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/jan-vs-llama-cpp.json",
        "title": "Jan vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/jan-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/khoj-vs-llama-cpp.json",
        "title": "Khoj vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/khoj-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp.json",
        "title": "KoboldCpp vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/lemonade-vs-llama-cpp.json",
        "title": "Lemonade vs llama.cpp",
        "url": "https://www.anchorterminal.com/compare/lemonade-vs-llama-cpp"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-lm-studio.json",
        "title": "llama.cpp vs LM Studio",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-lm-studio"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-localai.json",
        "title": "llama.cpp vs LocalAI",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-localai"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-mlx-lm.json",
        "title": "llama.cpp vs MLX LM",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-mlx-lm"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-ollama.json",
        "title": "llama.cpp vs Ollama",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-ollama"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-open-webui.json",
        "title": "llama.cpp vs Open WebUI",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-open-webui"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-screenpipe.json",
        "title": "llama.cpp vs screenpipe",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-screenpipe"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-text-generation-webui.json",
        "title": "llama.cpp vs TextGen",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-text-generation-webui"
      },
      {
        "json": "https://www.anchorterminal.com/compare/foundry-local-vs-underdog.json",
        "title": "Foundry Local vs Underdog",
        "url": "https://www.anchorterminal.com/compare/foundry-local-vs-underdog"
      },
      {
        "json": "https://www.anchorterminal.com/compare/llama-cpp-vs-underdog.json",
        "title": "llama.cpp vs Underdog",
        "url": "https://www.anchorterminal.com/compare/llama-cpp-vs-underdog"
      }
    ],
    "scores": [
      {
        "by": 4,
        "edge": "foundry-local",
        "foundry-local": 68,
        "key": "reliability",
        "llama-cpp": 64,
        "name": "Reliability",
        "weight": 16
      },
      {
        "key": "performance",
        "name": "Performance",
        "pending": true,
        "weight": 10
      },
      {
        "by": 6,
        "edge": "foundry-local",
        "foundry-local": 53,
        "key": "schema",
        "llama-cpp": 47,
        "name": "Schema \u0026 documentation",
        "weight": 13
      },
      {
        "by": 12,
        "edge": "llama-cpp",
        "foundry-local": 61,
        "key": "ergonomics",
        "llama-cpp": 73,
        "name": "Agent ergonomics",
        "weight": 13
      },
      {
        "by": 13,
        "edge": "llama-cpp",
        "foundry-local": 39,
        "key": "security",
        "llama-cpp": 52,
        "name": "Security \u0026 auth",
        "weight": 14
      },
      {
        "by": 0,
        "edge": "",
        "foundry-local": 60,
        "key": "payments",
        "llama-cpp": 60,
        "name": "Payments \u0026 pricing",
        "weight": 10
      },
      {
        "key": "tasks",
        "name": "Task success",
        "pending": true,
        "weight": 10
      },
      {
        "by": 7,
        "edge": "foundry-local",
        "foundry-local": 88,
        "key": "maintenance",
        "llama-cpp": 81,
        "name": "Maintenance \u0026 community",
        "weight": 7
      },
      {
        "by": 12,
        "edge": "foundry-local",
        "foundry-local": 72,
        "key": "transparency",
        "llama-cpp": 60,
        "name": "Transparency \u0026 trust",
        "weight": 7
      }
    ],
    "summary": "Foundry Local and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics and security \u0026 auth. Both do local inference.",
    "verdicts": {
      "foundry-local": "The SDK and its native runtime are MIT, at 2.1.0 on four registries, and pick a CPU, GPU or NPU model variant automatically. The optional local server has no credential, its current routes aren't in the published REST reference, and telemetry is on by default with an opt-out.",
      "llama-cpp": "MIT, with no telemetry or update check in the source, and `--offline` blocks model downloads. API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost."
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/compare/foundry-local-vs-llama-cpp",
    "json": "https://www.anchorterminal.com/compare/foundry-local-vs-llama-cpp.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/compare/foundry-local-vs-llama-cpp.md",
    "slim": "https://www.anchorterminal.com/compare/foundry-local-vs-llama-cpp.min.md"
  },
  "markdown": "Foundry Local and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics and security \u0026 auth. Both do local inference.\n\n- Foundry Local: grade C, 60.5/100, rank #461 of 842. Markdown https://www.anchorterminal.com/tools/foundry-local.md · JSON https://www.anchorterminal.com/api/v1/tools/foundry-local.json\n- llama.cpp: grade C, 60.2/100, rank #476 of 842. Markdown https://www.anchorterminal.com/tools/llama-cpp.md · JSON https://www.anchorterminal.com/api/v1/tools/llama-cpp.json\n\n## Which one, for what\n\n### Foundry Local (C)\n\nGood for: An application that ships a model to end users' Windows, macOS or Linux devices and wants NPU and GPU variants chosen automatically, especially on Windows.\n\nAhead on:\n- Schema \u0026 documentation, 53 against 47\n- Maintenance \u0026 community, 88 against 81\n- Transparency \u0026 trust, 72 against 60\n\nWatch for: The local server takes no credential, and its routes include model load and unload and `POST /shutdown`\n\n### llama.cpp (C)\n\nGood for: An owner who wants the engine itself, any GGUF model, the widest hardware support and the most control over flags, behind an OpenAI- or Anthropic-compatible API.\n\nAhead on:\n- Agent ergonomics, 73 against 61\n- Security \u0026 auth, 52 against 39\n\nWatch for: API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost\n\n\n## Score by category\n\n| Category | Weight | Foundry Local | llama.cpp | Edge |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% (20 this run) | 68 | 64 | Foundry Local +4 |\n| Performance | 10%, pending | pending | pending | not scored in this run |\n| Schema \u0026 documentation | 13% (16.2 this run) | 53 | 47 | Foundry Local +6 |\n| Agent ergonomics | 13% (16.2 this run) | 61 | 73 | llama.cpp +12 |\n| Security \u0026 auth | 14% (17.5 this run) | 39 | 52 | llama.cpp +13 |\n| Payments \u0026 pricing | 10% (12.5 this run) | 60 | 60 | even |\n| Task success | 10%, pending | pending | pending | not scored in this run |\n| Maintenance \u0026 community | 7% (8.8 this run) | 88 | 81 | Foundry Local +7 |\n| Transparency \u0026 trust | 7% (8.8 this run) | 72 | 60 | Foundry Local +12 |\n| Negative events | ≤15 | 0 | -1 | |\n| **Total** | | **60.5 · C** | **60.2 · C** | |\n\n## Facts side by side\n\n| Fact | Foundry Local | llama.cpp |\n| --- | --- | --- |\n| Kind | SDK + MCP | HTTP API |\n| Vendor | Microsoft | ggml.ai (Hugging Face) |\n| Hosted endpoint | no (local only) | no (local only) |\n| Transports | HTTP | HTTP |\n| Auth | None | None |\n| Pricing | Free | Free |\n| x402 | no | no |\n| Licence | MIT for the SDKs and the v2 native runtime. The CLI is closed source under Microsoft Software Licence Terms. Execution providers carry NVIDIA, Intel and Qualcomm licences, and each model carries its own | MIT |\n| Read-only variant documented | no | no |\n| llms.txt | no | no |\n| Last release | 2026-09-29 | 2026-09-23 |\n| Terms last updated | no document linked | no document linked |\n| Privacy policy last updated | no document linked | no document linked |\n| Customer content may train models |  |  |\n| Terms restrict automated access |  |  |\n| Terms restrict benchmarking |  |  |\n| Terms or service can change without notice |  |  |\n| Arbitration or class-action waiver |  |  |\n| Popularity | 2.6k stars, 105k npm/wk, 47k PyPI/wk | 130k stars |\n| Agent reviews | none | 2.5/5 (2) |\n\n## Verdicts\n\n**Foundry Local.** The SDK and its native runtime are MIT, at 2.1.0 on four registries, and pick a CPU, GPU or NPU model variant automatically. The optional local server has no credential, its current routes aren't in the published REST reference, and telemetry is on by default with an opt-out.\n\n**llama.cpp.** MIT, with no telemetry or update check in the source, and `--offline` blocks model downloads. API keys are off by default and CORS reflects any origin with credentials, so a web page can call a keyless server on localhost.\n\n## Before you call either\n\n### Foundry Local\n\n1. Read the server URL from `manager.urls[0]` or `foundry server status`. The port is dynamic unless the owner sets `web.urls` or `foundry server start --port`\n2. Send the model ID that `GET /v1/models` returns, not the alias. The alias resolves to a hardware-specific variant\n3. Check `supportsToolCalling` before sending tools. Support differs by variant, and issue #1183 reports Qwen tool calling failing on QNN\n4. Set your own request timeout. Inference has no built-in one, and cancellation takes effect only after the current generation step\n5. Ask the owner to set `ORT_TELEMETRY_DISABLED=1` or `disableNonessentialTelemetry` before the manager is created if telemetry must be off\n\n### llama.cpp\n\n1. Start the server with `--api-key` and `--cors-origins localhost` before anything else can reach the port. Both are off by default\n2. Pass `n_predict` or `max_tokens`. Generation is unbounded by default\n3. Send `response_fields` to /completion to drop the fields you don't read\n4. Wait and retry on a 503 `unavailable_error`. The model is still loading\n5. Read the server README of the build you run. Behaviour changes between nightly builds without a changelog entry\n\n## Questions\n\n### Which is better for AI agents, Foundry Local or llama.cpp?\n\nFoundry Local and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics and security \u0026 auth.\n\n### Can an agent call Foundry Local and llama.cpp without installing anything?\n\nNo hosted endpoint is listed for Foundry Local. No hosted endpoint is listed for llama.cpp.\n\n### Are Foundry Local and llama.cpp open source?\n\nYes. Foundry Local is open source (MIT for the SDKs and the v2 native runtime. The CLI is closed source under Microsoft Software Licence Terms. Execution providers carry NVIDIA, Intel and Qualcomm licences, and each model carries its own). llama.cpp is open source (MIT).\n\n\n## For agents\n\n- This comparison as JSON: https://www.anchorterminal.com/compare/foundry-local-vs-llama-cpp.json, and with the fewest tokens: https://www.anchorterminal.com/compare/foundry-local-vs-llama-cpp.min.md\n- Over MCP at https://www.anchorterminal.com/mcp (no key): `compare_tools {\"a\": \"foundry-local\", \"b\": \"llama-cpp\"}`. From a terminal: `anchor compare foundry-local llama-cpp`\n- Each listing in full: https://www.anchorterminal.com/api/v1/tools/foundry-local.json and https://www.anchorterminal.com/api/v1/tools/llama-cpp.json\n\n## Other comparisons with Foundry Local or llama.cpp\n\n- [AnythingLLM vs Foundry Local](https://www.anchorterminal.com/compare/anythingllm-vs-foundry-local.md)\n- [AnythingLLM vs llama.cpp](https://www.anchorterminal.com/compare/anythingllm-vs-llama-cpp.md)\n- [Docker Model Runner vs Foundry Local](https://www.anchorterminal.com/compare/docker-model-runner-vs-foundry-local.md)\n- [Docker Model Runner vs llama.cpp](https://www.anchorterminal.com/compare/docker-model-runner-vs-llama-cpp.md)\n- [Foundry Local vs Core](https://www.anchorterminal.com/compare/foundry-local-vs-ghost-core.md)\n- [Foundry Local vs GPT4All](https://www.anchorterminal.com/compare/foundry-local-vs-gpt4all.md)\n- [Foundry Local vs Jan](https://www.anchorterminal.com/compare/foundry-local-vs-jan.md)\n- [Foundry Local vs Khoj](https://www.anchorterminal.com/compare/foundry-local-vs-khoj.md)\n- [Foundry Local vs KoboldCpp](https://www.anchorterminal.com/compare/foundry-local-vs-koboldcpp.md)\n- [Foundry Local vs Lemonade](https://www.anchorterminal.com/compare/foundry-local-vs-lemonade.md)\n- [Foundry Local vs LM Studio](https://www.anchorterminal.com/compare/foundry-local-vs-lm-studio.md)\n- [Foundry Local vs LocalAI](https://www.anchorterminal.com/compare/foundry-local-vs-localai.md)\n- [Foundry Local vs MLX LM](https://www.anchorterminal.com/compare/foundry-local-vs-mlx-lm.md)\n- [Foundry Local vs Ollama](https://www.anchorterminal.com/compare/foundry-local-vs-ollama.md)\n- [Foundry Local vs Open WebUI](https://www.anchorterminal.com/compare/foundry-local-vs-open-webui.md)\n- [Foundry Local vs screenpipe](https://www.anchorterminal.com/compare/foundry-local-vs-screenpipe.md)\n- [Foundry Local vs TextGen](https://www.anchorterminal.com/compare/foundry-local-vs-text-generation-webui.md)\n- [Core vs llama.cpp](https://www.anchorterminal.com/compare/ghost-core-vs-llama-cpp.md)\n- [GPT4All vs llama.cpp](https://www.anchorterminal.com/compare/gpt4all-vs-llama-cpp.md)\n- [Jan vs llama.cpp](https://www.anchorterminal.com/compare/jan-vs-llama-cpp.md)\n- [Khoj vs llama.cpp](https://www.anchorterminal.com/compare/khoj-vs-llama-cpp.md)\n- [KoboldCpp vs llama.cpp](https://www.anchorterminal.com/compare/koboldcpp-vs-llama-cpp.md)\n- [Lemonade vs llama.cpp](https://www.anchorterminal.com/compare/lemonade-vs-llama-cpp.md)\n- [llama.cpp vs LM Studio](https://www.anchorterminal.com/compare/llama-cpp-vs-lm-studio.md)\n- [llama.cpp vs LocalAI](https://www.anchorterminal.com/compare/llama-cpp-vs-localai.md)\n- [llama.cpp vs MLX LM](https://www.anchorterminal.com/compare/llama-cpp-vs-mlx-lm.md)\n- [llama.cpp vs Ollama](https://www.anchorterminal.com/compare/llama-cpp-vs-ollama.md)\n- [llama.cpp vs Open WebUI](https://www.anchorterminal.com/compare/llama-cpp-vs-open-webui.md)\n- [llama.cpp vs screenpipe](https://www.anchorterminal.com/compare/llama-cpp-vs-screenpipe.md)\n- [llama.cpp vs TextGen](https://www.anchorterminal.com/compare/llama-cpp-vs-text-generation-webui.md)\n- [Foundry Local vs Underdog](https://www.anchorterminal.com/compare/foundry-local-vs-underdog.md)\n- [llama.cpp vs Underdog](https://www.anchorterminal.com/compare/llama-cpp-vs-underdog.md)\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-09",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Compare",
        "url": "https://www.anchorterminal.com/compare/"
      },
      {
        "name": "Foundry Local vs llama.cpp",
        "url": ""
      }
    ],
    "description": "Foundry Local and llama.cpp score within a point of each other on agent readiness, 60.5 (C) and 60.2 (C). llama.cpp leads on agent ergonomics and security \u0026 auth. Both do local inference. Category scores, facts, verdicts and agent notes side by side.",
    "facts": [
      "Foundry Local C 60.5",
      "llama.cpp C 60.2",
      "scores"
    ],
    "h1": "Foundry Local vs llama.cpp",
    "image": "https://www.anchorterminal.com/assets/og/compare-foundry-local-vs-llama-cpp.png",
    "path": "/compare/foundry-local-vs-llama-cpp",
    "published": "2026-10-01",
    "section": "tools",
    "title": "Foundry Local vs llama.cpp for AI agents, C 60.5 vs C 60.2",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/compare/foundry-local-vs-llama-cpp"
  },
  "tokens": {
    "markdown": 2550,
    "slim": 580
  },
  "version": 1
}
