{
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-05",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.3",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "tool": {
    "category": "gpu-compute",
    "endpoint": "https://fitllm.run/api/mcp",
    "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/fitllm.json",
    "kind": "mcp",
    "listed": "indexed",
    "liveUrl": "https://www.anchorterminal.com/api/v1/live/fitllm.json",
    "markdownUrl": "https://www.anchorterminal.com/tools/fitllm.md",
    "mcpTools": {
      "check": {
        "checker": "anchor-check/1.0",
        "totalTokens": 957,
        "counts": {
          "error": 0,
          "note": 1,
          "warn": 0
        },
        "findings": [
          {
            "rule": "TC24",
            "severity": "note",
            "message": "3 of 3 tools have no outputSchema",
            "fix": "Declare outputSchema for tools that return structured data, and return structuredContent that matches it."
          }
        ]
      },
      "checkedAt": "2026-10-04T22:25:52Z",
      "count": 3,
      "schemaTokens": 957,
      "status": "ok",
      "tools": [
        {
          "name": "check_llm_fit",
          "title": "Check if an LLM fits on hardware",
          "description": "Check whether a specific local LLM fits in the memory of a specific GPU or Apple Silicon Mac. Returns fits/tight/won't-fit verdict with the memory breakdown (weights, KV cache, linear-attention state when present, runtime overhead, reserve), max context, and a concrete fix if it doesn't fit. Use this whenever a user asks anything like \"can I run \u003cmodel\u003e on my \u003cGPU/Mac\u003e?\", \"will \u003cmodel\u003e fit in \u003cN\u003eGB?\", or \"what do I need to run \u003cmodel\u003e?\". Estimates using curated, config-derived architecture fields (MLA, sliding-window, hybrid attention, MoE modeled).",
          "inputSchema": {
            "$schema": "http://json-schema.org/draft-07/schema#",
            "additionalProperties": false,
            "properties": {
              "context_tokens": {
                "description": "Context length in tokens (default 8192). Alias: ctx (same field as the REST API).",
                "minimum": 1024,
                "type": "integer"
              },
              "ctx": {
                "description": "Alias of context_tokens — accepted because the REST API uses this name. Do not pass both with different values.",
                "minimum": 1024,
                "type": "integer"
              },
              "gpu": {
                "description": "GPU name, fuzzy — e.g. \"RTX 4090\", \"RX 7900 XTX\", \"A100 80GB\". Multi-GPU rigs: join with + — e.g. \"RTX 5090 + RTX 3090\" (VRAM pools across cards). Provide gpu OR mac_ram_gb.",
                "type": "string"
              },
              "gpu_count": {
                "description": "Number of identical copies of the gpu (e.g. gpu=\"RTX 3090\", gpu_count=2 for a 2×3090 rig). Default 1.",
                "maximum": 8,
                "minimum": 1,
                "type": "integer"
              },
              "kv_bits": {
                "description": "KV-cache quantization bits (default 16 = F16)",
                "enum": [
                  16,
                  8,
                  4
                ],
                "type": "number"
              },
              "mac_ram_gb": {
                "description": "Apple Silicon unified memory in GB — e.g. 16, 64, 512. Provide gpu OR mac_ram_gb.",
                "maximum": 2048,
                "minimum": 8,
                "type": "integer"
              },
              "model": {
                "description": "LLM name, fuzzy — e.g. \"GLM-4.7-Flash\", \"gpt-oss-20b\", \"gemma 31b\"",
                "type": "string"
              },
              "quant": {
                "description": "Weight quantization. GPU: Q4_K_M(default)/Q5_K_M/Q6_K/Q8_0/FP16. Mac: 4/8(default)/16 (bits).",
                "type": "string"
              }
            },
            "required": [
              "model"
            ],
            "type": "object"
          },
          "annotations": {
            "destructiveHint": false,
            "idempotentHint": true,
            "openWorldHint": false,
            "readOnlyHint": true
          }
        },
        {
          "name": "what_fits_on_hardware",
          "title": "What LLMs fit on this hardware",
          "description": "Rank which popular local LLMs fit on a given GPU or Apple Silicon Mac (at ~4-bit quantization, 8K context) — models that fit come first, biggest first, with max context each. Use when a user asks \"what can I run on my \u003cGPU/Mac/N GB\u003e?\", \"best local model for my machine?\", or gives hardware without naming a model.",
          "inputSchema": {
            "$schema": "http://json-schema.org/draft-07/schema#",
            "additionalProperties": false,
            "properties": {
              "gpu": {
                "description": "GPU name, fuzzy. Multi-GPU rigs: join with + (e.g. \"RTX 5090 + RTX 3090\"). Provide gpu OR mac_ram_gb.",
                "type": "string"
              },
              "gpu_count": {
                "description": "Number of identical copies of the gpu. Default 1.",
                "maximum": 8,
                "minimum": 1,
                "type": "integer"
              },
              "mac_ram_gb": {
                "description": "Apple Silicon unified memory GB. Provide gpu OR mac_ram_gb.",
                "maximum": 2048,
                "minimum": 8,
                "type": "integer"
              }
            },
            "type": "object"
          },
          "annotations": {
            "destructiveHint": false,
            "idempotentHint": true,
            "openWorldHint": false,
            "readOnlyHint": true
          }
        },
        {
          "name": "list_supported",
          "title": "List supported models \u0026 hardware",
          "description": "List the built-in model names and hardware names this fit-checker knows (for mapping user wording to exact names). Standard text-only HuggingFace transformer configs can also be checked via fitllm.run; unsupported architectures are rejected.",
          "inputSchema": {
            "$schema": "http://json-schema.org/draft-07/schema#",
            "properties": {},
            "type": "object"
          },
          "annotations": {
            "destructiveHint": false,
            "idempotentHint": true,
            "openWorldHint": false,
            "readOnlyHint": true
          }
        }
      ]
    },
    "name": "FitLLM",
    "note": "Indexed from the official MCP registry: facts and our own checks, not reviewed, so no score, grade or rank.",
    "packages": null,
    "pageJsonUrl": "https://www.anchorterminal.com/tools/fitllm.json",
    "popularity": {
      "githubStars": 8
    },
    "registryName": "run.fitllm/fitllm",
    "remotes": [
      {
        "type": "streamable-http",
        "url": "https://fitllm.run/api/mcp"
      }
    ],
    "repository": "https://github.com/click6067-ship-it/fitllm-engine",
    "reviewed": false,
    "slug": "fitllm",
    "source": "the official MCP registry",
    "sourceUrl": "https://registry.modelcontextprotocol.io/v0.1/servers?search=run.fitllm/fitllm",
    "summary": "Will this LLM fit on your GPU, multi-GPU rig or Mac? Exact VRAM \u0026 KV-cache math. Read-only.",
    "updatedAt": "2026-09-02T17:35:00Z",
    "url": "https://www.anchorterminal.com/tools/fitllm",
    "vendor": "fitllm.run",
    "vendorUrl": "https://fitllm.run",
    "version": "1.1.0",
    "websiteUrl": "https://fitllm.run",
    "where": "hosted",
    "why": [
      "vendor"
    ]
  }
}
