{
  "data": {
    "tool": {
      "category": "gpu-compute",
      "endpoint": "https://fitllm.run/api/mcp",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/fitllm.json",
      "kind": "mcp",
      "listed": "indexed",
      "liveUrl": "https://www.anchorterminal.com/api/v1/live/fitllm.json",
      "markdownUrl": "https://www.anchorterminal.com/tools/fitllm.md",
      "mcpTools": {
        "check": {
          "checker": "anchor-check/1.0",
          "totalTokens": 957,
          "counts": {
            "error": 0,
            "note": 1,
            "warn": 0
          },
          "findings": [
            {
              "rule": "TC24",
              "severity": "note",
              "message": "3 of 3 tools have no outputSchema",
              "fix": "Declare outputSchema for tools that return structured data, and return structuredContent that matches it."
            }
          ]
        },
        "checkedAt": "2026-10-04T22:25:52Z",
        "count": 3,
        "schemaTokens": 957,
        "status": "ok",
        "tools": [
          {
            "name": "check_llm_fit",
            "title": "Check if an LLM fits on hardware",
            "description": "Check whether a specific local LLM fits in the memory of a specific GPU or Apple Silicon Mac. Returns fits/tight/won't-fit verdict with the memory breakdown (weights, KV cache, linear-attention state when present, runtime overhead, reserve), max context, and a concrete fix if it doesn't fit. Use this whenever a user asks anything like \"can I run \u003cmodel\u003e on my \u003cGPU/Mac\u003e?\", \"will \u003cmodel\u003e fit in \u003cN\u003eGB?\", or \"what do I need to run \u003cmodel\u003e?\". Estimates using curated, config-derived architecture fields (MLA, sliding-window, hybrid attention, MoE modeled).",
            "inputSchema": {
              "$schema": "http://json-schema.org/draft-07/schema#",
              "additionalProperties": false,
              "properties": {
                "context_tokens": {
                  "description": "Context length in tokens (default 8192). Alias: ctx (same field as the REST API).",
                  "minimum": 1024,
                  "type": "integer"
                },
                "ctx": {
                  "description": "Alias of context_tokens — accepted because the REST API uses this name. Do not pass both with different values.",
                  "minimum": 1024,
                  "type": "integer"
                },
                "gpu": {
                  "description": "GPU name, fuzzy — e.g. \"RTX 4090\", \"RX 7900 XTX\", \"A100 80GB\". Multi-GPU rigs: join with + — e.g. \"RTX 5090 + RTX 3090\" (VRAM pools across cards). Provide gpu OR mac_ram_gb.",
                  "type": "string"
                },
                "gpu_count": {
                  "description": "Number of identical copies of the gpu (e.g. gpu=\"RTX 3090\", gpu_count=2 for a 2×3090 rig). Default 1.",
                  "maximum": 8,
                  "minimum": 1,
                  "type": "integer"
                },
                "kv_bits": {
                  "description": "KV-cache quantization bits (default 16 = F16)",
                  "enum": [
                    16,
                    8,
                    4
                  ],
                  "type": "number"
                },
                "mac_ram_gb": {
                  "description": "Apple Silicon unified memory in GB — e.g. 16, 64, 512. Provide gpu OR mac_ram_gb.",
                  "maximum": 2048,
                  "minimum": 8,
                  "type": "integer"
                },
                "model": {
                  "description": "LLM name, fuzzy — e.g. \"GLM-4.7-Flash\", \"gpt-oss-20b\", \"gemma 31b\"",
                  "type": "string"
                },
                "quant": {
                  "description": "Weight quantization. GPU: Q4_K_M(default)/Q5_K_M/Q6_K/Q8_0/FP16. Mac: 4/8(default)/16 (bits).",
                  "type": "string"
                }
              },
              "required": [
                "model"
              ],
              "type": "object"
            },
            "annotations": {
              "destructiveHint": false,
              "idempotentHint": true,
              "openWorldHint": false,
              "readOnlyHint": true
            }
          },
          {
            "name": "what_fits_on_hardware",
            "title": "What LLMs fit on this hardware",
            "description": "Rank which popular local LLMs fit on a given GPU or Apple Silicon Mac (at ~4-bit quantization, 8K context) — models that fit come first, biggest first, with max context each. Use when a user asks \"what can I run on my \u003cGPU/Mac/N GB\u003e?\", \"best local model for my machine?\", or gives hardware without naming a model.",
            "inputSchema": {
              "$schema": "http://json-schema.org/draft-07/schema#",
              "additionalProperties": false,
              "properties": {
                "gpu": {
                  "description": "GPU name, fuzzy. Multi-GPU rigs: join with + (e.g. \"RTX 5090 + RTX 3090\"). Provide gpu OR mac_ram_gb.",
                  "type": "string"
                },
                "gpu_count": {
                  "description": "Number of identical copies of the gpu. Default 1.",
                  "maximum": 8,
                  "minimum": 1,
                  "type": "integer"
                },
                "mac_ram_gb": {
                  "description": "Apple Silicon unified memory GB. Provide gpu OR mac_ram_gb.",
                  "maximum": 2048,
                  "minimum": 8,
                  "type": "integer"
                }
              },
              "type": "object"
            },
            "annotations": {
              "destructiveHint": false,
              "idempotentHint": true,
              "openWorldHint": false,
              "readOnlyHint": true
            }
          },
          {
            "name": "list_supported",
            "title": "List supported models \u0026 hardware",
            "description": "List the built-in model names and hardware names this fit-checker knows (for mapping user wording to exact names). Standard text-only HuggingFace transformer configs can also be checked via fitllm.run; unsupported architectures are rejected.",
            "inputSchema": {
              "$schema": "http://json-schema.org/draft-07/schema#",
              "properties": {},
              "type": "object"
            },
            "annotations": {
              "destructiveHint": false,
              "idempotentHint": true,
              "openWorldHint": false,
              "readOnlyHint": true
            }
          }
        ]
      },
      "name": "FitLLM",
      "note": "Indexed from the official MCP registry: facts and our own checks, not reviewed, so no score, grade or rank.",
      "packages": null,
      "pageJsonUrl": "https://www.anchorterminal.com/tools/fitllm.json",
      "popularity": {
        "githubStars": 8
      },
      "registryName": "run.fitllm/fitllm",
      "remotes": [
        {
          "type": "streamable-http",
          "url": "https://fitllm.run/api/mcp"
        }
      ],
      "repository": "https://github.com/click6067-ship-it/fitllm-engine",
      "reviewed": false,
      "slug": "fitllm",
      "source": "the official MCP registry",
      "sourceUrl": "https://registry.modelcontextprotocol.io/v0.1/servers?search=run.fitllm/fitllm",
      "summary": "Will this LLM fit on your GPU, multi-GPU rig or Mac? Exact VRAM \u0026 KV-cache math. Read-only.",
      "updatedAt": "2026-09-02T17:35:00Z",
      "url": "https://www.anchorterminal.com/tools/fitllm",
      "vendor": "fitllm.run",
      "vendorUrl": "https://fitllm.run",
      "version": "1.1.0",
      "websiteUrl": "https://fitllm.run",
      "where": "hosted",
      "why": [
        "vendor"
      ]
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/tools/fitllm",
    "json": "https://www.anchorterminal.com/tools/fitllm.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/tools/fitllm.md",
    "slim": "https://www.anchorterminal.com/tools/fitllm.min.md"
  },
  "markdown": "# FitLLM\n\n\u003e Indexed, not reviewed: facts from the official MCP registry and our own checks. No score, grade or rank, and not in the rankings until the panel reviews it. How the index works: https://www.anchorterminal.com/indexed/\n\n- Kind: MCP server, by fitllm.run (https://fitllm.run)\n- Category: GPU \u0026 serverless compute (https://www.anchorterminal.com/categories/gpu-compute.md)\n- Listed because: It's published in the registry under fitllm.run, a namespace the registry only gives to whoever proves they control that domain.\n- What the official MCP registry says: Will this LLM fit on your GPU, multi-GPU rig or Mac? Exact VRAM \u0026 KV-cache math. Read-only.\n\n## Facts\n\n- MCP registry: `run.fitllm/fitllm` 1.1.0\n- Endpoint: https://fitllm.run/api/mcp (streamable HTTP)\n- Source: https://github.com/click6067-ship-it/fitllm-engine\n- Website: https://fitllm.run\n- GitHub stars: 8\n- Registry entry updated: 2026-09-02\n\n## Tools\n\n- Tools it lists (3, about 957 tokens of context, `tools/list` without credentials over MCP 2025-11-25, checked 2026-10-04 22:25 UTC):\n  - `check_llm_fit` (read-only): Check whether a specific local LLM fits in the memory of a specific GPU or Apple Silicon Mac. Returns fits/tight/won't-fit verdict with the memory breakdown…\n  - `what_fits_on_hardware` (read-only): Rank which popular local LLMs fit on a given GPU or Apple Silicon Mac (at ~4-bit quantization, 8K context) — models that fit come first, biggest first, with…\n  - `list_supported` (read-only): List the built-in model names and hardware names this fit-checker knows (for mapping user wording to exact names). Standard text-only HuggingFace transformer…\n- How its tools read to an agent (0 errors, 0 warnings, 1 note, about 957 tokens; rules at https://www.anchorterminal.com/check.md; not part of the score):\n  - note TC24 server: 3 of 3 tools have no outputSchema\n\n- JSON: https://www.anchorterminal.com/api/v1/tools/fitllm.json\n- Being indexed says nothing about quality, and nobody can pay for it. Ask for a review: https://www.anchorterminal.com/builders/#claiming\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-04",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.3",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Terminal",
        "url": "https://www.anchorterminal.com/tools/"
      },
      {
        "name": "Indexed",
        "url": "https://www.anchorterminal.com/indexed/"
      },
      {
        "name": "FitLLM",
        "url": ""
      }
    ],
    "description": "FitLLM, an MCP server by fitllm.run, listed from the official MCP registry. Indexed, not reviewed: facts and our own checks, no score or ranking. Will this LLM fit on your GPU, multi-GPU rig or Mac? Exact VRAM \u0026 KV-cache math. Read-only.",
    "facts": [
      "not reviewed",
      "not ranked",
      "facts only"
    ],
    "h1": "FitLLM",
    "image": "https://www.anchorterminal.com/assets/og/indexed.png",
    "path": "/tools/fitllm",
    "published": "",
    "section": "indexed",
    "title": "FitLLM: MCP server, indexed from the official MCP registry",
    "toc": null,
    "updated": "2026-10-04",
    "url": "https://www.anchorterminal.com/tools/fitllm"
  },
  "tokens": {
    "markdown": 700,
    "slim": 630
  },
  "version": 1
}
