{
  "data": {
    "category": {
      "area": "web-data",
      "capabilities": [
        "docs.parse",
        "docs.ocr",
        "docs.extract",
        "docs.tables",
        "docs.chunk"
      ],
      "description": "APIs that turn scans, PDFs, invoices and forms into text, tables and structured fields an agent can use. Compared on text accuracy, table structure, field extraction, page references, processing time and price per page.",
      "indexed": [
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/diemdesk-mcp.json",
          "kind": "mcp",
          "name": "diemdesk.com MCP server",
          "slug": "diemdesk-mcp",
          "url": "https://www.anchorterminal.com/tools/diemdesk-mcp"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/nestr-mcp.json",
          "kind": "mcp",
          "name": "nestr.io MCP server",
          "slug": "nestr-mcp",
          "url": "https://www.anchorterminal.com/tools/nestr-mcp"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/acjlabs-receipt-extraction.json",
          "kind": "mcp",
          "name": "Receipt Extraction",
          "slug": "acjlabs-receipt-extraction",
          "url": "https://www.anchorterminal.com/tools/acjlabs-receipt-extraction"
        }
      ],
      "indexedCount": 3,
      "json": "https://www.anchorterminal.com/categories/document-extraction.json",
      "name": "Document parsing \u0026 extraction",
      "slug": "document-extraction",
      "test": "The same scans, invoices, tables and long PDFs through every API. We score text accuracy, table structure, field extraction, page references, processing time and price per page.",
      "title": "Document parsing, OCR and extraction APIs for AI agents",
      "toolCount": 16,
      "tools": [
        "google-cloud-document-ai",
        "amazon-textract",
        "landingai-agentic-document-extraction",
        "azure-document-intelligence",
        "extend",
        "reducto",
        "mindee",
        "llamaparse",
        "mistral-ocr",
        "opendocrouter",
        "adobe-pdf-extract",
        "veryfi",
        "unstructured",
        "invofox",
        "nanonets",
        "pdf-co"
      ],
      "url": "https://www.anchorterminal.com/categories/document-extraction"
    },
    "faq": [
      {
        "answer": "Google Cloud Document AI has the highest benchmark score of the 16 ranked document parsing, OCR and extraction APIs, 73.9 (BB). Amazon Textract is second with 73.5 (BB).",
        "question": "What are the highest-rated document parsing, OCR and extraction APIs for AI agents?"
      },
      {
        "answer": "2 of the 16 ranked here grade BB or better, the bar for agent-ready on the Anchor benchmark.",
        "question": "How many document parsing, OCR and extraction APIs are agent-ready?"
      },
      {
        "answer": "None of the ranked listings here accepts x402 for its main call yet.",
        "question": "Which document parsing, OCR and extraction APIs accept x402 payments?"
      },
      {
        "answer": "By published paid prices, OpenDocRouter, at $0.80 per 1,000 pages, the lowest of the 13 listings here with a paid price in this unit (free allowances aside). Plans, volume tiers and free allowances change the sum, so check the listing's price table.",
        "question": "Which of these document parsing, OCR and extraction APIs is cheapest?"
      },
      {
        "answer": "By the Anchor benchmark score out of 100, a weighted mean of the scored categories minus deductions for negative events, from public evidence re-checked as vendors change. Listings cannot pay for a place. The latest assessment behind this page is from 9 October 2026.",
        "question": "How is this list ranked?"
      }
    ],
    "howToChoose": [
      {
        "label": "Text accuracy on scans",
        "detail": "Test text accuracy on your own scans, since clean born-digital PDFs hide the errors that matter when an agent reads a faded, skewed or handwritten page."
      },
      {
        "label": "Table structure and layout",
        "detail": "Check that tables keep their rows, columns and merged cells, since a flattened table can look correct as text while the figures no longer line up for the agent."
      },
      {
        "label": "Page references for citations",
        "detail": "Look for page references on each extracted field, so an agent can cite the source page and a person can check a figure against the original document."
      },
      {
        "label": "Price per page and file limits",
        "detail": "Compare price per page at the volume you expect, and check the file types and sizes accepted, since long scanned PDFs can cost more or fail outright."
      }
    ],
    "picks": [
      {
        "also": {
          "name": "Amazon Textract",
          "slug": "amazon-textract",
          "why": "BB, 73.5/100"
        },
        "name": "Google Cloud Document AI",
        "need": "Highest score overall",
        "slug": "google-cloud-document-ai",
        "why": "BB, 73.9/100 on the benchmark"
      },
      {
        "name": "Amazon Textract",
        "need": "Reliability",
        "slug": "amazon-textract",
        "why": "96/100 on reliability, against 90 for the overall leader"
      },
      {
        "name": "Mistral OCR API",
        "need": "Schema \u0026 documentation",
        "slug": "mistral-ocr",
        "why": "89/100 on schema \u0026 documentation, against 81 for the overall leader"
      },
      {
        "name": "Reducto API + MCP",
        "need": "Agent ergonomics",
        "slug": "reducto",
        "why": "84/100 on agent ergonomics, against 70 for the overall leader"
      },
      {
        "name": "LandingAI Agentic Document Extraction",
        "need": "Maintenance \u0026 community",
        "slug": "landingai-agentic-document-extraction",
        "why": "85/100 on maintenance \u0026 community, against 80 for the overall leader"
      },
      {
        "also": {
          "name": "LlamaParse API + MCP",
          "slug": "llamaparse",
          "why": "$1.25 per 1,000 pages"
        },
        "name": "OpenDocRouter",
        "need": "Lowest paid price per 1,000 pages",
        "slug": "opendocrouter",
        "why": "$0.80 per 1,000 pages, the lowest of the 13 listings here with a paid price in this unit (free allowances aside)"
      },
      {
        "name": "Extend API + MCP",
        "need": "A hosted MCP endpoint",
        "slug": "extend",
        "why": "remote MCP server, nothing to install"
      },
      {
        "name": "Unstructured API + MCP",
        "need": "Self-hosting under an open licence",
        "slug": "unstructured",
        "why": "self-hosted, Apache-2 licence"
      }
    ],
    "ranked": 16,
    "shortlist": [
      {
        "bestFor": "Agents already on Google Cloud that need OCR, form and table extraction or chunks for retrieval, with IAM, audit logs and EU processing.",
        "grade": "BB",
        "name": "Google Cloud Document AI",
        "position": 1,
        "price": "$1.50 / 1k pages",
        "score": 73.9,
        "slug": "google-cloud-document-ai",
        "strengths": [
          "Discovery document for v1 (revision 20260929) with 42 methods and 326 schemas, and descriptions on 952 of 966 properties",
          "`fieldMask`, `imagelessMode` and page selectors on the process request limit what comes back and what is billed",
          "The Document AI API User role allows processing only, roles can be granted on one processor, and process calls write Data Access audit logs once enabled"
        ],
        "url": "https://www.anchorterminal.com/tools/google-cloud-document-ai",
        "verdict": "A public Discovery document with 42 methods, IAM roles that can limit a caller to processing, a `fieldMask` that trims responses, and a 99.9 per cent SLA on the US and EU endpoints. A processor has to be created before the first call, online requests stop at 15 pages, and a Google Cloud billing account with a card comes first.",
        "weaknesses": [
          "An online request reads at most 15 pages (30 with `imagelessMode`). Longer files need a batch job through Cloud Storage",
          "A processor must be created in a project and location before any document can be sent, and the endpoint host changes with the location",
          "No guidance on retrying quota errors was found on the quotas, limits or request pages, and there is no idempotency key"
        ],
        "where": "local",
        "x402": "no"
      },
      {
        "bestFor": "Teams already on AWS with documents in S3 that need OCR, tables and form fields at volume with IAM and CloudTrail controls, and for US invoices, receipts, identity documents and mortgage packages.",
        "grade": "BB",
        "name": "Amazon Textract",
        "position": 2,
        "price": "$1.50 / 1k pages",
        "score": 73.5,
        "slug": "amazon-textract",
        "strengths": [
          "IAM policies grant single operations such as `textract:DetectDocumentText`, with temporary credentials, and CloudTrail logs every call without the document bytes or the response",
          "Text detection costs $1.50 per 1,000 pages in US East (N. Virginia) and $0.60 after a million pages a month, published without a login",
          "`ClientRequestToken` on the `Start` operations returns the same `JobId` for a repeated call, so a retried submission does not start a second job"
        ],
        "url": "https://www.anchorterminal.com/tools/amazon-textract",
        "verdict": "IAM policies limit a credential to single operations, CloudTrail logs every call, and page prices start at $1.50 per 1,000 in US East. Multipage files need S3 and an asynchronous job. Text detection covers six languages. Under the AWS Service Terms AWS may store and use documents to improve the service unless an AI services opt-out policy is set.",
        "weaknesses": [
          "AWS may store and use processed documents to improve the service, in other Regions too, unless an AI services opt-out policy is set on the AWS organisation",
          "Synchronous calls take one page of at most 10 MB. Multipage PDF and TIFF files must sit in S3 and run as asynchronous jobs",
          "Text detection covers English, French, German, Italian, Portuguese and Spanish only. Handwriting and queries are English only, and vertical text is not read"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Agents that need Markdown with block-level page coordinates and schema extraction with citations, in the US or EU.",
        "grade": "B",
        "name": "LandingAI Agentic Document Extraction",
        "position": 3,
        "price": "$0.01 / credit",
        "score": 69,
        "slug": "landingai-agentic-document-extraction",
        "strengths": [
          "Public OpenAPI 3.1 description of the nine Gen2 operations, `llms.txt`, `llms-full.txt` and a Markdown twin of every docs page",
          "Credit rates published per page and per 1,000 characters, and each response reports the credits it consumed in `metadata.billing`",
          "1,000 free credits on signup with no card, then $0.01 a credit on the Explore and Team plans"
        ],
        "url": "https://www.anchorterminal.com/tools/landingai-agentic-document-extraction",
        "verdict": "Parse, Extract and Ground have a public OpenAPI description, Markdown docs, published credit rates and page limits, and a status page showing no incident in 90 days. The Explore plan's single API key cannot be revoked without support, and the published terms let LandingAI train on customer documents unless Zero Data Retention, a Team and Enterprise setting, is on.",
        "weaknesses": [
          "The Explore plan has one API key that cannot be deleted or revoked. A leaked key is replaced by emailing support",
          "The published terms let LandingAI use customer materials and output to train its models. Zero Data Retention, which stops that, is a Team and Enterprise setting",
          "The terms forbid accessing the Solution through any agent, robot or tool LandingAI does not supply, and forbid benchmarking. This matters before any probe is run"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Teams already on Azure that want OCR, layout to Markdown and prebuilt invoice, receipt, identity and tax models with Entra ID and regional processing.",
        "grade": "B",
        "name": "Azure Document Intelligence",
        "position": 4,
        "price": "$1.50 / 1k pages",
        "score": 66,
        "slug": "azure-document-intelligence",
        "strengths": [
          "OpenAPI 2.0 document for version 2024-11-30 in Microsoft's public spec repository, 27 operations, each with an example",
          "The layout model returns Markdown with headings and tables when `outputContentFormat=markdown` is set, and reads PDF, images, DOCX, XLSX, PPTX and HTML",
          "Input and results are deleted 24 hours after an analysis completes, and a delete call removes them sooner"
        ],
        "url": "https://www.anchorterminal.com/tools/azure-document-intelligence",
        "verdict": "A public OpenAPI document with 27 operations, Markdown output from the layout model, Microsoft Entra ID or header-only keys, and 24-hour retention with a delete call. Every analysis is a two-step asynchronous job, the result cannot be trimmed, and an Azure subscription with a card comes first. Three of the four SDKs last shipped in 2025 or earlier.",
        "weaknesses": [
          "Every analysis is asynchronous. A POST returns 202 with `Operation-Location`, and the caller polls for the result",
          "The result carries every word and line with polygons, with no parameter to leave them out. Only `pages` narrows it",
          "No idempotency key. Sending the same POST again starts and bills a second analysis"
        ],
        "where": "local",
        "x402": "no"
      },
      {
        "bestFor": "Teams building production extraction with evaluation sets and versioned processors, and for agents that need to build or tune extractors as well as run them.",
        "grade": "B",
        "name": "Extend API + MCP",
        "position": 5,
        "price": "$25 / 1k pages",
        "score": 62.9,
        "slug": "extend",
        "strengths": [
          "Hosted MCP with OAuth scoped to workspaces and to test or production, and a tools filter with nine groups",
          "Errors carry code, retryable, requestId and a docs link, and 429 guidance names Retry-After and jittered backoff",
          "Published credit prices and per-plan rate limits, 10,000 free credits to start"
        ],
        "url": "https://www.anchorterminal.com/tools/extend",
        "verdict": "Hosted MCP with OAuth scoped to workspaces and to test or production, and a tools filter with nine groups. Parse is billed on top of Extract, Split and Classify, so Performance Extract costs 5 credits a page.",
        "weaknesses": [
          "Parse is billed on top of Extract, Split and Classify, so Performance Extract costs 5 credits a page",
          "MCP loads every tool unless you pass ?tools=, including write and delete tools with no documented confirmation",
          "Three major-impact incidents on the status page since 25 August"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Messy PDFs, tables and spreadsheets where an agent needs parse, extract and split behind one small MCP.",
        "grade": "B",
        "name": "Reducto API + MCP",
        "position": 6,
        "price": "$10 / 1k pages",
        "score": 62.9,
        "slug": "reducto",
        "strengths": [
          "Nine MCP tools with when-to-use descriptions, jobid:// chaining and URL results for large outputs",
          "Per-1,000-page list prices for every endpoint, parsing included in Extract and Split",
          "Coded 429 bodies, and SDKs that retry 429s with backoff"
        ],
        "url": "https://www.anchorterminal.com/tools/reducto",
        "verdict": "Nine MCP tools with when-to-use descriptions, jobid:// chaining and URL results for large outputs. Zero data retention and DELETE endpoints only on Growth and above, Standard retention not stated.",
        "weaknesses": [
          "Zero data retention and DELETE endpoints only on Growth and above, Standard retention not stated",
          "No public changelog for the hosted API",
          "11 incidents since 23 July, mostly latency, with Parse degraded on 1 October"
        ],
        "where": "both",
        "x402": "no"
      },
      {
        "bestFor": "Teams with a fixed set of document types (invoices, receipts, IDs) who want short retention and EU processing.",
        "grade": "C",
        "name": "Mindee API",
        "position": 7,
        "price": "$44 / mo",
        "score": 61.3,
        "slug": "mindee",
        "strengths": [
          "Extracted data kept 12 hours by default (1 to 24 configurable), deletable on fetch, source files never stored",
          "EU or US processing selectable, and SOC 2 Type II per the docs",
          "Problem-details errors with 15 documented cases"
        ],
        "url": "https://www.anchorterminal.com/tools/mindee",
        "verdict": "Extracted data kept 12 hours by default (1 to 24 configurable), deletable on fetch, source files never stored. Every call needs a model_id created in the web platform first.",
        "weaknesses": [
          "Every call needs a model_id created in the web platform first",
          "No free tier after the 14-day trial, and the pricing page's platform fee and currency were unclear to us",
          "Organisation-wide keys that never expire and reach every model"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "RAG ingestion where cost per page matters and a tier can be picked per document, and for agents that want a small MCP surface.",
        "grade": "C",
        "name": "LlamaParse API + MCP",
        "position": 8,
        "price": "$1.25 / 1k pages",
        "score": 59.2,
        "slug": "llamaparse",
        "strengths": [
          "Free plan with 10,000 credits a month, no card",
          "Product-scoped MCP endpoints with 1 to 5 tools each, OAuth or API key, NA and EU",
          "API keys scoped to a project and, since 26 August 2026, able to expire"
        ],
        "url": "https://www.anchorterminal.com/tools/llamaparse",
        "verdict": "Free plan with 10,000 credits a month, no card. Breaking SDK changes shipped as minor releases (files.get renamed, classify v1 removed).",
        "weaknesses": [
          "Breaking SDK changes shipped as minor releases (files.get renamed, classify v1 removed)",
          "No 429 or Retry-After guidance, and most endpoint rate limits unpublished",
          "Agentic Plus parse plus Agentic Plus extract costs 95 credits a page, about $119 per 1,000 pages"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Turning PDFs and scans into Markdown for RAG or reading at a flat, low page price.",
        "grade": "C",
        "name": "Mistral OCR API",
        "position": 9,
        "price": "$4 / 1k pages",
        "score": 58.8,
        "slug": "mistral-ocr",
        "strengths": [
          "Single synchronous call returns Markdown per page, no upload step for public URLs",
          "Flat price, $4 per 1,000 pages and $5 with annotations",
          "Tables as HTML or Markdown, headers and footers split out, block bounding boxes and page, block or word confidence"
        ],
        "url": "https://www.anchorterminal.com/tools/mistral-ocr",
        "verdict": "Single synchronous call returns Markdown per page, no upload step for public URLs. OCR API at 99.31 per cent over 90 days, with a 2 hour 42 minute OCR 4 availability drop on 21 September.",
        "weaknesses": [
          "OCR API at 99.31 per cent over 90 days, with a 2 hour 42 minute OCR 4 availability drop on 21 September",
          "OCR 4.0 lasted about three months before retiring on 30 September",
          "Free tier data may be used for training"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Agents that need page-level markdown from PDFs and images and want to choose or compare models on price and published benchmark scores with one key.",
        "grade": "C",
        "name": "OpenDocRouter",
        "position": 10,
        "price": "$48.82 / 1k pages",
        "score": 56.6,
        "slug": "opendocrouter",
        "strengths": [
          "Public OpenAPI 3.1 document for all six operations, and the docs page is also served as Markdown at /docs.md",
          "`GET /v1/models` lists 11 models without a key, each with token prices, an average and a maximum charge per page, and ParseBench scores",
          "Failed, cached and blank pages are free, and every page comes back with its own status, token usage and charge"
        ],
        "url": "https://www.anchorterminal.com/tools/opendocrouter",
        "verdict": "One endpoint reaches 11 parsing models, each with a dated recipe version, a published token price and a ParseBench score, and failed pages aren't charged. The service launched on 7 October 2026, so it has no status page, incident history, SLA or changelog yet, and no key scopes were found in the reviewed documentation.",
        "weaknesses": [
          "Launched on 7 October 2026. The domain was registered on 23 September 2026 and both SDKs have one release, 1.0.0",
          "No status page, SLA or service changelog was found. LlamaIndex's status page lists only LlamaParse components",
          "No scoped, expiring or read-only keys in the reviewed documentation, and no idempotency key"
        ],
        "where": "hosted",
        "x402": "no"
      }
    ],
    "updated": "2026-10-09"
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/best/document-extraction/",
    "json": "https://www.anchorterminal.com/best/document-extraction/index.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/best/document-extraction/index.md",
    "slim": "https://www.anchorterminal.com/best/document-extraction/index.min.md"
  },
  "markdown": "The 10 highest-scoring of 16 document parsing, OCR and extraction APIs on the Anchor benchmark, with a pick for each need and where each one falls short. Scores come from public evidence, re-checked as vendors change.\n\n- Ranked: 16 · agent-ready (BB or better): 2 · accept x402: 0 · hosted endpoints: 14\n- Full ranked table: https://www.anchorterminal.com/categories/document-extraction.md\n- Head-to-head comparisons: https://www.anchorterminal.com/compare/document-extraction/index.md (109)\n- Methodology: https://www.anchorterminal.com/benchmark/index.md\n\n## The shortlist\n\n| # | Tool | Grade | Score | Best for | Price | Where |\n| --- | --- | --- | --- | --- | --- | --- |\n| 1 | [Google Cloud Document AI](https://www.anchorterminal.com/tools/google-cloud-document-ai.md) | BB | 73.9 | Agents already on Google Cloud that need OCR, form and table extraction or chunks for retrieval, with IAM, audit logs and EU processing. | $1.50 / 1k pages | local |\n| 2 | [Amazon Textract](https://www.anchorterminal.com/tools/amazon-textract.md) | BB | 73.5 | Teams already on AWS with documents in S3 that need OCR, tables and form fields at volume with IAM and CloudTrail controls, and for US invoices, receipts, identity documents and mortgage packages. | $1.50 / 1k pages | hosted |\n| 3 | [LandingAI Agentic Document Extraction](https://www.anchorterminal.com/tools/landingai-agentic-document-extraction.md) | B | 69 | Agents that need Markdown with block-level page coordinates and schema extraction with citations, in the US or EU. | $0.01 / credit | hosted |\n| 4 | [Azure Document Intelligence](https://www.anchorterminal.com/tools/azure-document-intelligence.md) | B | 66 | Teams already on Azure that want OCR, layout to Markdown and prebuilt invoice, receipt, identity and tax models with Entra ID and regional processing. | $1.50 / 1k pages | local |\n| 5 | [Extend API + MCP](https://www.anchorterminal.com/tools/extend.md) | B | 62.9 | Teams building production extraction with evaluation sets and versioned processors, and for agents that need to build or tune extractors as well as run them. | $25 / 1k pages | hosted |\n| 6 | [Reducto API + MCP](https://www.anchorterminal.com/tools/reducto.md) | B | 62.9 | Messy PDFs, tables and spreadsheets where an agent needs parse, extract and split behind one small MCP. | $10 / 1k pages | hosted and local |\n| 7 | [Mindee API](https://www.anchorterminal.com/tools/mindee.md) | C | 61.3 | Teams with a fixed set of document types (invoices, receipts, IDs) who want short retention and EU processing. | $44 / mo | hosted |\n| 8 | [LlamaParse API + MCP](https://www.anchorterminal.com/tools/llamaparse.md) | C | 59.2 | RAG ingestion where cost per page matters and a tier can be picked per document, and for agents that want a small MCP surface. | $1.25 / 1k pages | hosted |\n| 9 | [Mistral OCR API](https://www.anchorterminal.com/tools/mistral-ocr.md) | C | 58.8 | Turning PDFs and scans into Markdown for RAG or reading at a flat, low page price. | $4 / 1k pages | hosted |\n| 10 | [OpenDocRouter](https://www.anchorterminal.com/tools/opendocrouter.md) | C | 56.6 | Agents that need page-level markdown from PDFs and images and want to choose or compare models on price and published benchmark scores with one key. | $48.82 / 1k pages | hosted |\n\n## Picks by need\n\n- Highest score overall: [Google Cloud Document AI](https://www.anchorterminal.com/tools/google-cloud-document-ai.md), BB, 73.9/100 on the benchmark. Also [Amazon Textract](https://www.anchorterminal.com/tools/amazon-textract.md), BB, 73.5/100.\n- Reliability: [Amazon Textract](https://www.anchorterminal.com/tools/amazon-textract.md), 96/100 on reliability, against 90 for the overall leader.\n- Schema \u0026 documentation: [Mistral OCR API](https://www.anchorterminal.com/tools/mistral-ocr.md), 89/100 on schema \u0026 documentation, against 81 for the overall leader.\n- Agent ergonomics: [Reducto API + MCP](https://www.anchorterminal.com/tools/reducto.md), 84/100 on agent ergonomics, against 70 for the overall leader.\n- Maintenance \u0026 community: [LandingAI Agentic Document Extraction](https://www.anchorterminal.com/tools/landingai-agentic-document-extraction.md), 85/100 on maintenance \u0026 community, against 80 for the overall leader.\n- Lowest paid price per 1,000 pages: [OpenDocRouter](https://www.anchorterminal.com/tools/opendocrouter.md), $0.80 per 1,000 pages, the lowest of the 13 listings here with a paid price in this unit (free allowances aside). Also [LlamaParse API + MCP](https://www.anchorterminal.com/tools/llamaparse.md), $1.25 per 1,000 pages.\n- A hosted MCP endpoint: [Extend API + MCP](https://www.anchorterminal.com/tools/extend.md), remote MCP server, nothing to install.\n- Self-hosting under an open licence: [Unstructured API + MCP](https://www.anchorterminal.com/tools/unstructured.md), self-hosted, Apache-2 licence.\n\n## How to choose\n\n- Text accuracy on scans: Test text accuracy on your own scans, since clean born-digital PDFs hide the errors that matter when an agent reads a faded, skewed or handwritten page.\n- Table structure and layout: Check that tables keep their rows, columns and merged cells, since a flattened table can look correct as text while the figures no longer line up for the agent.\n- Page references for citations: Look for page references on each extracted field, so an agent can cite the source page and a person can check a figure against the original document.\n- Price per page and file limits: Compare price per page at the volume you expect, and check the file types and sizes accepted, since long scanned PDFs can cost more or fail outright.\n\n- How the benchmark tests this category: The same scans, invoices, tables and long PDFs through every API. We score text accuracy, table structure, field extraction, page references, processing time and price per page.\n\n## Each one in detail\n\n### 1. Google Cloud Document AI, BB 73.9/100\n\nGoogle Cloud's service for OCR, layout parsing, chunking, form and table extraction, classification and splitting of documents. Work runs through processors created per project and location. Access is a REST and gRPC API with client libraries in eight languages.\n\n- Verdict: A public Discovery document with 42 methods, IAM roles that can limit a caller to processing, a `fieldMask` that trims responses, and a 99.9 per cent SLA on the US and EU endpoints. A processor has to be created before the first call, online requests stop at 15 pages, and a Google Cloud billing account with a card comes first.\n- Choose it for: Agents already on Google Cloud that need OCR, form and table extraction or chunks for retrieval, with IAM, audit logs and EU processing.\n- Strength: Discovery document for v1 (revision 20260929) with 42 methods and 326 schemas, and descriptions on 952 of 966 properties\n- Strength: `fieldMask`, `imagelessMode` and page selectors on the process request limit what comes back and what is billed\n- Strength: The Document AI API User role allows processing only, roles can be granted on one processor, and process calls write Data Access audit logs once enabled\n- Weakness: An online request reads at most 15 pages (30 with `imagelessMode`). Longer files need a batch job through Cloud Storage\n- Weakness: A processor must be created in a project and location before any document can be sent, and the endpoint host changes with the location\n- Weakness: No guidance on retrying quota errors was found on the quotas, limits or request pages, and there is no idempotency key\n- Price: $1.50 / 1k pages · Auth: OAuth · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/google-cloud-document-ai.md\n\n### 2. Amazon Textract, BB 73.5/100\n\nAmazon Textract is AWS's document OCR and analysis API. It reads printed and handwritten text from scans and PDFs and returns tables, form fields, layout elements, answers to queries, and invoice, receipt and identity document fields as JSON.\n\n- Verdict: IAM policies limit a credential to single operations, CloudTrail logs every call, and page prices start at $1.50 per 1,000 in US East. Multipage files need S3 and an asynchronous job. Text detection covers six languages. Under the AWS Service Terms AWS may store and use documents to improve the service unless an AI services opt-out policy is set.\n- Choose it for: Teams already on AWS with documents in S3 that need OCR, tables and form fields at volume with IAM and CloudTrail controls, and for US invoices, receipts, identity documents and mortgage packages.\n- Strength: IAM policies grant single operations such as `textract:DetectDocumentText`, with temporary credentials, and CloudTrail logs every call without the document bytes or the response\n- Strength: Text detection costs $1.50 per 1,000 pages in US East (N. Virginia) and $0.60 after a million pages a month, published without a login\n- Strength: `ClientRequestToken` on the `Start` operations returns the same `JobId` for a repeated call, so a retried submission does not start a second job\n- Weakness: AWS may store and use processed documents to improve the service, in other Regions too, unless an AI services opt-out policy is set on the AWS organisation\n- Weakness: Synchronous calls take one page of at most 10 MB. Multipage PDF and TIFF files must sit in S3 and run as asynchronous jobs\n- Weakness: Text detection covers English, French, German, Italian, Portuguese and Spanish only. Handwriting and queries are English only, and vertical text is not read\n- Price: $1.50 / 1k pages · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/amazon-textract.md\n- Against #1: https://www.anchorterminal.com/compare/amazon-textract-vs-google-cloud-document-ai.md\n\n### 3. LandingAI Agentic Document Extraction, B 69/100\n\nLandingAI's Agentic Document Extraction is a hosted API that parses PDFs, images, Office files and spreadsheets into Markdown with page coordinates, then extracts fields against a JSON schema. Python and TypeScript libraries and a CLI call it.\n\n- Verdict: Parse, Extract and Ground have a public OpenAPI description, Markdown docs, published credit rates and page limits, and a status page showing no incident in 90 days. The Explore plan's single API key cannot be revoked without support, and the published terms let LandingAI train on customer documents unless Zero Data Retention, a Team and Enterprise setting, is on.\n- Choose it for: Agents that need Markdown with block-level page coordinates and schema extraction with citations, in the US or EU.\n- Strength: Public OpenAPI 3.1 description of the nine Gen2 operations, `llms.txt`, `llms-full.txt` and a Markdown twin of every docs page\n- Strength: Credit rates published per page and per 1,000 characters, and each response reports the credits it consumed in `metadata.billing`\n- Strength: 1,000 free credits on signup with no card, then $0.01 a credit on the Explore and Team plans\n- Weakness: The Explore plan has one API key that cannot be deleted or revoked. A leaked key is replaced by emailing support\n- Weakness: The published terms let LandingAI use customer materials and output to train its models. Zero Data Retention, which stops that, is a Team and Enterprise setting\n- Weakness: The terms forbid accessing the Solution through any agent, robot or tool LandingAI does not supply, and forbid benchmarking. This matters before any probe is run\n- Price: $0.01 / credit · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/landingai-agentic-document-extraction.md\n- Against #1: https://www.anchorterminal.com/compare/google-cloud-document-ai-vs-landingai-agentic-document-extraction.md\n\n### 4. Azure Document Intelligence, B 66/100\n\nMicrosoft's Azure service for OCR, layout analysis and field extraction from PDFs, images and Office files. It has prebuilt models for invoices, receipts, identity and tax documents, and custom models. Access is a REST API with four SDKs.\n\n- Verdict: A public OpenAPI document with 27 operations, Markdown output from the layout model, Microsoft Entra ID or header-only keys, and 24-hour retention with a delete call. Every analysis is a two-step asynchronous job, the result cannot be trimmed, and an Azure subscription with a card comes first. Three of the four SDKs last shipped in 2025 or earlier.\n- Choose it for: Teams already on Azure that want OCR, layout to Markdown and prebuilt invoice, receipt, identity and tax models with Entra ID and regional processing.\n- Strength: OpenAPI 2.0 document for version 2024-11-30 in Microsoft's public spec repository, 27 operations, each with an example\n- Strength: The layout model returns Markdown with headings and tables when `outputContentFormat=markdown` is set, and reads PDF, images, DOCX, XLSX, PPTX and HTML\n- Strength: Input and results are deleted 24 hours after an analysis completes, and a delete call removes them sooner\n- Weakness: Every analysis is asynchronous. A POST returns 202 with `Operation-Location`, and the caller polls for the result\n- Weakness: The result carries every word and line with polygons, with no parameter to leave them out. Only `pages` narrows it\n- Weakness: No idempotency key. Sending the same POST again starts and bills a second analysis\n- Price: $1.50 / 1k pages · Auth: OAuth or key · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/azure-document-intelligence.md\n- Against #1: https://www.anchorterminal.com/compare/azure-document-intelligence-vs-google-cloud-document-ai.md\n\n### 5. Extend API + MCP, B 62.9/100\n\nHosted parse, extract, classify, split and PDF form-fill APIs with versioned processors, evaluation sets and workflows.\n\n- Verdict: Hosted MCP with OAuth scoped to workspaces and to test or production, and a tools filter with nine groups. Parse is billed on top of Extract, Split and Classify, so Performance Extract costs 5 credits a page.\n- Choose it for: Teams building production extraction with evaluation sets and versioned processors, and for agents that need to build or tune extractors as well as run them.\n- Strength: Hosted MCP with OAuth scoped to workspaces and to test or production, and a tools filter with nine groups\n- Strength: Errors carry code, retryable, requestId and a docs link, and 429 guidance names Retry-After and jittered backoff\n- Strength: Published credit prices and per-plan rate limits, 10,000 free credits to start\n- Weakness: Parse is billed on top of Extract, Split and Classify, so Performance Extract costs 5 credits a page\n- Weakness: MCP loads every tool unless you pass ?tools=, including write and delete tools with no documented confirmation\n- Weakness: Three major-impact incidents on the status page since 25 August\n- Price: $25 / 1k pages · Auth: OAuth or key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/extend.md\n- Against #1: https://www.anchorterminal.com/compare/extend-vs-google-cloud-document-ai.md\n\n### 6. Reducto API + MCP, B 62.9/100\n\nHosted parse, extract, split, classify and edit endpoints for PDFs, scans, spreadsheets and Office files, built on its own r-1 parsing model.\n\n- Verdict: Nine MCP tools with when-to-use descriptions, jobid:// chaining and URL results for large outputs. Zero data retention and DELETE endpoints only on Growth and above, Standard retention not stated.\n- Choose it for: Messy PDFs, tables and spreadsheets where an agent needs parse, extract and split behind one small MCP.\n- Strength: Nine MCP tools with when-to-use descriptions, jobid:// chaining and URL results for large outputs\n- Strength: Per-1,000-page list prices for every endpoint, parsing included in Extract and Split\n- Strength: Coded 429 bodies, and SDKs that retry 429s with backoff\n- Weakness: Zero data retention and DELETE endpoints only on Growth and above, Standard retention not stated\n- Weakness: No public changelog for the hosted API\n- Weakness: 11 incidents since 23 July, mostly latency, with Parse degraded on 1 October\n- Price: $10 / 1k pages · Auth: API key · x402: no · Where: hosted and local\n- Full assessment: https://www.anchorterminal.com/tools/reducto.md\n- Against #1: https://www.anchorterminal.com/compare/google-cloud-document-ai-vs-reducto.md\n\n### 7. Mindee API, C 61.3/100\n\nHosted extraction, classification, split, crop and raw-text OCR models that you define in the Mindee platform, then call by model ID through an async REST API.\n\n- Verdict: Extracted data kept 12 hours by default (1 to 24 configurable), deletable on fetch, source files never stored. Every call needs a model_id created in the web platform first.\n- Choose it for: Teams with a fixed set of document types (invoices, receipts, IDs) who want short retention and EU processing.\n- Strength: Extracted data kept 12 hours by default (1 to 24 configurable), deletable on fetch, source files never stored\n- Strength: EU or US processing selectable, and SOC 2 Type II per the docs\n- Strength: Problem-details errors with 15 documented cases\n- Weakness: Every call needs a model_id created in the web platform first\n- Weakness: No free tier after the 14-day trial, and the pricing page's platform fee and currency were unclear to us\n- Weakness: Organisation-wide keys that never expire and reach every model\n- Price: $44 / mo · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/mindee.md\n- Against #1: https://www.anchorterminal.com/compare/google-cloud-document-ai-vs-mindee.md\n\n### 8. LlamaParse API + MCP, C 59.2/100\n\nLlamaIndex's hosted Parse, Extract, Classify, Split and Index APIs on one LlamaCloud key, billed in credits by tier.\n\n- Verdict: Free plan with 10,000 credits a month, no card. Breaking SDK changes shipped as minor releases (files.get renamed, classify v1 removed).\n- Choose it for: RAG ingestion where cost per page matters and a tier can be picked per document, and for agents that want a small MCP surface.\n- Strength: Free plan with 10,000 credits a month, no card\n- Strength: Product-scoped MCP endpoints with 1 to 5 tools each, OAuth or API key, NA and EU\n- Strength: API keys scoped to a project and, since 26 August 2026, able to expire\n- Weakness: Breaking SDK changes shipped as minor releases (files.get renamed, classify v1 removed)\n- Weakness: No 429 or Retry-After guidance, and most endpoint rate limits unpublished\n- Weakness: Agentic Plus parse plus Agentic Plus extract costs 95 credits a page, about $119 per 1,000 pages\n- Price: $1.25 / 1k pages · Auth: OAuth or key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/llamaparse.md\n- Against #1: https://www.anchorterminal.com/compare/google-cloud-document-ai-vs-llamaparse.md\n\n### 9. Mistral OCR API, C 58.8/100\n\nMistral's OCR API for extracting content from documents.\n\n- Verdict: Single synchronous call returns Markdown per page, no upload step for public URLs. OCR API at 99.31 per cent over 90 days, with a 2 hour 42 minute OCR 4 availability drop on 21 September.\n- Choose it for: Turning PDFs and scans into Markdown for RAG or reading at a flat, low page price.\n- Strength: Single synchronous call returns Markdown per page, no upload step for public URLs\n- Strength: Flat price, $4 per 1,000 pages and $5 with annotations\n- Strength: Tables as HTML or Markdown, headers and footers split out, block bounding boxes and page, block or word confidence\n- Weakness: OCR API at 99.31 per cent over 90 days, with a 2 hour 42 minute OCR 4 availability drop on 21 September\n- Weakness: OCR 4.0 lasted about three months before retiring on 30 September\n- Weakness: Free tier data may be used for training\n- Price: $4 / 1k pages · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/mistral-ocr.md\n- Against #1: https://www.anchorterminal.com/compare/google-cloud-document-ai-vs-mistral-ocr.md\n\n### 10. OpenDocRouter, C 56.6/100\n\nOpenDocRouter is a hosted API from LlamaIndex that sends PDFs and images to a document parsing model the caller picks and returns markdown for each page. It launched on 7 October 2026 and bills prepaid credit by tokens used.\n\n- Verdict: One endpoint reaches 11 parsing models, each with a dated recipe version, a published token price and a ParseBench score, and failed pages aren't charged. The service launched on 7 October 2026, so it has no status page, incident history, SLA or changelog yet, and no key scopes were found in the reviewed documentation.\n- Choose it for: Agents that need page-level markdown from PDFs and images and want to choose or compare models on price and published benchmark scores with one key.\n- Strength: Public OpenAPI 3.1 document for all six operations, and the docs page is also served as Markdown at /docs.md\n- Strength: `GET /v1/models` lists 11 models without a key, each with token prices, an average and a maximum charge per page, and ParseBench scores\n- Strength: Failed, cached and blank pages are free, and every page comes back with its own status, token usage and charge\n- Weakness: Launched on 7 October 2026. The domain was registered on 23 September 2026 and both SDKs have one release, 1.0.0\n- Weakness: No status page, SLA or service changelog was found. LlamaIndex's status page lists only LlamaParse components\n- Weakness: No scoped, expiring or read-only keys in the reviewed documentation, and no idempotency key\n- Price: $48.82 / 1k pages · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/opendocrouter.md\n- Against #1: https://www.anchorterminal.com/compare/google-cloud-document-ai-vs-opendocrouter.md\n\n6 more are ranked in the full table: https://www.anchorterminal.com/categories/document-extraction.md\n\n## Head to head\n\n- [Amazon Textract vs Google Cloud Document AI](https://www.anchorterminal.com/compare/amazon-textract-vs-google-cloud-document-ai.md)\n- [Google Cloud Document AI vs LandingAI Agentic Document Extraction](https://www.anchorterminal.com/compare/google-cloud-document-ai-vs-landingai-agentic-document-extraction.md)\n- [Azure Document Intelligence vs Google Cloud Document AI](https://www.anchorterminal.com/compare/azure-document-intelligence-vs-google-cloud-document-ai.md)\n- [Extend API + MCP vs Google Cloud Document AI](https://www.anchorterminal.com/compare/extend-vs-google-cloud-document-ai.md)\n- [Amazon Textract vs LandingAI Agentic Document Extraction](https://www.anchorterminal.com/compare/amazon-textract-vs-landingai-agentic-document-extraction.md)\n- [Amazon Textract vs Azure Document Intelligence](https://www.anchorterminal.com/compare/amazon-textract-vs-azure-document-intelligence.md)\n- [Amazon Textract vs Extend API + MCP](https://www.anchorterminal.com/compare/amazon-textract-vs-extend.md)\n- [Azure Document Intelligence vs LandingAI Agentic Document Extraction](https://www.anchorterminal.com/compare/azure-document-intelligence-vs-landingai-agentic-document-extraction.md)\n- [Extend API + MCP vs LandingAI Agentic Document Extraction](https://www.anchorterminal.com/compare/extend-vs-landingai-agentic-document-extraction.md)\n- [Azure Document Intelligence vs Extend API + MCP](https://www.anchorterminal.com/compare/azure-document-intelligence-vs-extend.md)\n\n## Questions\n\n### What are the highest-rated document parsing, OCR and extraction APIs for AI agents?\n\nGoogle Cloud Document AI has the highest benchmark score of the 16 ranked document parsing, OCR and extraction APIs, 73.9 (BB). Amazon Textract is second with 73.5 (BB).\n\n### How many document parsing, OCR and extraction APIs are agent-ready?\n\n2 of the 16 ranked here grade BB or better, the bar for agent-ready on the Anchor benchmark.\n\n### Which document parsing, OCR and extraction APIs accept x402 payments?\n\nNone of the ranked listings here accepts x402 for its main call yet.\n\n### Which of these document parsing, OCR and extraction APIs is cheapest?\n\nBy published paid prices, OpenDocRouter, at $0.80 per 1,000 pages, the lowest of the 13 listings here with a paid price in this unit (free allowances aside). Plans, volume tiers and free allowances change the sum, so check the listing's price table.\n\n### How is this list ranked?\n\nBy the Anchor benchmark score out of 100, a weighted mean of the scored categories minus deductions for negative events, from public evidence re-checked as vendors change. Listings cannot pay for a place. The latest assessment behind this page is from 9 October 2026.\n\n## How this list is made\n\nThe order is the Anchor benchmark score, the same number as on each listing. Each listing is graded from public evidence against the benchmark checklist, and the picks are worked out from those grades, prices and facts. No listing pays for its place, and paid audits or listing help never change a score.\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-10",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Best of",
        "url": "https://www.anchorterminal.com/best/"
      },
      {
        "name": "Document parsing \u0026 extraction",
        "url": ""
      }
    ],
    "description": "Google Cloud Document AI (BB), Amazon Textract (BB) and LandingAI Agentic Document Extraction (B) lead the 16 ranked document parsing, OCR and extraction APIs. Picks by need, strengths, weaknesses and prices from the Anchor benchmark.",
    "facts": [
      "Google Cloud Document AI BB",
      "Amazon Textract BB",
      "LandingAI Agentic Document Extraction B"
    ],
    "h1": "Best document parsing, OCR and extraction APIs for AI agents",
    "image": "https://www.anchorterminal.com/assets/og/best-document-extraction.png",
    "path": "/best/document-extraction/",
    "published": "",
    "section": "tools",
    "title": "Best document parsing, OCR and extraction APIs for AI agents in 2026",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/best/document-extraction/"
  },
  "tokens": {
    "markdown": 6350,
    "slim": 1580
  },
  "version": 1
}
