{
  "data": {
    "category": {
      "area": "models",
      "capabilities": [
        "guard.injection",
        "guard.pii",
        "guard.moderation",
        "guard.policy",
        "guard.self-host"
      ],
      "description": "APIs and libraries that check what goes into and comes out of a model: prompt injection, jailbreaks, personal data, toxic content and off-policy answers. Compared on what they detect, false positives, the latency they add and where the data goes.",
      "indexed": [
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/icemoon-automation-studio.json",
          "kind": "mcp",
          "name": "Icemoon — control a real iPhone with AI",
          "slug": "icemoon-automation-studio",
          "url": "https://www.anchorterminal.com/tools/icemoon-automation-studio"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/cognivators-mcp-safeguard.json",
          "kind": "mcp",
          "name": "mcp-safeguard",
          "slug": "cognivators-mcp-safeguard",
          "url": "https://www.anchorterminal.com/tools/cognivators-mcp-safeguard"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/midplane.json",
          "kind": "mcp",
          "name": "Midplane",
          "slug": "midplane",
          "url": "https://www.anchorterminal.com/tools/midplane"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/overwing-mcp.json",
          "kind": "mcp",
          "name": "Overwing",
          "slug": "overwing-mcp",
          "url": "https://www.anchorterminal.com/tools/overwing-mcp"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/safeprompt-mcp.json",
          "kind": "mcp",
          "name": "safeprompt.dev MCP server",
          "slug": "safeprompt-mcp",
          "url": "https://www.anchorterminal.com/tools/safeprompt-mcp"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/skillssafe-scanner.json",
          "kind": "mcp",
          "name": "SkillsSafe Security Scanner",
          "slug": "skillssafe-scanner",
          "url": "https://www.anchorterminal.com/tools/skillssafe-scanner"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/aliengiraffe-spotdb.json",
          "kind": "mcp",
          "name": "spotdb",
          "slug": "aliengiraffe-spotdb",
          "url": "https://www.anchorterminal.com/tools/aliengiraffe-spotdb"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/smartmemory-stratum-mcp.json",
          "kind": "mcp",
          "name": "Stratum MCP",
          "slug": "smartmemory-stratum-mcp",
          "url": "https://www.anchorterminal.com/tools/smartmemory-stratum-mcp"
        },
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/thinkneo-control-plane.json",
          "kind": "mcp",
          "name": "ThinkNEO Control Plane",
          "slug": "thinkneo-control-plane",
          "url": "https://www.anchorterminal.com/tools/thinkneo-control-plane"
        }
      ],
      "indexedCount": 9,
      "json": "https://www.anchorterminal.com/categories/guardrails.json",
      "name": "Guardrails \u0026 safety filters",
      "slug": "guardrails",
      "test": "A set of prompts with injections, jailbreaks, personal data and clean inputs run through each filter. We count what is caught and what is wrongly blocked, and measure the latency each adds.",
      "title": "Guardrails and safety filters for AI agents",
      "toolCount": 16,
      "tools": [
        "google-model-armor",
        "amazon-bedrock-guardrails",
        "openai-moderation",
        "openai-guardrails",
        "nemo-guardrails",
        "microsoft-presidio",
        "prisma-airs",
        "cisco-ai-defense-inspection",
        "azure-ai-content-safety",
        "granite-guardian",
        "lakera-guard",
        "mistral-moderation",
        "llamafirewall",
        "guardrails-ai",
        "llama-guard",
        "galileo"
      ],
      "url": "https://www.anchorterminal.com/categories/guardrails"
    },
    "faq": [
      {
        "answer": "Google Cloud Model Armor has the highest benchmark score of the 16 ranked guardrails and safety filters, 77.9 (BB). Amazon Bedrock Guardrails is second with 74.8 (BB).",
        "question": "What are the highest-rated guardrails and safety filters for AI agents?"
      },
      {
        "answer": "3 of the 16 ranked here grade BB or better, the bar for agent-ready on the Anchor benchmark.",
        "question": "How many guardrails and safety filters are agent-ready?"
      },
      {
        "answer": "None of the ranked listings here accepts x402 for its main call yet.",
        "question": "Which guardrails and safety filters accept x402 payments?"
      },
      {
        "answer": "By the Anchor benchmark score out of 100, a weighted mean of the scored categories minus deductions for negative events, from public evidence re-checked as vendors change. Listings cannot pay for a place. The latest assessment behind this page is from 8 October 2026.",
        "question": "How is this list ranked?"
      }
    ],
    "howToChoose": [
      {
        "label": "Injection and jailbreak coverage",
        "detail": "Check which attack types are detected, including indirect injection in tool results, because agents read untrusted pages and documents that can carry instructions."
      },
      {
        "label": "Wrongful blocks on clean input",
        "detail": "Check the false positive rate on clean inputs, because a filter that blocks legitimate tool calls stops the agent's task without any attack taking place."
      },
      {
        "label": "Latency added to each call",
        "detail": "Check the latency each check adds at your payload size, since guardrails run on every model call and the delay compounds across a multi-step agent run."
      },
      {
        "label": "Self-hosting option",
        "detail": "Check whether the filter can run in your own infrastructure, because a hosted check sends every prompt, including personal data, to another party."
      }
    ],
    "picks": [
      {
        "also": {
          "name": "Amazon Bedrock Guardrails",
          "slug": "amazon-bedrock-guardrails",
          "why": "BB, 74.8/100"
        },
        "name": "Google Cloud Model Armor",
        "need": "Highest score overall",
        "slug": "google-model-armor",
        "why": "BB, 77.9/100 on the benchmark"
      },
      {
        "name": "Amazon Bedrock Guardrails",
        "need": "Schema \u0026 documentation",
        "slug": "amazon-bedrock-guardrails",
        "why": "92/100 on schema \u0026 documentation, against 78 for the overall leader"
      },
      {
        "name": "Amazon Bedrock Guardrails",
        "need": "Agent ergonomics",
        "slug": "amazon-bedrock-guardrails",
        "why": "93/100 on agent ergonomics, against 75 for the overall leader"
      },
      {
        "name": "OpenAI Guardrails",
        "need": "Maintenance \u0026 community",
        "slug": "openai-guardrails",
        "why": "89/100 on maintenance \u0026 community, against 85 for the overall leader"
      },
      {
        "name": "Galileo API + MCP",
        "need": "A hosted MCP endpoint",
        "slug": "galileo",
        "why": "remote MCP server, nothing to install"
      },
      {
        "also": {
          "name": "NVIDIA NeMo Guardrails",
          "slug": "nemo-guardrails",
          "why": "self-hosted, Apache-2 licence"
        },
        "name": "OpenAI Guardrails",
        "need": "Self-hosting under an open licence",
        "slug": "openai-guardrails",
        "why": "self-hosted, MIT licence"
      }
    ],
    "ranked": 16,
    "shortlist": [
      {
        "bestFor": "Teams already on Google Cloud who want prompt and response screening with real PII detection, document and URL scanning, and audit logs, at the lowest paid rate in the category.",
        "grade": "BB",
        "name": "Google Cloud Model Armor",
        "position": 1,
        "price": "Freemium",
        "score": 77.9,
        "slug": "google-model-armor",
        "strengths": [
          "2 million free tokens a month, then $0.10 per million",
          "No incidents for Model Armor on the Google Cloud status page in the last 90 days",
          "Each screening method has its own IAM permission and writes Data Access audit logs once the operator enables them"
        ],
        "url": "https://www.anchorterminal.com/tools/google-model-armor",
        "verdict": "2 million free tokens a month, then $0.10 per million. OAuth only, and a template must exist in the same location as the endpoint before the first call.",
        "weaknesses": [
          "OAuth only, and a template must exist in the same location as the endpoint before the first call",
          "Filter versions v1 and v2 retire on 17 December 2026, a date that moved from 29 November within the same month",
          "No SLA listed for Model Armor"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "A team already on AWS that wants one versioned policy covering topics, PII masking, grounding and prompt attacks in front of any model.",
        "grade": "BB",
        "name": "Amazon Bedrock Guardrails",
        "position": 2,
        "price": "Pay per use",
        "score": 74.8,
        "slug": "amazon-bedrock-guardrails",
        "strengths": [
          "ApplyGuardrail works with any model, self-hosted or third party, without invoking Bedrock inference",
          "InvokeGuardrailChecks takes the checks inline and returns severity and confidence scores, so no guardrail resource is needed",
          "IAM can grant bedrock:ApplyGuardrail on one guardrail ARN and nothing else, and calls land in CloudTrail as data events"
        ],
        "url": "https://www.anchorterminal.com/tools/amazon-bedrock-guardrails",
        "verdict": "ApplyGuardrail works with any model, self-hosted or third party, without invoking Bedrock inference. Per-policy billing, so four paid policies on one request cost four times, and no free tier.",
        "weaknesses": [
          "Per-policy billing, so four paid policies on one request cost four times, and no free tier",
          "Classic tier covers English, French and Spanish only, and Standard tier uses cross-Region inference that can move prompts within a geography",
          "Quota numbers are mostly in the Service Quotas console, with public figures only for two US regions"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "A free harm-category filter for an agent already on OpenAI.",
        "grade": "BB",
        "name": "OpenAI Moderation API",
        "position": 3,
        "price": "Free",
        "score": 71.3,
        "slug": "openai-moderation",
        "strengths": [
          "Free, on any OpenAI project key",
          "A restricted key can be limited to the moderation endpoint",
          "Text and images in the same request, with per-category scores"
        ],
        "url": "https://www.anchorterminal.com/tools/openai-moderation",
        "verdict": "Free, on any OpenAI project key. No prompt-injection, jailbreak or PII detection.",
        "weaknesses": [
          "No prompt-injection, jailbreak or PII detection",
          "One model snapshot from 26 September 2024, and scores can shift when the latest alias moves",
          "Fixed categories with no custom policies or per-request category choice"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Teams already on the OpenAI client or Agents SDK that want several checks from one config file with little code.",
        "grade": "B",
        "name": "OpenAI Guardrails",
        "position": 4,
        "price": "Free · OSS",
        "score": 69.5,
        "slug": "openai-guardrails",
        "strengths": [
          "MIT licence, source on GitHub, and three PyPI releases in the 90 days to 8 October 2026 (0.3.0, 0.3.2, 0.3.3)",
          "Twelve built-in checks set in one versioned JSON file across pre-flight, input and output stages",
          "`GuardrailAgent` runs the prompt injection check before and after every tool call in the OpenAI Agents SDK"
        ],
        "url": "https://www.anchorterminal.com/tools/openai-guardrails",
        "verdict": "MIT-licensed wrapper that adds twelve configurable checks to OpenAI client calls from one JSON file, with tool-level injection checks for the Agents SDK. The README labels it a preview at version 0.3.3, and by default a check that fails to run is reported as passed unless `raise_guardrail_errors=True` is set.",
        "weaknesses": [
          "By default a check that fails to run returns `tripwire_triggered=False`, so the request continues. Strict mode is opt-in",
          "The README titles the package a preview, the version is 0.3.3, and no release was published between 15 December 2025 and 21 July 2026",
          "With `stream=True` the output checks run alongside the stream, and the docs say violating content may appear briefly"
        ],
        "where": "library",
        "x402": "no"
      },
      {
        "bestFor": "Teams that want to compose several checks (their own, NVIDIA's and third-party APIs) behind one OpenAI-compatible endpoint.",
        "grade": "B",
        "name": "NVIDIA NeMo Guardrails",
        "position": 5,
        "price": "Free · OSS",
        "score": 68.4,
        "slug": "nemo-guardrails",
        "strengths": [
          "Apache-2.0, 7,200 stars and nine releases between 9 October 2025 and 16 September 2026",
          "Input, output, retrieval, dialogue, tool-input and tool-output rails in one config",
          "Adapters for about 20 hosted guardrail services plus NVIDIA's NemoGuard models"
        ],
        "url": "https://www.anchorterminal.com/tools/nemo-guardrails",
        "verdict": "Apache-2.0, 7,200 stars and nine releases between 9 October 2025 and 16 September 2026. Usage telemetry and a heartbeat every 10 minutes to NVIDIA by default.",
        "weaknesses": [
          "Usage telemetry and a heartbeat every 10 minutes to NVIDIA by default",
          "Six breaking changes in 0.24.0, and the project is still pre-1.0",
          "No authentication on the server, by design"
        ],
        "where": "library",
        "x402": "no"
      },
      {
        "bestFor": "Detecting and masking personal data in prompts, outputs, logs and images on the owner's own machines, with detection tuned by entity, threshold and custom recognisers.",
        "grade": "B",
        "name": "Presidio",
        "position": 6,
        "price": "Free · OSS",
        "score": 66,
        "slug": "microsoft-presidio",
        "strengths": [
          "MIT licence, source on GitHub, and nothing to buy. No account, key or card is needed to install or run it",
          "OpenAPI 3.0 document for the analyser and anonymiser REST services, with request examples and 400 and 422 error shapes",
          "CI runs each package on Python 3.10, 3.11, 3.12, 3.13 and 3.14, with CodeQL and Dependabot configured"
        ],
        "url": "https://www.anchorterminal.com/tools/microsoft-presidio",
        "verdict": "MIT-licensed personal data detector with a public OpenAPI document, tests on Python 3.10 to 3.14 and about 1.2 million weekly PyPI downloads. The REST containers have no authentication, the project states no SLA or support, and it covers personal data only, with no prompt injection or content moderation checks.",
        "weaknesses": [
          "The REST containers have no authentication by design. The FAQ says to put a gateway or proxy in front",
          "SUPPORT.md states no SLA and no official support. The project is run by volunteers since leaving Microsoft",
          "One release in the 90 days to 8 October 2026 (2.2.364 on 22 July), and CHANGELOG.md has no section for it"
        ],
        "where": "local",
        "x402": "no"
      },
      {
        "bestFor": "A company already buying Palo Alto Networks through Strata Cloud Manager that wants prompt, response and MCP tool scanning with DLP and URL filtering from the same vendor.",
        "grade": "B",
        "name": "Prisma AIRS AI Runtime Security API",
        "position": 7,
        "price": "Paid",
        "score": 62.8,
        "slug": "prisma-airs",
        "strengths": [
          "One call scans a prompt, a response and an MCP tool event for up to ten detection types set by the security profile",
          "Public OpenAPI 3.0.3 documents for the scan API (4 operations) and the management API (21 operations)",
          "API keys carry a rotation period, expiry and revocation, and OAuth service accounts take custom roles with per-entity permissions"
        ],
        "url": "https://www.anchorterminal.com/tools/prisma-airs",
        "verdict": "A public OpenAPI document, ten detection types in one call, tool-call scanning, a remote MCP server, rotating API keys and OAuth roles suit teams already on Strata Cloud Manager. Access needs Software NGFW credits bought through sales, with no public price, trial or self-serve signup, and payloads flagged malicious are kept for up to 10 years.",
        "weaknesses": [
          "No public price, free tier or trial. Capacity is bought as Software NGFW credits, in steps of 1 billion tokens a month, through sales",
          "Payloads judged malicious are kept for up to 10 years, including after the subscription ends, per the privacy datasheet of 13 July 2026",
          "The EULA of August 2026 forbids publishing benchmark or comparison tests and forbids load testing of subscriptions"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "A company already buying Cisco security through Security Cloud Control that wants its own application to decide what to do with each verdict.",
        "grade": "C",
        "name": "Cisco AI Defense Inspection API",
        "position": 8,
        "price": "Paid",
        "score": 61.3,
        "slug": "cisco-ai-defense-inspection",
        "strengths": [
          "One call returns `is_safe`, an action, classifications, severity, matched rules and, since April 2026, the type and position of each suspected PII value",
          "Each connection has its own API key with an expiry date, revocation and regeneration, and the key works only on the Inspection API",
          "Published limits of 60,000 calls a minute, a one million token context per inspection and one million tokens in parallel per organisation"
        ],
        "url": "https://www.anchorterminal.com/tools/cisco-ai-defense-inspection",
        "verdict": "Per-connection API keys with expiry, revocation and regeneration, published limits of 60,000 calls a minute, weekly dated release notes and a one-year retention statement suit companies already on Cisco Security Cloud Control. Access needs a subscription bought through sales, with no public price or trial for the API, and the developer changelog has one entry from February 2025.",
        "weaknesses": [
          "No public price, free tier or trial for the Inspection API. Subscriptions are bought through sales and activated with a claim code",
          "The Toxicity rule was removed on 16 September 2026 in the release note that announced it, with no earlier notice found",
          "The developer changelog lists only v1.0.0 of 28 February 2025, while release notes record later changes to the API's responses and rules"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "An agent on Azure that retrieves documents and needs indirect-injection checks next to harm-category moderation.",
        "grade": "C",
        "name": "Azure AI Content Safety (Prompt Shields)",
        "position": 9,
        "price": "$338 / mo",
        "score": 60.7,
        "slug": "azure-ai-content-safety",
        "strengths": [
          "Prompt Shields checks up to five retrieved documents for indirect injection, not only the user prompt",
          "5,000 free text records and 5,000 free images a month on F0",
          "FAQ and data-privacy page agree that inputs aren't stored or trained on and stay in the resource's region"
        ],
        "url": "https://www.anchorterminal.com/tools/azure-ai-content-safety",
        "verdict": "Prompt Shields checks up to five retrieved documents for indirect injection, not only the user prompt. Needs an Azure subscription with a card, a resource and a region that has the feature, before the first call.",
        "weaknesses": [
          "Needs an Azure subscription with a card, a resource and a region that has the feature, before the first call",
          "Python SDK is 1.0.0 from December 2023 and has no Prompt Shields method",
          "No retry or 429 guidance in the Content Safety docs"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "A team with a GPU that wants one English-language judge for harm, jailbreaks, RAG groundedness, function-call checks and house rules, under a permissive licence with no gate.",
        "grade": "C",
        "name": "Granite Guardian",
        "position": 10,
        "price": "Free",
        "score": 60.1,
        "slug": "granite-guardian",
        "strengths": [
          "Weights are ungated on Hugging Face under the Apache 2.0 licence, with IBM's own GGUF builds and an Ollama library entry",
          "Built-in criteria cover harm, social bias, jailbreaking, violence, profanity, sexual content, unethical behaviour, three RAG checks and function-call hallucination",
          "A custom criterion is one natural-language sentence in the prompt, and the answer is `yes` or `no` inside `\u003cscore\u003e` tags"
        ],
        "url": "https://www.anchorterminal.com/tools/granite-guardian",
        "verdict": "One ungated Apache 2.0 model judges harm, jailbreaks, RAG groundedness, function-call errors and custom criteria, with signed weights and published evaluation code. It is trained and tested on English only, each call checks one criterion, the 4.1 prompt format differs from 3.x, and IBM's watsonx.ai lists only the deprecated 3.0 model.",
        "weaknesses": [
          "Trained and tested on English only, per the model card",
          "Each call judges one criterion, so checking several risks takes several calls or a separate LoRA adapter built on the 3.2 model",
          "Version 4.1 moved the criterion into a `\u003cguardian\u003e` block in the last user message, where 3.x cookbooks pass `guardian_config`"
        ],
        "where": "local",
        "x402": "no"
      }
    ],
    "updated": "2026-10-08"
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/best/guardrails/",
    "json": "https://www.anchorterminal.com/best/guardrails/index.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/best/guardrails/index.md",
    "slim": "https://www.anchorterminal.com/best/guardrails/index.min.md"
  },
  "markdown": "The 10 highest-scoring of 16 guardrails and safety filters on the Anchor benchmark, with a pick for each need and where each one falls short. Scores come from public evidence, re-checked as vendors change.\n\n- Ranked: 16 · agent-ready (BB or better): 3 · accept x402: 0 · hosted endpoints: 9\n- Full ranked table: https://www.anchorterminal.com/categories/guardrails.md\n- Head-to-head comparisons: https://www.anchorterminal.com/compare/guardrails/index.md (118)\n- Methodology: https://www.anchorterminal.com/benchmark/index.md\n\n## The shortlist\n\n| # | Tool | Grade | Score | Best for | Price | Where |\n| --- | --- | --- | --- | --- | --- | --- |\n| 1 | [Google Cloud Model Armor](https://www.anchorterminal.com/tools/google-model-armor.md) | BB | 77.9 | Teams already on Google Cloud who want prompt and response screening with real PII detection, document and URL scanning, and audit logs, at the lowest paid rate in the category. | Freemium | hosted |\n| 2 | [Amazon Bedrock Guardrails](https://www.anchorterminal.com/tools/amazon-bedrock-guardrails.md) | BB | 74.8 | A team already on AWS that wants one versioned policy covering topics, PII masking, grounding and prompt attacks in front of any model. | Pay per use | hosted |\n| 3 | [OpenAI Moderation API](https://www.anchorterminal.com/tools/openai-moderation.md) | BB | 71.3 | A free harm-category filter for an agent already on OpenAI. | Free | hosted |\n| 4 | [OpenAI Guardrails](https://www.anchorterminal.com/tools/openai-guardrails.md) | B | 69.5 | Teams already on the OpenAI client or Agents SDK that want several checks from one config file with little code. | Free · OSS | library |\n| 5 | [NVIDIA NeMo Guardrails](https://www.anchorterminal.com/tools/nemo-guardrails.md) | B | 68.4 | Teams that want to compose several checks (their own, NVIDIA's and third-party APIs) behind one OpenAI-compatible endpoint. | Free · OSS | library |\n| 6 | [Presidio](https://www.anchorterminal.com/tools/microsoft-presidio.md) | B | 66 | Detecting and masking personal data in prompts, outputs, logs and images on the owner's own machines, with detection tuned by entity, threshold and custom recognisers. | Free · OSS | local |\n| 7 | [Prisma AIRS AI Runtime Security API](https://www.anchorterminal.com/tools/prisma-airs.md) | B | 62.8 | A company already buying Palo Alto Networks through Strata Cloud Manager that wants prompt, response and MCP tool scanning with DLP and URL filtering from the same vendor. | Paid | hosted |\n| 8 | [Cisco AI Defense Inspection API](https://www.anchorterminal.com/tools/cisco-ai-defense-inspection.md) | C | 61.3 | A company already buying Cisco security through Security Cloud Control that wants its own application to decide what to do with each verdict. | Paid | hosted |\n| 9 | [Azure AI Content Safety (Prompt Shields)](https://www.anchorterminal.com/tools/azure-ai-content-safety.md) | C | 60.7 | An agent on Azure that retrieves documents and needs indirect-injection checks next to harm-category moderation. | $338 / mo | hosted |\n| 10 | [Granite Guardian](https://www.anchorterminal.com/tools/granite-guardian.md) | C | 60.1 | A team with a GPU that wants one English-language judge for harm, jailbreaks, RAG groundedness, function-call checks and house rules, under a permissive licence with no gate. | Free | local |\n\n## Picks by need\n\n- Highest score overall: [Google Cloud Model Armor](https://www.anchorterminal.com/tools/google-model-armor.md), BB, 77.9/100 on the benchmark. Also [Amazon Bedrock Guardrails](https://www.anchorterminal.com/tools/amazon-bedrock-guardrails.md), BB, 74.8/100.\n- Schema \u0026 documentation: [Amazon Bedrock Guardrails](https://www.anchorterminal.com/tools/amazon-bedrock-guardrails.md), 92/100 on schema \u0026 documentation, against 78 for the overall leader.\n- Agent ergonomics: [Amazon Bedrock Guardrails](https://www.anchorterminal.com/tools/amazon-bedrock-guardrails.md), 93/100 on agent ergonomics, against 75 for the overall leader.\n- Maintenance \u0026 community: [OpenAI Guardrails](https://www.anchorterminal.com/tools/openai-guardrails.md), 89/100 on maintenance \u0026 community, against 85 for the overall leader.\n- A hosted MCP endpoint: [Galileo API + MCP](https://www.anchorterminal.com/tools/galileo.md), remote MCP server, nothing to install.\n- Self-hosting under an open licence: [OpenAI Guardrails](https://www.anchorterminal.com/tools/openai-guardrails.md), self-hosted, MIT licence. Also [NVIDIA NeMo Guardrails](https://www.anchorterminal.com/tools/nemo-guardrails.md), self-hosted, Apache-2 licence.\n\n## How to choose\n\n- Injection and jailbreak coverage: Check which attack types are detected, including indirect injection in tool results, because agents read untrusted pages and documents that can carry instructions.\n- Wrongful blocks on clean input: Check the false positive rate on clean inputs, because a filter that blocks legitimate tool calls stops the agent's task without any attack taking place.\n- Latency added to each call: Check the latency each check adds at your payload size, since guardrails run on every model call and the delay compounds across a multi-step agent run.\n- Self-hosting option: Check whether the filter can run in your own infrastructure, because a hosted check sends every prompt, including personal data, to another party.\n\n- How the benchmark tests this category: A set of prompts with injections, jailbreaks, personal data and clean inputs run through each filter. We count what is caught and what is wrongly blocked, and measure the latency each adds.\n\n## Each one in detail\n\n### 1. Google Cloud Model Armor, BB 77.9/100\n\nGoogle Cloud's prompt and response screening service.\n\n- Verdict: 2 million free tokens a month, then $0.10 per million. OAuth only, and a template must exist in the same location as the endpoint before the first call.\n- Choose it for: Teams already on Google Cloud who want prompt and response screening with real PII detection, document and URL scanning, and audit logs, at the lowest paid rate in the category.\n- Strength: 2 million free tokens a month, then $0.10 per million\n- Strength: No incidents for Model Armor on the Google Cloud status page in the last 90 days\n- Strength: Each screening method has its own IAM permission and writes Data Access audit logs once the operator enables them\n- Weakness: OAuth only, and a template must exist in the same location as the endpoint before the first call\n- Weakness: Filter versions v1 and v2 retire on 17 December 2026, a date that moved from 29 November within the same month\n- Weakness: No SLA listed for Model Armor\n- Price: Freemium · Auth: OAuth · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/google-model-armor.md\n\n### 2. Amazon Bedrock Guardrails, BB 74.8/100\n\nConfigurable guardrail policies (content filters with a prompt-attack category, denied topics, word filters, PII and regex filters, contextual grounding, Automated Reasoning checks) applied to any model through the ApplyGuardrail API, or inline through InvokeGuardrailChecks.\n\n- Verdict: ApplyGuardrail works with any model, self-hosted or third party, without invoking Bedrock inference. Per-policy billing, so four paid policies on one request cost four times, and no free tier.\n- Choose it for: A team already on AWS that wants one versioned policy covering topics, PII masking, grounding and prompt attacks in front of any model.\n- Strength: ApplyGuardrail works with any model, self-hosted or third party, without invoking Bedrock inference\n- Strength: InvokeGuardrailChecks takes the checks inline and returns severity and confidence scores, so no guardrail resource is needed\n- Strength: IAM can grant bedrock:ApplyGuardrail on one guardrail ARN and nothing else, and calls land in CloudTrail as data events\n- Weakness: Per-policy billing, so four paid policies on one request cost four times, and no free tier\n- Weakness: Classic tier covers English, French and Spanish only, and Standard tier uses cross-Region inference that can move prompts within a geography\n- Weakness: Quota numbers are mostly in the Service Quotas console, with public figures only for two US regions\n- Price: Pay per use · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/amazon-bedrock-guardrails.md\n- Against #1: https://www.anchorterminal.com/compare/amazon-bedrock-guardrails-vs-google-model-armor.md\n\n### 3. OpenAI Moderation API, BB 71.3/100\n\nFree classifier endpoint that scores text and images against 13 harm categories (harassment, hate, illicit, self-harm, sexual, violence and their sub-types) and returns a flagged boolean plus per-category scores.\n\n- Verdict: Free, on any OpenAI project key. No prompt-injection, jailbreak or PII detection.\n- Choose it for: A free harm-category filter for an agent already on OpenAI.\n- Strength: Free, on any OpenAI project key\n- Strength: A restricted key can be limited to the moderation endpoint\n- Strength: Text and images in the same request, with per-category scores\n- Weakness: No prompt-injection, jailbreak or PII detection\n- Weakness: One model snapshot from 26 September 2024, and scores can shift when the latest alias moves\n- Weakness: Fixed categories with no custom policies or per-request category choice\n- Price: Free · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/openai-moderation.md\n- Against #1: https://www.anchorterminal.com/compare/google-model-armor-vs-openai-moderation.md\n\n### 4. OpenAI Guardrails, B 69.5/100\n\nOpenAI's open-source Python library that wraps the OpenAI client and runs configured checks on inputs, outputs and tool calls, including moderation, jailbreak, prompt injection, personal data, URL and off-topic checks. It is labelled a preview.\n\n- Verdict: MIT-licensed wrapper that adds twelve configurable checks to OpenAI client calls from one JSON file, with tool-level injection checks for the Agents SDK. The README labels it a preview at version 0.3.3, and by default a check that fails to run is reported as passed unless `raise_guardrail_errors=True` is set.\n- Choose it for: Teams already on the OpenAI client or Agents SDK that want several checks from one config file with little code.\n- Strength: MIT licence, source on GitHub, and three PyPI releases in the 90 days to 8 October 2026 (0.3.0, 0.3.2, 0.3.3)\n- Strength: Twelve built-in checks set in one versioned JSON file across pre-flight, input and output stages\n- Strength: `GuardrailAgent` runs the prompt injection check before and after every tool call in the OpenAI Agents SDK\n- Weakness: By default a check that fails to run returns `tripwire_triggered=False`, so the request continues. Strict mode is opt-in\n- Weakness: The README titles the package a preview, the version is 0.3.3, and no release was published between 15 December 2025 and 21 July 2026\n- Weakness: With `stream=True` the output checks run alongside the stream, and the docs say violating content may appear briefly\n- Price: Free · OSS · Auth: API key · x402: no · Where: library\n- Full assessment: https://www.anchorterminal.com/tools/openai-guardrails.md\n- Against #1: https://www.anchorterminal.com/compare/google-model-armor-vs-openai-guardrails.md\n\n### 5. NVIDIA NeMo Guardrails, B 68.4/100\n\nOpen-source Python toolkit that runs input, output, retrieval, dialogue and tool rails around any LLM.\n\n- Verdict: Apache-2.0, 7,200 stars and nine releases between 9 October 2025 and 16 September 2026. Usage telemetry and a heartbeat every 10 minutes to NVIDIA by default.\n- Choose it for: Teams that want to compose several checks (their own, NVIDIA's and third-party APIs) behind one OpenAI-compatible endpoint.\n- Strength: Apache-2.0, 7,200 stars and nine releases between 9 October 2025 and 16 September 2026\n- Strength: Input, output, retrieval, dialogue, tool-input and tool-output rails in one config\n- Strength: Adapters for about 20 hosted guardrail services plus NVIDIA's NemoGuard models\n- Weakness: Usage telemetry and a heartbeat every 10 minutes to NVIDIA by default\n- Weakness: Six breaking changes in 0.24.0, and the project is still pre-1.0\n- Weakness: No authentication on the server, by design\n- Price: Free · OSS · Auth: None · x402: no · Where: library\n- Full assessment: https://www.anchorterminal.com/tools/nemo-guardrails.md\n- Against #1: https://www.anchorterminal.com/compare/google-model-armor-vs-nemo-guardrails.md\n\n### 6. Presidio, B 66/100\n\nOpen-source Python library and Docker services that detect personal data in text and images and replace, mask, hash or encrypt it. Created at Microsoft and run since June 2026 by the community organisation Data Privacy Stack.\n\n- Verdict: MIT-licensed personal data detector with a public OpenAPI document, tests on Python 3.10 to 3.14 and about 1.2 million weekly PyPI downloads. The REST containers have no authentication, the project states no SLA or support, and it covers personal data only, with no prompt injection or content moderation checks.\n- Choose it for: Detecting and masking personal data in prompts, outputs, logs and images on the owner's own machines, with detection tuned by entity, threshold and custom recognisers.\n- Strength: MIT licence, source on GitHub, and nothing to buy. No account, key or card is needed to install or run it\n- Strength: OpenAPI 3.0 document for the analyser and anonymiser REST services, with request examples and 400 and 422 error shapes\n- Strength: CI runs each package on Python 3.10, 3.11, 3.12, 3.13 and 3.14, with CodeQL and Dependabot configured\n- Weakness: The REST containers have no authentication by design. The FAQ says to put a gateway or proxy in front\n- Weakness: SUPPORT.md states no SLA and no official support. The project is run by volunteers since leaving Microsoft\n- Weakness: One release in the 90 days to 8 October 2026 (2.2.364 on 22 July), and CHANGELOG.md has no section for it\n- Price: Free · OSS · Auth: None · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/microsoft-presidio.md\n- Against #1: https://www.anchorterminal.com/compare/google-model-armor-vs-microsoft-presidio.md\n\n### 7. Prisma AIRS AI Runtime Security API, B 62.8/100\n\nHosted scan API from Palo Alto Networks that checks prompts, model responses and tool calls for prompt injection, sensitive data, toxic content, malicious URLs and code, and off-topic content against a security profile. Also reachable as a remote MCP server.\n\n- Verdict: A public OpenAPI document, ten detection types in one call, tool-call scanning, a remote MCP server, rotating API keys and OAuth roles suit teams already on Strata Cloud Manager. Access needs Software NGFW credits bought through sales, with no public price, trial or self-serve signup, and payloads flagged malicious are kept for up to 10 years.\n- Choose it for: A company already buying Palo Alto Networks through Strata Cloud Manager that wants prompt, response and MCP tool scanning with DLP and URL filtering from the same vendor.\n- Strength: One call scans a prompt, a response and an MCP tool event for up to ten detection types set by the security profile\n- Strength: Public OpenAPI 3.0.3 documents for the scan API (4 operations) and the management API (21 operations)\n- Strength: API keys carry a rotation period, expiry and revocation, and OAuth service accounts take custom roles with per-entity permissions\n- Weakness: No public price, free tier or trial. Capacity is bought as Software NGFW credits, in steps of 1 billion tokens a month, through sales\n- Weakness: Payloads judged malicious are kept for up to 10 years, including after the subscription ends, per the privacy datasheet of 13 July 2026\n- Weakness: The EULA of August 2026 forbids publishing benchmark or comparison tests and forbids load testing of subscriptions\n- Price: Paid · Auth: OAuth or key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/prisma-airs.md\n- Against #1: https://www.anchorterminal.com/compare/google-model-armor-vs-prisma-airs.md\n\n### 8. Cisco AI Defense Inspection API, C 61.3/100\n\nHosted inspection API from Cisco that checks chat messages, HTTP requests and responses, and MCP messages for prompt injection, personal data, harmful content and policy violations, and returns an allow or block verdict for the calling application to enforce.\n\n- Verdict: Per-connection API keys with expiry, revocation and regeneration, published limits of 60,000 calls a minute, weekly dated release notes and a one-year retention statement suit companies already on Cisco Security Cloud Control. Access needs a subscription bought through sales, with no public price or trial for the API, and the developer changelog has one entry from February 2025.\n- Choose it for: A company already buying Cisco security through Security Cloud Control that wants its own application to decide what to do with each verdict.\n- Strength: One call returns `is_safe`, an action, classifications, severity, matched rules and, since April 2026, the type and position of each suspected PII value\n- Strength: Each connection has its own API key with an expiry date, revocation and regeneration, and the key works only on the Inspection API\n- Strength: Published limits of 60,000 calls a minute, a one million token context per inspection and one million tokens in parallel per organisation\n- Weakness: No public price, free tier or trial for the Inspection API. Subscriptions are bought through sales and activated with a claim code\n- Weakness: The Toxicity rule was removed on 16 September 2026 in the release note that announced it, with no earlier notice found\n- Weakness: The developer changelog lists only v1.0.0 of 28 February 2025, while release notes record later changes to the API's responses and rules\n- Price: Paid · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/cisco-ai-defense-inspection.md\n- Against #1: https://www.anchorterminal.com/compare/cisco-ai-defense-inspection-vs-google-model-armor.md\n\n### 9. Azure AI Content Safety (Prompt Shields), C 60.7/100\n\nMicrosoft's API for analysing harmful text and images, detecting prompt injection and checking groundedness.\n\n- Verdict: Prompt Shields checks up to five retrieved documents for indirect injection, not only the user prompt. Needs an Azure subscription with a card, a resource and a region that has the feature, before the first call.\n- Choose it for: An agent on Azure that retrieves documents and needs indirect-injection checks next to harm-category moderation.\n- Strength: Prompt Shields checks up to five retrieved documents for indirect injection, not only the user prompt\n- Strength: 5,000 free text records and 5,000 free images a month on F0\n- Strength: FAQ and data-privacy page agree that inputs aren't stored or trained on and stay in the resource's region\n- Weakness: Needs an Azure subscription with a card, a resource and a region that has the feature, before the first call\n- Weakness: Python SDK is 1.0.0 from December 2023 and has no Prompt Shields method\n- Weakness: No retry or 429 guidance in the Content Safety docs\n- Price: $338 / mo · Auth: OAuth or key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/azure-ai-content-safety.md\n- Against #1: https://www.anchorterminal.com/compare/azure-ai-content-safety-vs-google-model-armor.md\n\n### 10. Granite Guardian, C 60.1/100\n\nGranite Guardian is IBM's family of open-weight judge models. The current 8-billion-parameter release answers yes or no on whether a prompt, response, retrieved context or function call meets a built-in or custom criterion, and the owner runs it.\n\n- Verdict: One ungated Apache 2.0 model judges harm, jailbreaks, RAG groundedness, function-call errors and custom criteria, with signed weights and published evaluation code. It is trained and tested on English only, each call checks one criterion, the 4.1 prompt format differs from 3.x, and IBM's watsonx.ai lists only the deprecated 3.0 model.\n- Choose it for: A team with a GPU that wants one English-language judge for harm, jailbreaks, RAG groundedness, function-call checks and house rules, under a permissive licence with no gate.\n- Strength: Weights are ungated on Hugging Face under the Apache 2.0 licence, with IBM's own GGUF builds and an Ollama library entry\n- Strength: Built-in criteria cover harm, social bias, jailbreaking, violence, profanity, sexual content, unethical behaviour, three RAG checks and function-call hallucination\n- Strength: A custom criterion is one natural-language sentence in the prompt, and the answer is `yes` or `no` inside `\u003cscore\u003e` tags\n- Weakness: Trained and tested on English only, per the model card\n- Weakness: Each call judges one criterion, so checking several risks takes several calls or a separate LoRA adapter built on the 3.2 model\n- Weakness: Version 4.1 moved the criterion into a `\u003cguardian\u003e` block in the last user message, where 3.x cookbooks pass `guardian_config`\n- Price: Free · Auth: None · x402: no · Where: local\n- Full assessment: https://www.anchorterminal.com/tools/granite-guardian.md\n- Against #1: https://www.anchorterminal.com/compare/google-model-armor-vs-granite-guardian.md\n\n6 more are ranked in the full table: https://www.anchorterminal.com/categories/guardrails.md\n\n## Head to head\n\n- [Amazon Bedrock Guardrails vs Google Cloud Model Armor](https://www.anchorterminal.com/compare/amazon-bedrock-guardrails-vs-google-model-armor.md)\n- [Google Cloud Model Armor vs OpenAI Moderation API](https://www.anchorterminal.com/compare/google-model-armor-vs-openai-moderation.md)\n- [Google Cloud Model Armor vs OpenAI Guardrails](https://www.anchorterminal.com/compare/google-model-armor-vs-openai-guardrails.md)\n- [Google Cloud Model Armor vs NVIDIA NeMo Guardrails](https://www.anchorterminal.com/compare/google-model-armor-vs-nemo-guardrails.md)\n- [Amazon Bedrock Guardrails vs OpenAI Moderation API](https://www.anchorterminal.com/compare/amazon-bedrock-guardrails-vs-openai-moderation.md)\n- [Amazon Bedrock Guardrails vs OpenAI Guardrails](https://www.anchorterminal.com/compare/amazon-bedrock-guardrails-vs-openai-guardrails.md)\n- [Amazon Bedrock Guardrails vs NVIDIA NeMo Guardrails](https://www.anchorterminal.com/compare/amazon-bedrock-guardrails-vs-nemo-guardrails.md)\n- [OpenAI Guardrails vs OpenAI Moderation API](https://www.anchorterminal.com/compare/openai-guardrails-vs-openai-moderation.md)\n- [NVIDIA NeMo Guardrails vs OpenAI Moderation API](https://www.anchorterminal.com/compare/nemo-guardrails-vs-openai-moderation.md)\n- [NVIDIA NeMo Guardrails vs OpenAI Guardrails](https://www.anchorterminal.com/compare/nemo-guardrails-vs-openai-guardrails.md)\n\n## Questions\n\n### What are the highest-rated guardrails and safety filters for AI agents?\n\nGoogle Cloud Model Armor has the highest benchmark score of the 16 ranked guardrails and safety filters, 77.9 (BB). Amazon Bedrock Guardrails is second with 74.8 (BB).\n\n### How many guardrails and safety filters are agent-ready?\n\n3 of the 16 ranked here grade BB or better, the bar for agent-ready on the Anchor benchmark.\n\n### Which guardrails and safety filters accept x402 payments?\n\nNone of the ranked listings here accepts x402 for its main call yet.\n\n### How is this list ranked?\n\nBy the Anchor benchmark score out of 100, a weighted mean of the scored categories minus deductions for negative events, from public evidence re-checked as vendors change. Listings cannot pay for a place. The latest assessment behind this page is from 8 October 2026.\n\n## How this list is made\n\nThe order is the Anchor benchmark score, the same number as on each listing. Each listing is graded from public evidence against the benchmark checklist, and the picks are worked out from those grades, prices and facts. No listing pays for its place, and paid audits or listing help never change a score.\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-10",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Best of",
        "url": "https://www.anchorterminal.com/best/"
      },
      {
        "name": "Guardrails \u0026 safety filters",
        "url": ""
      }
    ],
    "description": "Google Cloud Model Armor (BB), Amazon Bedrock Guardrails (BB) and OpenAI Moderation API (BB) lead the 16 ranked guardrails and safety filters. Picks by need, strengths, weaknesses and prices from the Anchor benchmark.",
    "facts": [
      "Google Cloud Model Armor BB",
      "Amazon Bedrock Guardrails BB",
      "OpenAI Moderation API BB"
    ],
    "h1": "Best guardrails and safety filters for AI agents",
    "image": "https://www.anchorterminal.com/assets/og/best-guardrails.png",
    "path": "/best/guardrails/",
    "published": "",
    "section": "tools",
    "title": "Best guardrails and safety filters for AI agents in 2026, ranked",
    "toc": null,
    "updated": "2026-10-08",
    "url": "https://www.anchorterminal.com/best/guardrails/"
  },
  "tokens": {
    "markdown": 6150,
    "slim": 1530
  },
  "version": 1
}
