{
  "data": {
    "category": {
      "area": "voice",
      "capabilities": [
        "voice.agent",
        "voice.speech-to-speech",
        "voice.pipeline",
        "voice.tools",
        "voice.telephony"
      ],
      "description": "Platforms that run a spoken conversation for you, either with one speech-to-speech model or a configurable pipeline of speech-to-text, a language model and text-to-speech. Compared on response latency, handling interruptions, tool calls, recovery after a misunderstanding and cost per completed call.",
      "indexed": [
        {
          "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/topness-callwright.json",
          "kind": "mcp",
          "name": "callwright",
          "slug": "topness-callwright",
          "url": "https://www.anchorterminal.com/tools/topness-callwright"
        }
      ],
      "indexedCount": 1,
      "json": "https://www.anchorterminal.com/categories/voice-agents.json",
      "name": "Conversational voice agents",
      "slug": "voice-agents",
      "test": "The same booking and support tasks on every platform. We measure response latency, how interruptions are handled, tool-call success, recovery after a misunderstanding and the total cost per completed call, and record whether the product runs a direct speech-to-speech model or an STT, model and TTS pipeline, since those are different setups.",
      "title": "Conversational voice-agent APIs",
      "toolCount": 14,
      "tools": [
        "elevenlabs-agents",
        "retell-ai",
        "deepgram-voice-agent",
        "assemblyai-voice-agent",
        "bland-ai",
        "vapi",
        "openai-realtime",
        "ultravox",
        "gemini-live",
        "hume-evi",
        "vocily",
        "bolna",
        "synthflow",
        "vogent"
      ],
      "url": "https://www.anchorterminal.com/categories/voice-agents"
    },
    "faq": [
      {
        "answer": "ElevenLabs Agents API + MCP has the highest benchmark score of the 14 ranked conversational voice-agent APIs, 71.3 (BB). Retell AI API + MCP is second with 69.1 (B).",
        "question": "What are the highest-rated conversational voice-agent APIs for AI agents?"
      },
      {
        "answer": "1 of the 14 ranked here grade BB or better, the bar for agent-ready on the Anchor benchmark.",
        "question": "How many conversational voice-agent APIs are agent-ready?"
      },
      {
        "answer": "None of the ranked listings here accepts x402 for its main call yet.",
        "question": "Which conversational voice-agent APIs accept x402 payments?"
      },
      {
        "answer": "By the Anchor benchmark score out of 100, a weighted mean of the scored categories minus deductions for negative events, from public evidence re-checked as vendors change. Listings cannot pay for a place. The latest assessment behind this page is from 9 October 2026.",
        "question": "How is this list ranked?"
      }
    ],
    "howToChoose": [
      {
        "label": "Speech-to-speech or pipeline",
        "detail": "Check whether the product runs one speech-to-speech model or a separate recognition, language and synthesis chain, since each changes latency and what you can swap."
      },
      {
        "label": "Handling of interruptions",
        "detail": "Ask how the platform stops speaking when a caller interrupts, because a slow stop leaves the agent talking over the customer."
      },
      {
        "label": "Tool calls during a live call",
        "detail": "Check how a tool call is confirmed to the caller and what happens when it fails, since a long pause or a wrong booking is what the caller hears."
      },
      {
        "label": "Cost per completed call",
        "detail": "Work out the cost of a finished call from per-minute, language model and telephony charges together, because a low per-minute rate can hide a costly language model."
      }
    ],
    "picks": [
      {
        "also": {
          "name": "Retell AI API + MCP",
          "slug": "retell-ai",
          "why": "B, 69.1/100"
        },
        "name": "ElevenLabs Agents API + MCP",
        "need": "Highest score overall",
        "slug": "elevenlabs-agents",
        "why": "BB, 71.3/100 on the benchmark"
      },
      {
        "name": "Vapi API + MCP",
        "need": "Reliability",
        "slug": "vapi",
        "why": "70/100 on reliability, against 60 for the overall leader"
      },
      {
        "name": "Vocily AI",
        "need": "Agent ergonomics",
        "slug": "vocily",
        "why": "84/100 on agent ergonomics, against 77 for the overall leader"
      },
      {
        "name": "Vapi API + MCP",
        "need": "Maintenance \u0026 community",
        "slug": "vapi",
        "why": "88/100 on maintenance \u0026 community, against 85 for the overall leader"
      },
      {
        "name": "Bolna API + MCP",
        "need": "Self-hosting under an open licence",
        "slug": "bolna",
        "why": "self-hosted, MIT licence"
      }
    ],
    "ranked": 14,
    "shortlist": [
      {
        "bestFor": "Teams that want a managed pipeline with ElevenLabs voices, a choice of LLM, wide telephony integration and web and mobile SDKs.",
        "grade": "BB",
        "name": "ElevenLabs Agents API + MCP",
        "position": 1,
        "price": "$22 / mo",
        "score": 71.3,
        "slug": "elevenlabs-agents",
        "strengths": [
          "API keys scoped to endpoint groups, with per-key credit limits and service accounts",
          "Hosted MCP server that signs in with OAuth, with EU, India and Singapore endpoints",
          "Public OpenAPI, llms.txt and an errors page with codes, fixes and request IDs"
        ],
        "url": "https://www.anchorterminal.com/tools/elevenlabs-agents",
        "verdict": "API keys scoped to endpoint groups, with per-key credit limits and service accounts. $0.08 a minute excludes LLM tokens and carrier minutes.",
        "weaknesses": [
          "$0.08 a minute excludes LLM tokens and carrier minutes",
          "At least six incidents since July where agent calls failed or didn't start",
          "Conversation data kept 2 years by default"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Teams that want a self-serve phone agent with clear per-component pricing, scoped keys and a choice of LLM and voice from Retell's menu.",
        "grade": "B",
        "name": "Retell AI API + MCP",
        "position": 2,
        "price": "$2 / mo",
        "score": 69.1,
        "slug": "retell-ai",
        "strengths": [
          "Published per-minute price for every component, from $0.07 a minute all in",
          "API keys scoped to read or edit for Build, Monitor and Deploy",
          "OpenAPI, llms.txt and a deprecation feed with RSS"
        ],
        "url": "https://www.anchorterminal.com/tools/retell-ai",
        "verdict": "Published per-minute price for every component, from $0.07 a minute all in. Call data kept indefinitely unless retention is set per agent.",
        "weaknesses": [
          "Call data kept indefinitely unless retention is set per agent",
          "MCP tools carry no read-only or destructive annotations",
          "Web and phone calls disrupted for 69 minutes on 5 September 2026"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Developers who already run telephony (Twilio, Amazon Connect, Genesys, AudioCodes) and want a cheap managed pipeline with their choice of LLM.",
        "grade": "B",
        "name": "Deepgram Voice Agent API",
        "position": 3,
        "price": "Pay per use",
        "score": 68.1,
        "slug": "deepgram-voice-agent",
        "strengths": [
          "$0.075 a minute all in on Standard, $0.050 with your own LLM and TTS, $200 of credit with no card",
          "AsyncAPI description of the agent socket plus a public OpenAPI file",
          "Role-based API keys with expiry and 30-second browser tokens"
        ],
        "url": "https://www.anchorterminal.com/tools/deepgram-voice-agent",
        "verdict": "$0.075 a minute all in on Standard, $0.050 with your own LLM and TTS, $200 of credit with no card. No phone numbers or SIP, so you bridge Twilio or another carrier yourself.",
        "weaknesses": [
          "No phone numbers or SIP, so you bridge Twilio or another carrier yourself",
          "Fifteen status incidents since July, four of them an hour or longer on parts the agent uses",
          "Call audio is kept for model improvement unless `mip_opt_out` is set"
        ],
        "where": "both",
        "x402": "no"
      },
      {
        "bestFor": "A developer who wants one managed pipeline at a flat rate and is content with AssemblyAI's own speech models and voices, with an optional model of their own.",
        "grade": "B",
        "name": "AssemblyAI Voice Agent API",
        "position": 4,
        "price": "Pay per use",
        "score": 64.4,
        "slug": "assemblyai-voice-agent",
        "strengths": [
          "One published rate of $4.50 an hour ($0.075 a minute), billed per second, with $50 of credit and no card for new accounts",
          "AsyncAPI 3.0 file for the socket, an OpenAPI file for the token endpoint, llms.txt and a Markdown copy of every docs page",
          "Session errors carry a code, a message and sometimes the offending field, and the docs name the three codes that are safe to retry"
        ],
        "url": "https://www.anchorterminal.com/tools/assemblyai-voice-agent",
        "verdict": "One $0.075 a minute rate covers speech-to-text, the managed model, speech output, recordings and hosting, and the socket has an AsyncAPI file, typed error codes and a 30-second resume window. No concurrent-session limit is published, the status page has no Voice Agent component, and phone calls are inbound only through a Twilio trunk the customer owns.",
        "weaknesses": [
          "No number is published for the concurrent-session limit that the `concurrency_exceeded` error enforces",
          "status.assemblyai.com lists the Asynchronous API, Streaming API and LLM Gateway, with no Voice Agent component",
          "Telephony is inbound only, over a SIP trunk on a Twilio account the customer owns"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Teams that want one predictable per-minute bill for outbound and inbound phone agents, and personal agents that need their own US number.",
        "grade": "B",
        "name": "Bland AI API + MCP",
        "position": 5,
        "price": "$299 / mo",
        "score": 63.8,
        "slug": "bland-ai",
        "strengths": [
          "One per-minute rate covering STT, LLM, TTS and telephony, $0.14 on Start and $0.12 on Build",
          "Start plan with 2 credits and an inbound number, no card",
          "Hosted MCP server with 42 tools labelled read, write or destructive, with confirmation on destructive ones"
        ],
        "url": "https://www.anchorterminal.com/tools/bland-ai",
        "verdict": "One per-minute rate covering STT, LLM, TTS and telephony, $0.14 on Start and $0.12 on Build. No OpenAPI document.",
        "weaknesses": [
          "No OpenAPI document",
          "Start caps at 10 concurrent calls and 100 calls a day",
          "Failed calls and outbound attempts cost $0.015 each"
        ],
        "where": "both",
        "x402": "no"
      },
      {
        "bestFor": "Developers who want to choose every provider, bring their own keys and tune cost against latency.",
        "grade": "B",
        "name": "Vapi API + MCP",
        "position": 6,
        "price": "$29 / mo",
        "score": 63.5,
        "slug": "vapi",
        "strengths": [
          "Any mix of transcriber, model and voice provider, or your own keys and endpoints",
          "OpenAPI file, llms.txt and a weekly changelog",
          "Public keys limited to allowed origins and assistants"
        ],
        "url": "https://www.anchorterminal.com/tools/vapi",
        "verdict": "Voice agents can combine supported transcription, model and voice providers, including customer-supplied endpoints. Per-minute cost depends on the selected providers.",
        "weaknesses": [
          "Real per-minute cost depends on provider choices",
          "Usage only has 4 concurrent lines and 14 days of retention",
          "MCP tools that place calls or buy numbers carry no annotations"
        ],
        "where": "both",
        "x402": "no"
      },
      {
        "bestFor": "Teams building their own voice agent on one speech-to-speech model who want browser, server and phone transports from one vendor and can run a backend for secrets and tools.",
        "grade": "B",
        "name": "OpenAI Realtime API",
        "position": 7,
        "price": "Pay per use",
        "score": 63.3,
        "slug": "openai-realtime",
        "strengths": [
          "The OpenAPI 3.1 file in `openai/openai-openapi` covers nine Realtime paths and types 13 client events and 48 server events as schemas.",
          "Browser clients use client secrets that expire after 10 seconds to 2 hours, 10 minutes by default, so the API key stays on a server.",
          "Remote MCP tools accept an `allowed_tools` list, and `require_approval` defaults to `always` in the OpenAPI file."
        ],
        "url": "https://www.anchorterminal.com/tools/openai-realtime",
        "verdict": "A generally available speech-to-speech API with WebRTC, WebSocket and SIP transports, a public OpenAPI file that types its events, and per-token prices. Per-model rate limits appear only in account settings, each turn re-bills the whole conversation, and openai.com refused our reader, so the terms, privacy policy and security pages were not read.",
        "weaknesses": [
          "openai.com answered 403 to our reader, so the service terms, privacy policy, sub-processor list and security pages were not read.",
          "Realtime rate limits are shown only in account settings. The public table gives numbers for GPT-6 model families only.",
          "Every response re-sends the whole conversation, so later turns cost more. Audio is $32 in and $64 out per 1M tokens on `gpt-realtime-2.1`."
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Cost-sensitive voice agents that bring their own telephony and want a speech-native model.",
        "grade": "C",
        "name": "Ultravox Realtime API",
        "position": 8,
        "price": "$100 / mo",
        "score": 58.5,
        "slug": "ultravox",
        "strengths": [
          "Flat $0.05 a minute including model and built-in voices",
          "Speech-native model with no separate STT stage, v0.7 weights public under MIT",
          "429 and 503 responses with Retry-After and backoff guidance"
        ],
        "url": "https://www.anchorterminal.com/tools/ultravox",
        "verdict": "Flat $0.05 a minute including model and built-in voices. No MCP server.",
        "weaknesses": [
          "No MCP server",
          "News page, deprecation guide and Python client last updated in December 2025",
          "Plain API keys with no scopes"
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Teams building their own voice or vision assistant on a single speech-to-speech model with function calling and Search grounding, who can run a backend for tokens and audio transport.",
        "grade": "C",
        "name": "Gemini Live API",
        "position": 9,
        "price": "$14 / 1k req",
        "score": 57.1,
        "slug": "gemini-live",
        "strengths": [
          "`gemini-3.8-live` went generally available on 15 September 2026 with asynchronous function calling as the default and three response scheduling modes.",
          "Ephemeral tokens can be single use, expire in 30 minutes by default and be locked to a model and session configuration.",
          "Prices are public per million tokens with per-minute equivalents, $0.005 a minute of audio in and $0.018 a minute of audio out."
        ],
        "url": "https://www.anchorterminal.com/tools/gemini-live",
        "verdict": "A speech-to-speech API with per-token prices published, a free tier and a generally available model, `gemini-3.8-live`, since 15 September 2026. The documented raw WebSocket connection carries the API key in the URL, connections reset about every 10 minutes, and no SLA or readable incident history was found for the Developer API.",
        "weaknesses": [
          "The WebSocket guide authenticates with the API key as a `key` query parameter, and ephemeral tokens as an `access_token` query parameter.",
          "The status page at aistudio.google.com/status renders in the browser only, so no incident history could be read, and no SLA was found for the Developer API.",
          "No machine-readable contract for the WebSocket messages was found. The Discovery document types only the setup and token schemas."
        ],
        "where": "hosted",
        "x402": "no"
      },
      {
        "bestFor": "Consumer and coaching products where the caller's tone matters and English is enough on EVI 3.",
        "grade": "C",
        "name": "Hume EVI (Empathic Voice Interface)",
        "position": 10,
        "price": "$14 / mo",
        "score": 56.7,
        "slug": "hume-evi",
        "strengths": [
          "Speech-language model that reads the caller's tone and answers with matching prosody",
          "4 to 7 cents a minute including Hume's own model",
          "External LLMs or a custom endpoint for tool calling"
        ],
        "url": "https://www.anchorterminal.com/tools/hume-evi",
        "verdict": "Speech-language model that reads the caller's tone and answers with matching prosody. One account-wide key, sent in the WebSocket and Twilio webhook URLs. Hume ends API access on 13 November 2026.",
        "weaknesses": [
          "Hume ends access to the TTS and EVI APIs on 13 November 2026 and deletes account data after that date",
          "One account-wide key, sent in the WebSocket and Twilio webhook URLs",
          "Privacy page contradicts itself on training use of EVI data"
        ],
        "where": "hosted",
        "x402": "no"
      }
    ],
    "updated": "2026-10-09"
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/best/voice-agents/",
    "json": "https://www.anchorterminal.com/best/voice-agents/index.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/best/voice-agents/index.md",
    "slim": "https://www.anchorterminal.com/best/voice-agents/index.min.md"
  },
  "markdown": "The 10 highest-scoring of 14 conversational voice-agent APIs on the Anchor benchmark, with a pick for each need and where each one falls short. Scores come from public evidence, re-checked as vendors change.\n\n- Ranked: 14 · agent-ready (BB or better): 1 · accept x402: 0 · hosted endpoints: 14\n- Full ranked table: https://www.anchorterminal.com/categories/voice-agents.md\n- Head-to-head comparisons: https://www.anchorterminal.com/compare/voice-agents/index.md (91)\n- Methodology: https://www.anchorterminal.com/benchmark/index.md\n\n## The shortlist\n\n| # | Tool | Grade | Score | Best for | Price | Where |\n| --- | --- | --- | --- | --- | --- | --- |\n| 1 | [ElevenLabs Agents API + MCP](https://www.anchorterminal.com/tools/elevenlabs-agents.md) | BB | 71.3 | Teams that want a managed pipeline with ElevenLabs voices, a choice of LLM, wide telephony integration and web and mobile SDKs. | $22 / mo | hosted |\n| 2 | [Retell AI API + MCP](https://www.anchorterminal.com/tools/retell-ai.md) | B | 69.1 | Teams that want a self-serve phone agent with clear per-component pricing, scoped keys and a choice of LLM and voice from Retell's menu. | $2 / mo | hosted |\n| 3 | [Deepgram Voice Agent API](https://www.anchorterminal.com/tools/deepgram-voice-agent.md) | B | 68.1 | Developers who already run telephony (Twilio, Amazon Connect, Genesys, AudioCodes) and want a cheap managed pipeline with their choice of LLM. | Pay per use | hosted and local |\n| 4 | [AssemblyAI Voice Agent API](https://www.anchorterminal.com/tools/assemblyai-voice-agent.md) | B | 64.4 | A developer who wants one managed pipeline at a flat rate and is content with AssemblyAI's own speech models and voices, with an optional model of their own. | Pay per use | hosted |\n| 5 | [Bland AI API + MCP](https://www.anchorterminal.com/tools/bland-ai.md) | B | 63.8 | Teams that want one predictable per-minute bill for outbound and inbound phone agents, and personal agents that need their own US number. | $299 / mo | hosted and local |\n| 6 | [Vapi API + MCP](https://www.anchorterminal.com/tools/vapi.md) | B | 63.5 | Developers who want to choose every provider, bring their own keys and tune cost against latency. | $29 / mo | hosted and local |\n| 7 | [OpenAI Realtime API](https://www.anchorterminal.com/tools/openai-realtime.md) | B | 63.3 | Teams building their own voice agent on one speech-to-speech model who want browser, server and phone transports from one vendor and can run a backend for secrets and tools. | Pay per use | hosted |\n| 8 | [Ultravox Realtime API](https://www.anchorterminal.com/tools/ultravox.md) | C | 58.5 | Cost-sensitive voice agents that bring their own telephony and want a speech-native model. | $100 / mo | hosted |\n| 9 | [Gemini Live API](https://www.anchorterminal.com/tools/gemini-live.md) | C | 57.1 | Teams building their own voice or vision assistant on a single speech-to-speech model with function calling and Search grounding, who can run a backend for tokens and audio transport. | $14 / 1k req | hosted |\n| 10 | [Hume EVI (Empathic Voice Interface)](https://www.anchorterminal.com/tools/hume-evi.md) | C | 56.7 | Consumer and coaching products where the caller's tone matters and English is enough on EVI 3. | $14 / mo | hosted |\n\n## Picks by need\n\n- Highest score overall: [ElevenLabs Agents API + MCP](https://www.anchorterminal.com/tools/elevenlabs-agents.md), BB, 71.3/100 on the benchmark. Also [Retell AI API + MCP](https://www.anchorterminal.com/tools/retell-ai.md), B, 69.1/100.\n- Reliability: [Vapi API + MCP](https://www.anchorterminal.com/tools/vapi.md), 70/100 on reliability, against 60 for the overall leader.\n- Agent ergonomics: [Vocily AI](https://www.anchorterminal.com/tools/vocily.md), 84/100 on agent ergonomics, against 77 for the overall leader.\n- Maintenance \u0026 community: [Vapi API + MCP](https://www.anchorterminal.com/tools/vapi.md), 88/100 on maintenance \u0026 community, against 85 for the overall leader.\n- Self-hosting under an open licence: [Bolna API + MCP](https://www.anchorterminal.com/tools/bolna.md), self-hosted, MIT licence.\n\n## How to choose\n\n- Speech-to-speech or pipeline: Check whether the product runs one speech-to-speech model or a separate recognition, language and synthesis chain, since each changes latency and what you can swap.\n- Handling of interruptions: Ask how the platform stops speaking when a caller interrupts, because a slow stop leaves the agent talking over the customer.\n- Tool calls during a live call: Check how a tool call is confirmed to the caller and what happens when it fails, since a long pause or a wrong booking is what the caller hears.\n- Cost per completed call: Work out the cost of a finished call from per-minute, language model and telephony charges together, because a low per-minute rate can hide a costly language model.\n\n- How the benchmark tests this category: The same booking and support tasks on every platform. We measure response latency, how interruptions are handled, tool-call success, recovery after a misunderstanding and the total cost per completed call, and record whether the product runs a direct speech-to-speech model or an STT, model and TTS pipeline, since those are different setups.\n\n## Each one in detail\n\n### 1. ElevenLabs Agents API + MCP, BB 71.3/100\n\nElevenAgents (formerly Conversational AI) runs hosted voice agents as a pipeline of a fine-tuned ElevenLabs ASR model, an LLM of your choice or your own, ElevenLabs TTS and a proprietary turn-taking model.\n\n- Verdict: API keys scoped to endpoint groups, with per-key credit limits and service accounts. $0.08 a minute excludes LLM tokens and carrier minutes.\n- Choose it for: Teams that want a managed pipeline with ElevenLabs voices, a choice of LLM, wide telephony integration and web and mobile SDKs.\n- Strength: API keys scoped to endpoint groups, with per-key credit limits and service accounts\n- Strength: Hosted MCP server that signs in with OAuth, with EU, India and Singapore endpoints\n- Strength: Public OpenAPI, llms.txt and an errors page with codes, fixes and request IDs\n- Weakness: $0.08 a minute excludes LLM tokens and carrier minutes\n- Weakness: At least six incidents since July where agent calls failed or didn't start\n- Weakness: Conversation data kept 2 years by default\n- Price: $22 / mo · Auth: OAuth or key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/elevenlabs-agents.md\n\n### 2. Retell AI API + MCP, B 69.1/100\n\nHosted platform for phone and web voice agents, built as single-prompt agents or node-based conversation flows.\n\n- Verdict: Published per-minute price for every component, from $0.07 a minute all in. Call data kept indefinitely unless retention is set per agent.\n- Choose it for: Teams that want a self-serve phone agent with clear per-component pricing, scoped keys and a choice of LLM and voice from Retell's menu.\n- Strength: Published per-minute price for every component, from $0.07 a minute all in\n- Strength: API keys scoped to read or edit for Build, Monitor and Deploy\n- Strength: OpenAPI, llms.txt and a deprecation feed with RSS\n- Weakness: Call data kept indefinitely unless retention is set per agent\n- Weakness: MCP tools carry no read-only or destructive annotations\n- Weakness: Web and phone calls disrupted for 69 minutes on 5 September 2026\n- Price: $2 / mo · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/retell-ai.md\n- Against #1: https://www.anchorterminal.com/compare/elevenlabs-agents-vs-retell-ai.md\n\n### 3. Deepgram Voice Agent API, B 68.1/100\n\nOne WebSocket that runs Deepgram STT (Flux or Nova-3), a managed or bring-your-own LLM and Deepgram or third-party TTS, with turn-taking, barge-in and function calling.\n\n- Verdict: $0.075 a minute all in on Standard, $0.050 with your own LLM and TTS, $200 of credit with no card. No phone numbers or SIP, so you bridge Twilio or another carrier yourself.\n- Choose it for: Developers who already run telephony (Twilio, Amazon Connect, Genesys, AudioCodes) and want a cheap managed pipeline with their choice of LLM.\n- Strength: $0.075 a minute all in on Standard, $0.050 with your own LLM and TTS, $200 of credit with no card\n- Strength: AsyncAPI description of the agent socket plus a public OpenAPI file\n- Strength: Role-based API keys with expiry and 30-second browser tokens\n- Weakness: No phone numbers or SIP, so you bridge Twilio or another carrier yourself\n- Weakness: Fifteen status incidents since July, four of them an hour or longer on parts the agent uses\n- Weakness: Call audio is kept for model improvement unless `mip_opt_out` is set\n- Price: Pay per use · Auth: API key · x402: no · Where: hosted and local\n- Full assessment: https://www.anchorterminal.com/tools/deepgram-voice-agent.md\n- Against #1: https://www.anchorterminal.com/compare/deepgram-voice-agent-vs-elevenlabs-agents.md\n\n### 4. AssemblyAI Voice Agent API, B 64.4/100\n\nAssemblyAI's Voice Agent API runs a spoken conversation over one WebSocket, combining its own speech-to-text, language model and text-to-speech. Agents are stored through a REST API and reached from a browser, a server or an inbound Twilio SIP number.\n\n- Verdict: One $0.075 a minute rate covers speech-to-text, the managed model, speech output, recordings and hosting, and the socket has an AsyncAPI file, typed error codes and a 30-second resume window. No concurrent-session limit is published, the status page has no Voice Agent component, and phone calls are inbound only through a Twilio trunk the customer owns.\n- Choose it for: A developer who wants one managed pipeline at a flat rate and is content with AssemblyAI's own speech models and voices, with an optional model of their own.\n- Strength: One published rate of $4.50 an hour ($0.075 a minute), billed per second, with $50 of credit and no card for new accounts\n- Strength: AsyncAPI 3.0 file for the socket, an OpenAPI file for the token endpoint, llms.txt and a Markdown copy of every docs page\n- Strength: Session errors carry a code, a message and sometimes the offending field, and the docs name the three codes that are safe to retry\n- Weakness: No number is published for the concurrent-session limit that the `concurrency_exceeded` error enforces\n- Weakness: status.assemblyai.com lists the Asynchronous API, Streaming API and LLM Gateway, with no Voice Agent component\n- Weakness: Telephony is inbound only, over a SIP trunk on a Twilio account the customer owns\n- Price: Pay per use · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/assemblyai-voice-agent.md\n- Against #1: https://www.anchorterminal.com/compare/assemblyai-voice-agent-vs-elevenlabs-agents.md\n\n### 5. Bland AI API + MCP, B 63.8/100\n\nPhone-first voice-agent platform that runs its own speech recognition, language model, TTS and telephony, billed as one per-minute rate.\n\n- Verdict: One per-minute rate covering STT, LLM, TTS and telephony, $0.14 on Start and $0.12 on Build. No OpenAPI document.\n- Choose it for: Teams that want one predictable per-minute bill for outbound and inbound phone agents, and personal agents that need their own US number.\n- Strength: One per-minute rate covering STT, LLM, TTS and telephony, $0.14 on Start and $0.12 on Build\n- Strength: Start plan with 2 credits and an inbound number, no card\n- Strength: Hosted MCP server with 42 tools labelled read, write or destructive, with confirmation on destructive ones\n- Weakness: No OpenAPI document\n- Weakness: Start caps at 10 concurrent calls and 100 calls a day\n- Weakness: Failed calls and outbound attempts cost $0.015 each\n- Price: $299 / mo · Auth: API key · x402: no · Where: hosted and local\n- Full assessment: https://www.anchorterminal.com/tools/bland-ai.md\n- Against #1: https://www.anchorterminal.com/compare/bland-ai-vs-elevenlabs-agents.md\n\n### 6. Vapi API + MCP, B 63.5/100\n\nDeveloper platform for phone and web voice agents.\n\n- Verdict: Voice agents can combine supported transcription, model and voice providers, including customer-supplied endpoints. Per-minute cost depends on the selected providers.\n- Choose it for: Developers who want to choose every provider, bring their own keys and tune cost against latency.\n- Strength: Any mix of transcriber, model and voice provider, or your own keys and endpoints\n- Strength: OpenAPI file, llms.txt and a weekly changelog\n- Strength: Public keys limited to allowed origins and assistants\n- Weakness: Real per-minute cost depends on provider choices\n- Weakness: Usage only has 4 concurrent lines and 14 days of retention\n- Weakness: MCP tools that place calls or buy numbers carry no annotations\n- Price: $29 / mo · Auth: API key · x402: no · Where: hosted and local\n- Full assessment: https://www.anchorterminal.com/tools/vapi.md\n- Against #1: https://www.anchorterminal.com/compare/elevenlabs-agents-vs-vapi.md\n\n### 7. OpenAI Realtime API, B 63.3/100\n\nOpenAI's Realtime API runs spoken conversations with speech-to-speech models such as `gpt-realtime-2.1`. Clients connect over WebRTC, WebSocket or SIP, and sessions support interruptions, function calling and remote MCP tools.\n\n- Verdict: A generally available speech-to-speech API with WebRTC, WebSocket and SIP transports, a public OpenAPI file that types its events, and per-token prices. Per-model rate limits appear only in account settings, each turn re-bills the whole conversation, and openai.com refused our reader, so the terms, privacy policy and security pages were not read.\n- Choose it for: Teams building their own voice agent on one speech-to-speech model who want browser, server and phone transports from one vendor and can run a backend for secrets and tools.\n- Strength: The OpenAPI 3.1 file in `openai/openai-openapi` covers nine Realtime paths and types 13 client events and 48 server events as schemas.\n- Strength: Browser clients use client secrets that expire after 10 seconds to 2 hours, 10 minutes by default, so the API key stays on a server.\n- Strength: Remote MCP tools accept an `allowed_tools` list, and `require_approval` defaults to `always` in the OpenAPI file.\n- Weakness: openai.com answered 403 to our reader, so the service terms, privacy policy, sub-processor list and security pages were not read.\n- Weakness: Realtime rate limits are shown only in account settings. The public table gives numbers for GPT-6 model families only.\n- Weakness: Every response re-sends the whole conversation, so later turns cost more. Audio is $32 in and $64 out per 1M tokens on `gpt-realtime-2.1`.\n- Price: Pay per use · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/openai-realtime.md\n- Against #1: https://www.anchorterminal.com/compare/elevenlabs-agents-vs-openai-realtime.md\n\n### 8. Ultravox Realtime API, C 58.5/100\n\nHosted voice agents on Ultravox, an open-weight model that takes speech directly with no speech-to-text step and answers through a TTS voice.\n\n- Verdict: Flat $0.05 a minute including model and built-in voices. No MCP server.\n- Choose it for: Cost-sensitive voice agents that bring their own telephony and want a speech-native model.\n- Strength: Flat $0.05 a minute including model and built-in voices\n- Strength: Speech-native model with no separate STT stage, v0.7 weights public under MIT\n- Strength: 429 and 503 responses with Retry-After and backoff guidance\n- Weakness: No MCP server\n- Weakness: News page, deprecation guide and Python client last updated in December 2025\n- Weakness: Plain API keys with no scopes\n- Price: $100 / mo · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/ultravox.md\n- Against #1: https://www.anchorterminal.com/compare/elevenlabs-agents-vs-ultravox.md\n\n### 9. Gemini Live API, C 57.1/100\n\nGoogle's Live API runs real-time spoken conversations with Gemini audio-to-audio models over a stateful WebSocket. It takes audio, images and text, speaks back, and supports interruptions, function calling and Google Search grounding.\n\n- Verdict: A speech-to-speech API with per-token prices published, a free tier and a generally available model, `gemini-3.8-live`, since 15 September 2026. The documented raw WebSocket connection carries the API key in the URL, connections reset about every 10 minutes, and no SLA or readable incident history was found for the Developer API.\n- Choose it for: Teams building their own voice or vision assistant on a single speech-to-speech model with function calling and Search grounding, who can run a backend for tokens and audio transport.\n- Strength: `gemini-3.8-live` went generally available on 15 September 2026 with asynchronous function calling as the default and three response scheduling modes.\n- Strength: Ephemeral tokens can be single use, expire in 30 minutes by default and be locked to a model and session configuration.\n- Strength: Prices are public per million tokens with per-minute equivalents, $0.005 a minute of audio in and $0.018 a minute of audio out.\n- Weakness: The WebSocket guide authenticates with the API key as a `key` query parameter, and ephemeral tokens as an `access_token` query parameter.\n- Weakness: The status page at aistudio.google.com/status renders in the browser only, so no incident history could be read, and no SLA was found for the Developer API.\n- Weakness: No machine-readable contract for the WebSocket messages was found. The Discovery document types only the setup and token schemas.\n- Price: $14 / 1k req · Auth: API key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/gemini-live.md\n- Against #1: https://www.anchorterminal.com/compare/elevenlabs-agents-vs-gemini-live.md\n\n### 10. Hume EVI (Empathic Voice Interface), C 56.7/100\n\nHume's hosted voice-agent service, accessed through a WebSocket API.\n\n- Verdict: Speech-language model that reads the caller's tone and answers with matching prosody. One account-wide key, sent in the WebSocket and Twilio webhook URLs. Hume ends API access on 13 November 2026.\n- Choose it for: Consumer and coaching products where the caller's tone matters and English is enough on EVI 3.\n- Strength: Speech-language model that reads the caller's tone and answers with matching prosody\n- Strength: 4 to 7 cents a minute including Hume's own model\n- Strength: External LLMs or a custom endpoint for tool calling\n- Weakness: Hume ends access to the TTS and EVI APIs on 13 November 2026 and deletes account data after that date\n- Weakness: One account-wide key, sent in the WebSocket and Twilio webhook URLs\n- Weakness: Privacy page contradicts itself on training use of EVI data\n- Price: $14 / mo · Auth: OAuth or key · x402: no · Where: hosted\n- Full assessment: https://www.anchorterminal.com/tools/hume-evi.md\n- Against #1: https://www.anchorterminal.com/compare/elevenlabs-agents-vs-hume-evi.md\n\n4 more are ranked in the full table: https://www.anchorterminal.com/categories/voice-agents.md\n\n## Head to head\n\n- [ElevenLabs Agents API + MCP vs Retell AI API + MCP](https://www.anchorterminal.com/compare/elevenlabs-agents-vs-retell-ai.md)\n- [Deepgram Voice Agent API vs ElevenLabs Agents API + MCP](https://www.anchorterminal.com/compare/deepgram-voice-agent-vs-elevenlabs-agents.md)\n- [AssemblyAI Voice Agent API vs ElevenLabs Agents API + MCP](https://www.anchorterminal.com/compare/assemblyai-voice-agent-vs-elevenlabs-agents.md)\n- [Bland AI API + MCP vs ElevenLabs Agents API + MCP](https://www.anchorterminal.com/compare/bland-ai-vs-elevenlabs-agents.md)\n- [Deepgram Voice Agent API vs Retell AI API + MCP](https://www.anchorterminal.com/compare/deepgram-voice-agent-vs-retell-ai.md)\n- [AssemblyAI Voice Agent API vs Retell AI API + MCP](https://www.anchorterminal.com/compare/assemblyai-voice-agent-vs-retell-ai.md)\n- [Bland AI API + MCP vs Retell AI API + MCP](https://www.anchorterminal.com/compare/bland-ai-vs-retell-ai.md)\n- [AssemblyAI Voice Agent API vs Deepgram Voice Agent API](https://www.anchorterminal.com/compare/assemblyai-voice-agent-vs-deepgram-voice-agent.md)\n- [Bland AI API + MCP vs Deepgram Voice Agent API](https://www.anchorterminal.com/compare/bland-ai-vs-deepgram-voice-agent.md)\n- [AssemblyAI Voice Agent API vs Bland AI API + MCP](https://www.anchorterminal.com/compare/assemblyai-voice-agent-vs-bland-ai.md)\n\n## Questions\n\n### What are the highest-rated conversational voice-agent APIs for AI agents?\n\nElevenLabs Agents API + MCP has the highest benchmark score of the 14 ranked conversational voice-agent APIs, 71.3 (BB). Retell AI API + MCP is second with 69.1 (B).\n\n### How many conversational voice-agent APIs are agent-ready?\n\n1 of the 14 ranked here grade BB or better, the bar for agent-ready on the Anchor benchmark.\n\n### Which conversational voice-agent APIs accept x402 payments?\n\nNone of the ranked listings here accepts x402 for its main call yet.\n\n### How is this list ranked?\n\nBy the Anchor benchmark score out of 100, a weighted mean of the scored categories minus deductions for negative events, from public evidence re-checked as vendors change. Listings cannot pay for a place. The latest assessment behind this page is from 9 October 2026.\n\n## How this list is made\n\nThe order is the Anchor benchmark score, the same number as on each listing. Each listing is graded from public evidence against the benchmark checklist, and the picks are worked out from those grades, prices and facts. No listing pays for its place, and paid audits or listing help never change a score.\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-10",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Best of",
        "url": "https://www.anchorterminal.com/best/"
      },
      {
        "name": "Conversational voice agents",
        "url": ""
      }
    ],
    "description": "ElevenLabs Agents API + MCP (BB), Retell AI API + MCP (B) and Deepgram Voice Agent API (B) lead the 14 ranked conversational voice-agent APIs. Picks by need, strengths, weaknesses and prices from the Anchor benchmark.",
    "facts": [
      "ElevenLabs Agents API + MCP BB",
      "Retell AI API + MCP B",
      "Deepgram Voice Agent API B"
    ],
    "h1": "Best conversational voice-agent APIs",
    "image": "https://www.anchorterminal.com/assets/og/best-voice-agents.png",
    "path": "/best/voice-agents/",
    "published": "",
    "section": "tools",
    "title": "Best conversational voice-agent APIs in 2026, ranked | Anchor Terminal",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/best/voice-agents/"
  },
  "tokens": {
    "markdown": 5550,
    "slim": 1430
  },
  "version": 1
}
