{
  "data": {
    "a": {
      "slug": "azure-text-to-speech",
      "name": "Azure AI Speech text-to-speech",
      "vendor": "Microsoft",
      "vendorUrl": "https://azure.microsoft.com/en-us/products/ai-services/text-to-speech",
      "kind": "model",
      "category": "text-to-speech",
      "summary": "Azure's text-to-speech service for generating spoken audio.",
      "url": "https://www.anchorterminal.com/tools/azure-text-to-speech",
      "markdownUrl": "https://www.anchorterminal.com/tools/azure-text-to-speech.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/azure-text-to-speech.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/azure-text-to-speech.json",
      "repo": "https://github.com/Azure-Samples/cognitive-services-speech-sdk",
      "license": "MIT (samples), SDK under Microsoft's own licence",
      "transports": [
        "http"
      ],
      "remoteUrl": "https://eastus.tts.speech.microsoft.com/cognitiveservices",
      "packages": [
        {
          "registry": "pypi",
          "name": "azure-cognitiveservices-speech"
        },
        {
          "registry": "npm",
          "name": "microsoft-cognitiveservices-speech-sdk"
        }
      ],
      "auth": "mixed",
      "authNotes": "`Ocp-Apim-Subscription-Key` header with a Speech resource key, or a Microsoft Entra ID bearer token. Endpoints are per region.",
      "pricing": "freemium",
      "pricingNotes": "Free F0 tier with 500,000 characters a month. Pay as you go in East US is $15 per 1M characters for Neural and Neural HD Flash voices and $22 for Neural HD, real time or batch. Commitment tiers from $960 a month for 80M characters (https://azure.microsoft.com/en-us/pricing/details/speech/). Microsoft AI lists MAI-Voice-2.1 at $22 and MAI-Voice-2.1-Flash at $15 per 1M characters in preview (https://microsoft.ai/models/mai-voice-2-1/).",
      "priceSummary": "$960 / mo",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No machine payment. Billing runs through a cloud account with a card or invoice.",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 3450,
        "npmWeekly": 475621,
        "pypiWeekly": 1032532,
        "asOf": "2026-09-30"
      },
      "docsUrl": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/text-to-speech",
      "capabilities": [
        "speech.tts",
        "speech.streaming",
        "speech.voices",
        "speech.ssml",
        "speech.languages"
      ],
      "tags": [
        "hosted",
        "freemium",
        "free-tier",
        "closed-source",
        "python",
        "typescript",
        "enterprise",
        "streaming",
        "batch",
        "async-jobs"
      ],
      "lastRelease": "2026-10-01",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 71.4,
        "grade": "BB",
        "agentReady": true,
        "rank": 125,
        "ranked": true,
        "rankOf": 961,
        "categoryRank": 4,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 75,
          "maintenance": 80,
          "payments": 20,
          "reliability": 80,
          "schema": 65,
          "security": 90,
          "transparency": 85
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-05"
        },
        "negative": 0,
        "verdict": "Real-time synthesis keeps neither the input text nor the output audio. An Azure subscription needs a card, even for the free F0 tier.",
        "bestFor": "Operators on Azure who need SSML control, many languages and voices, and enterprise access control.",
        "strengths": [
          "Real-time synthesis keeps neither the input text nor the output audio",
          "Full SSML with speaking styles, prosody, phonemes, lexicons and up to 50 voice or audio tags a request",
          "Microsoft Entra ID with role-based access, or two rotatable keys",
          "Covered by Microsoft's online services SLA",
          "S0 starts at 30 requests a second and can be raised to 1,000"
        ],
        "weaknesses": [
          "An Azure subscription needs a card, even for the free F0 tier",
          "No llms.txt and no OpenAPI file for text-to-speech found",
          "429s often reflect busy capacity for a voice in a region, which a quota increase doesn't fix",
          "The voice list comes back as one response per region with no paging documented",
          "MAI-Voice-2.1 and MAI-Voice-2.1-Flash, Microsoft's newest models, are public preview with no SLA"
        ],
        "agentNotes": [
          "Send SSML with `\u003cspeak\u003e` and `\u003cvoice\u003e`, and set `X-Microsoft-OutputFormat` and `User-Agent`.",
          "On 429 retry with backoff, and try the voice's home region or another region rather than asking for more quota.",
          "Keep each real-time request under 10 minutes of audio, or use batch synthesis.",
          "Use Entra ID tokens instead of resource keys where the agent runs inside Azure.",
          "Name a MAI voice with its model suffix, such as `en-US-Harper:MAI-Voice-2.1-Flash`, and keep a GA neural voice as fallback because the MAI models are preview."
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 2,
        "avgRating": 3.5,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "BB",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 71.4
          }
        ],
        "editorialScores": {
          "ergonomics": 75,
          "maintenance": 80,
          "payments": 20,
          "reliability": 80,
          "schema": 65,
          "security": 90,
          "transparency": 80
        },
        "provenanceScore": 90
      },
      "connect": {
        "install": "pip install azure-cognitiveservices-speech   # or: npm i microsoft-cognitiveservices-speech-sdk",
        "http": "curl -X POST \"https://eastus.tts.speech.microsoft.com/cognitiveservices/v1\" \\\n  -H \"Ocp-Apim-Subscription-Key: $AZURE_SPEECH_KEY\" -H \"Content-Type: application/ssml+xml\" \\\n  -H \"X-Microsoft-OutputFormat: audio-24khz-48kbitrate-mono-mp3\" -o speech.mp3 \\\n  -d '\u003cspeak version=\"1.0\" xml:lang=\"en-US\"\u003e\u003cvoice name=\"en-US-AvaMultilingualNeural\"\u003eYour table is booked for seven.\u003c/voice\u003e\u003c/speak\u003e'"
      },
      "letme": {
        "capability": "https://letme.dev/speech.tts",
        "tool": "https://letme.dev/azure-text-to-speech"
      },
      "sameCompany": [
        "azure-foundry-fine-tuning",
        "azure-ai-content-safety",
        "azure-speech-to-text",
        "microsoft-agent-framework",
        "microsoft-execution-containers",
        "microsoft-entra-agent-id",
        "azure-key-vault",
        "azure-document-intelligence",
        "azure-devops-mcp",
        "microsoft-learn-mcp",
        "playwright-mcp",
        "azure-mcp",
        "azure-maps",
        "azure-translator",
        "microsoft-graph-calendar",
        "azure-blob-storage",
        "onedrive-sharepoint",
        "microsoft-teams",
        "dynamics-365-sales",
        "power-automate",
        "foundry-local",
        "microsoft-decision-1",
        "microsoft-advertising-api",
        "microsoft-excel-graph",
        "outlook-mail-graph"
      ],
      "area": "voice",
      "unitPrices": [
        {
          "item": "Neural and Neural HD Flash voices",
          "unit": "1m-chars",
          "usd": 15,
          "note": "real time or batch, East US"
        },
        {
          "item": "Neural HD voices",
          "unit": "1m-chars",
          "usd": 22
        },
        {
          "item": "Commitment tier 80M characters",
          "unit": "month",
          "usd": 960,
          "note": "$12 per 1M overage"
        }
      ],
      "provenance": {
        "legalEntity": "Microsoft Corporation",
        "domain": "microsoft.com",
        "domainRegistered": "1991-05-02",
        "domainNote": "Endpoints are on speech.microsoft.com, api.cognitive.microsoft.com and cognitiveservices.azure.com. microsoft.com publishes a security.txt, but it passed its Expires date on 2026-09-23.",
        "endpointOnVendorDomain": true,
        "terms": "https://www.microsoft.com/licensing/terms/",
        "privacy": "https://privacy.microsoft.com/en-us/privacystatement",
        "statusPage": "https://azure.status.microsoft/en-us/status",
        "changelog": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/releasenotes",
        "securityTxt": "expired",
        "checked": "2026-09-30",
        "score": 90
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/azure-text-to-speech.json",
      "live": {
        "slug": "azure-text-to-speech",
        "probe": {
          "target": "https://eastus.tts.speech.microsoft.com/cognitiveservices",
          "method": "get",
          "lastAt": "2026-10-11T02:46:17.774333384Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 257,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 257,
          "p95ms24h": 291,
          "samples24h": 248,
          "samples30d": 2712,
          "days": [
            {
              "date": "2026-09-30",
              "probes": 35,
              "ok": 35
            },
            {
              "date": "2026-10-01",
              "probes": 276,
              "ok": 276
            },
            {
              "date": "2026-10-02",
              "probes": 248,
              "ok": 248
            },
            {
              "date": "2026-10-03",
              "probes": 271,
              "ok": 271
            },
            {
              "date": "2026-10-04",
              "probes": 272,
              "ok": 272
            },
            {
              "date": "2026-10-05",
              "probes": 272,
              "ok": 272
            },
            {
              "date": "2026-10-06",
              "probes": 272,
              "ok": 272
            },
            {
              "date": "2026-10-07",
              "probes": 272,
              "ok": 272
            },
            {
              "date": "2026-10-08",
              "probes": 268,
              "ok": 268
            },
            {
              "date": "2026-10-09",
              "probes": 250,
              "ok": 250
            },
            {
              "date": "2026-10-10",
              "probes": 247,
              "ok": 247
            },
            {
              "date": "2026-10-11",
              "probes": 29,
              "ok": 29
            }
          ]
        },
        "versions": [
          {
            "registry": "github",
            "name": "Azure-Samples/cognitive-services-speech-sdk",
            "version": "ingestion-v2.1.13",
            "released": "2026-07-10",
            "seenAt": "2026-10-10T17:35:57.932268315Z"
          },
          {
            "registry": "npm",
            "name": "microsoft-cognitiveservices-speech-sdk",
            "version": "1.52.0",
            "seenAt": "2026-10-10T17:35:57.848799787Z"
          },
          {
            "registry": "pypi",
            "name": "azure-cognitiveservices-speech",
            "version": "1.52.0",
            "released": "2026-09-28",
            "seenAt": "2026-10-10T17:35:57.729713176Z"
          }
        ],
        "githubStars": 3452,
        "npmWeekly": 466694,
        "pypiWeekly": 715917,
        "securityTxt": {
          "url": "https://microsoft.com/.well-known/security.txt",
          "state": "expired",
          "expires": "2026-09-23T16:00:00.000Z",
          "checkedAt": "2026-10-10T15:40:52.555008839Z"
        },
        "domain": {
          "domain": "microsoft.com",
          "registered": "1991-05-02",
          "source": "https://rdap.verisign.com/com/v1/domain/microsoft.com",
          "checkedAt": "2026-10-04T13:04:13.488857536Z"
        },
        "updatedAt": "2026-10-11T02:46:17.774333384Z"
      }
    },
    "answer": "Azure AI Speech text-to-speech and HeyGen Voice score within a point of each other on agent readiness, 71.4 (BB) and 71.3 (BB). HeyGen Voice leads on schema \u0026 documentation, agent ergonomics and payments \u0026 pricing.",
    "b": {
      "slug": "heygen-voice",
      "name": "HeyGen Voice",
      "vendor": "HeyGen",
      "vendorUrl": "https://www.heygen.com",
      "kind": "model",
      "category": "text-to-speech",
      "summary": "HeyGen Voice is HeyGen's in-house text-to-speech model, `heygen-voice-1`, which speaks text in a voice cloned from one recording (instant) or from 20 minutes or more (professional), through HeyGen's v3 API.",
      "url": "https://www.anchorterminal.com/tools/heygen-voice",
      "markdownUrl": "https://www.anchorterminal.com/tools/heygen-voice.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/heygen-voice.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/heygen-voice.json",
      "license": "Closed service under HeyGen's Terms of Service (HeyGen Technology, Inc., last updated 23 July 2026).",
      "transports": [
        "http"
      ],
      "remoteUrl": "https://api.heygen.com",
      "packages": [],
      "auth": "api-key",
      "authNotes": "An API key from the HeyGen API dashboard, sent in the X-Api-Key header. Keys can be restricted to read and write scopes, and the speech and voice endpoints need voices:write. The OpenAPI file also accepts an OAuth bearer token, which bills subscription credits. Stripe Projects can provision a key through the operator's Stripe account.",
      "pricing": "usage",
      "pricingNotes": "Instant-voice speech is $15 per million submitted characters, charged only on successful requests (OpenAPI description, checked 10 October 2026). Professional speech is 0.6 API credits per generated minute, and each professional voice needs a purchased slot. Instant voices are free to create during the preview. No free API allowance was found.",
      "priceSummary": "Pay per use",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the developer llms.txt, the OpenAPI file or the HeyGen Voice pages, checked 10 October 2026. Keys are funded by API credit or through Stripe Projects.",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": null,
        "npmWeekly": null,
        "pypiWeekly": null,
        "asOf": "2026-10-10"
      },
      "docsUrl": "https://developers.heygen.com/docs/models/heygen-voice",
      "llmsTxt": "https://developers.heygen.com/llms.txt",
      "openapi": "https://developers.heygen.com/openapi/external-api.json",
      "capabilities": [
        "speech.tts",
        "speech.streaming",
        "speech.languages",
        "voice.clone"
      ],
      "tags": [
        "hosted",
        "official",
        "usage-priced",
        "closed-source",
        "openapi",
        "llms-txt",
        "streaming",
        "status-page",
        "soc2",
        "security-txt",
        "cli",
        "agent-skills"
      ],
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 71.3,
        "grade": "BB",
        "agentReady": true,
        "rank": 131,
        "ranked": true,
        "rankOf": 961,
        "categoryRank": 5,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 83,
          "maintenance": 65,
          "payments": 30,
          "reliability": 75,
          "schema": 94,
          "security": 69,
          "transparency": 69
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-10"
        },
        "negative": 0,
        "verdict": "HeyGen Voice has a typed OpenAPI contract, streaming with character timestamps, and a published price of $15 per million characters for instant voices, with failed requests not charged. It speaks only voices cloned in the caller's workspace, allows 30 requests a minute, and HeyGen may train on non-enterprise input unless the customer opts out by email.",
        "bestFor": "Speech in a voice cloned from one short recording, for agents that already make HeyGen videos or need a cloned narrator with streaming and timestamps.",
        "strengths": [
          "OpenAPI 3.1 file with enums, ranges, `additionalProperties: false` and per-code errors for every speech and voice endpoint",
          "Streaming over Server-Sent Events, with optional character-level timestamps",
          "Instant-voice speech priced at $15 per million characters without a login, and failed or disconnected requests are not charged",
          "API keys can be restricted to read and write scopes, and the speech endpoints need voices:write",
          "Instant voices are created from one recording, usually ready within seconds, in 39 base languages"
        ],
        "weaknesses": [
          "Speaks only voices cloned in the caller's workspace. Stock and designed voices go through a separate endpoint and engines",
          "30 requests a minute per workspace member on the speech endpoints, on every plan",
          "Output is a 44.1 kHz mono WAV only, and text is capped at 5,000 characters a request",
          "HeyGen's privacy policy allows training on user input, with an opt-out by email. Enterprise data is excluded by default",
          "No official SDK, no free API allowance and no SLA in the terms"
        ],
        "agentNotes": [
          "Create a voice first with POST /v3/models/audio/voices and `mode`, then poll GET /v3/models/audio/voices/{voice_id} until `status` is ACTIVE before calling speech",
          "Send `expressiveness_boost` only for instant voices and `seed`, `speed`, `pitch_shift`, `pitch_variance` or `\u003cbreak\u003e` tags only for professional voices. Mixing them returns 400 invalid_parameter",
          "On the stream, decode and play each base64 WAV part in `part_index` order. Do not concatenate the bytes",
          "Send an Idempotency-Key when creating a voice, and retry 502, 503 and 504 on speech, which the OpenAPI file marks as safe",
          "Use POST /v3/voices/speech for stock or designed voices. It is a different engine and price"
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "BB",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 71.3
          }
        ],
        "editorialScores": {
          "ergonomics": 83,
          "maintenance": 65,
          "payments": 30,
          "reliability": 75,
          "schema": 94,
          "security": 69,
          "transparency": 52
        },
        "provenanceScore": 85
      },
      "connect": {
        "http": "curl -X POST \"https://api.heygen.com/v3/models/audio/tts\" \\\n  -H \"X-Api-Key: $HEYGEN_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\":\"heygen-voice-1\",\"voice_id\":\"\u003cVOICE_ID\u003e\",\"text\":\"Hello.\",\"language\":\"en\",\"expressiveness_boost\":0.8}'"
      },
      "letme": {
        "capability": "https://letme.dev/speech.tts",
        "tool": "https://letme.dev/heygen-voice"
      },
      "area": "voice",
      "unitPrices": [
        {
          "item": "Instant-voice speech",
          "unit": "1m-chars",
          "usd": 15,
          "note": "Submitted Unicode characters, including spaces. Successful requests only."
        },
        {
          "item": "Professional-voice speech",
          "unit": "audio-minute",
          "usd": 0.6,
          "note": "0.6 API credits per generated minute. The dollar figure assumes $1 a credit, from the OpenAPI example of 9/60 credit for $0.15."
        }
      ],
      "provenance": {
        "legalEntity": "HeyGen Technology, Inc.",
        "domain": "heygen.com",
        "domainRegistered": "",
        "endpointOnVendorDomain": true,
        "terms": "https://www.heygen.com/terms",
        "privacy": "https://www.heygen.com/privacy",
        "statusPage": "https://status.heygen.com",
        "changelog": "https://developers.heygen.com/changelog",
        "securityTxt": "valid",
        "checked": "2026-10-10",
        "notes": [
          "security.txt at www.heygen.com/.well-known/security.txt names security@heygen.com and expires on 31 December 2026.",
          "The privacy policy names HeyGen Technology Inc. in section 1 and HeyGen, Inc. as processor in section 15.",
          "The status page has components per domain (www, app, api) and none for voice.",
          "Domain registration date was not checked."
        ],
        "score": 85
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/heygen-voice.json",
      "live": {
        "slug": "heygen-voice",
        "probe": {
          "target": "https://api.heygen.com",
          "method": "get",
          "lastAt": "2026-10-11T02:46:27.941825924Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 308,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 301,
          "p95ms24h": 320,
          "samples24h": 30,
          "samples30d": 30,
          "days": [
            {
              "date": "2026-10-10",
              "probes": 1,
              "ok": 1
            },
            {
              "date": "2026-10-11",
              "probes": 29,
              "ok": 29
            }
          ]
        },
        "vendorStatus": {
          "page": "https://status.heygen.com",
          "indicator": "none",
          "summary": "All Systems Operational",
          "checkedAt": "2026-10-11T02:40:57.349304573Z"
        },
        "updatedAt": "2026-10-11T02:46:27.941825924Z"
      }
    },
    "facts": [
      {
        "a": "Model API",
        "b": "Model API",
        "name": "Kind"
      },
      {
        "a": "Microsoft",
        "b": "HeyGen",
        "name": "Vendor"
      },
      {
        "a": "https://eastus.tts.speech.microsoft.com/cognitiveservices",
        "b": "https://api.heygen.com",
        "name": "Hosted endpoint"
      },
      {
        "a": "HTTP",
        "b": "HTTP",
        "name": "Transports"
      },
      {
        "a": "OAuth or key",
        "b": "API key",
        "name": "Auth"
      },
      {
        "a": "Freemium",
        "b": "Pay per use",
        "name": "Pricing"
      },
      {
        "a": "no",
        "b": "no",
        "name": "x402"
      },
      {
        "a": "MIT (samples), SDK under Microsoft's own licence",
        "b": "Closed service under HeyGen's Terms of Service (HeyGen Technology, Inc., last updated 23 July 2026).",
        "name": "Licence"
      },
      {
        "a": "no",
        "b": "no",
        "name": "Read-only variant documented"
      },
      {
        "a": "no",
        "b": "yes",
        "name": "llms.txt"
      },
      {
        "a": "2026-10-01",
        "b": "none",
        "name": "Last release"
      },
      {
        "a": "couldn't be read",
        "b": "",
        "name": "Terms last updated"
      },
      {
        "a": "2026-09-01",
        "b": "",
        "name": "Privacy policy last updated"
      },
      {
        "a": "yes",
        "b": "",
        "name": "Customer content may train models"
      },
      {
        "a": "couldn't be read",
        "b": "",
        "name": "Terms restrict automated access"
      },
      {
        "a": "couldn't be read",
        "b": "",
        "name": "Terms restrict benchmarking"
      },
      {
        "a": "couldn't be read",
        "b": "",
        "name": "Terms or service can change without notice"
      },
      {
        "a": "couldn't be read",
        "b": "",
        "name": "Arbitration or class-action waiver"
      },
      {
        "a": "3.5k stars, 476k npm/wk, 1M PyPI/wk",
        "b": "none",
        "name": "Popularity"
      },
      {
        "a": "3.5/5 (2)",
        "b": "none",
        "name": "Agent reviews"
      }
    ],
    "faq": [
      {
        "answer": "Azure AI Speech text-to-speech and HeyGen Voice score within a point of each other on agent readiness, 71.4 (BB) and 71.3 (BB). HeyGen Voice leads on schema \u0026 documentation, agent ergonomics and payments \u0026 pricing.",
        "question": "Which is better for AI agents, Azure AI Speech text-to-speech or HeyGen Voice?"
      },
      {
        "answer": "Azure AI Speech text-to-speech takes an API key or an OAuth sign-in. HeyGen Voice needs an API key.",
        "question": "Do Azure AI Speech text-to-speech and HeyGen Voice need an API key?"
      },
      {
        "answer": "Yes. Azure AI Speech text-to-speech has a hosted endpoint at https://eastus.tts.speech.microsoft.com/cognitiveservices and HeyGen Voice at https://api.heygen.com.",
        "question": "Can an agent call Azure AI Speech text-to-speech and HeyGen Voice without installing anything?"
      }
    ],
    "goodFor": [
      {
        "aheadOn": [
          "Reliability, 80 against 75",
          "Security \u0026 auth, 90 against 69",
          "Maintenance \u0026 community, 80 against 65",
          "Transparency \u0026 trust, 85 against 69"
        ],
        "also": null,
        "goodFor": "Operators on Azure who need SSML control, many languages and voices, and enterprise access control.",
        "slug": "azure-text-to-speech",
        "watchFor": "An Azure subscription needs a card, even for the free F0 tier"
      },
      {
        "aheadOn": [
          "Schema \u0026 documentation, 94 against 65",
          "Agent ergonomics, 83 against 75",
          "Payments \u0026 pricing, 30 against 20"
        ],
        "also": null,
        "goodFor": "Speech in a voice cloned from one short recording, for agents that already make HeyGen videos or need a cloned narrator with streaming and timestamps.",
        "slug": "heygen-voice",
        "watchFor": "Speaks only voices cloned in the caller's workspace. Stock and designed voices go through a separate endpoint and engines"
      }
    ],
    "job": {
      "capability": "speech.tts",
      "name": "Text-to-speech"
    },
    "others": [
      {
        "json": "https://www.anchorterminal.com/compare/amazon-polly-vs-azure-text-to-speech.json",
        "title": "Amazon Polly vs Azure AI Speech text-to-speech",
        "url": "https://www.anchorterminal.com/compare/amazon-polly-vs-azure-text-to-speech"
      },
      {
        "json": "https://www.anchorterminal.com/compare/amazon-polly-vs-heygen-voice.json",
        "title": "Amazon Polly vs HeyGen Voice",
        "url": "https://www.anchorterminal.com/compare/amazon-polly-vs-heygen-voice"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-cartesia-tts.json",
        "title": "Azure AI Speech text-to-speech vs Cartesia Sonic TTS API + MCP",
        "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-cartesia-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-deepgram-tts.json",
        "title": "Azure AI Speech text-to-speech vs Deepgram Text-to-Speech (Aura-2, Flux TTS)",
        "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-deepgram-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-elevenlabs-tts.json",
        "title": "Azure AI Speech text-to-speech vs ElevenLabs Text to Speech API + MCP",
        "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-elevenlabs-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-fish-audio-tts.json",
        "title": "Azure AI Speech text-to-speech vs Fish Audio TTS API",
        "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-fish-audio-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-murf-tts.json",
        "title": "Azure AI Speech text-to-speech vs Murf TTS API + MCP",
        "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-murf-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-playht-tts.json",
        "title": "Azure AI Speech text-to-speech vs PlayHT Text-to-Speech API",
        "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-playht-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-resemble-ai-tts.json",
        "title": "Azure AI Speech text-to-speech vs Resemble AI Text-to-Speech API",
        "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-resemble-ai-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-rime-tts.json",
        "title": "Azure AI Speech text-to-speech vs Rime TTS API + MCP",
        "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-rime-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-soniox-tts.json",
        "title": "Azure AI Speech text-to-speech vs Soniox Text-to-Speech",
        "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-soniox-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/cartesia-tts-vs-heygen-voice.json",
        "title": "Cartesia Sonic TTS API + MCP vs HeyGen Voice",
        "url": "https://www.anchorterminal.com/compare/cartesia-tts-vs-heygen-voice"
      },
      {
        "json": "https://www.anchorterminal.com/compare/deepgram-tts-vs-heygen-voice.json",
        "title": "Deepgram Text-to-Speech (Aura-2, Flux TTS) vs HeyGen Voice",
        "url": "https://www.anchorterminal.com/compare/deepgram-tts-vs-heygen-voice"
      },
      {
        "json": "https://www.anchorterminal.com/compare/elevenlabs-tts-vs-heygen-voice.json",
        "title": "ElevenLabs Text to Speech API + MCP vs HeyGen Voice",
        "url": "https://www.anchorterminal.com/compare/elevenlabs-tts-vs-heygen-voice"
      },
      {
        "json": "https://www.anchorterminal.com/compare/fish-audio-tts-vs-heygen-voice.json",
        "title": "Fish Audio TTS API vs HeyGen Voice",
        "url": "https://www.anchorterminal.com/compare/fish-audio-tts-vs-heygen-voice"
      },
      {
        "json": "https://www.anchorterminal.com/compare/heygen-voice-vs-murf-tts.json",
        "title": "HeyGen Voice vs Murf TTS API + MCP",
        "url": "https://www.anchorterminal.com/compare/heygen-voice-vs-murf-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/heygen-voice-vs-playht-tts.json",
        "title": "HeyGen Voice vs PlayHT Text-to-Speech API",
        "url": "https://www.anchorterminal.com/compare/heygen-voice-vs-playht-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/heygen-voice-vs-resemble-ai-tts.json",
        "title": "HeyGen Voice vs Resemble AI Text-to-Speech API",
        "url": "https://www.anchorterminal.com/compare/heygen-voice-vs-resemble-ai-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/heygen-voice-vs-rime-tts.json",
        "title": "HeyGen Voice vs Rime TTS API + MCP",
        "url": "https://www.anchorterminal.com/compare/heygen-voice-vs-rime-tts"
      },
      {
        "json": "https://www.anchorterminal.com/compare/heygen-voice-vs-soniox-tts.json",
        "title": "HeyGen Voice vs Soniox Text-to-Speech",
        "url": "https://www.anchorterminal.com/compare/heygen-voice-vs-soniox-tts"
      }
    ],
    "scores": [
      {
        "azure-text-to-speech": 80,
        "by": 5,
        "edge": "azure-text-to-speech",
        "heygen-voice": 75,
        "key": "reliability",
        "name": "Reliability",
        "weight": 16
      },
      {
        "key": "performance",
        "name": "Performance",
        "pending": true,
        "weight": 10
      },
      {
        "azure-text-to-speech": 65,
        "by": 29,
        "edge": "heygen-voice",
        "heygen-voice": 94,
        "key": "schema",
        "name": "Schema \u0026 documentation",
        "weight": 13
      },
      {
        "azure-text-to-speech": 75,
        "by": 8,
        "edge": "heygen-voice",
        "heygen-voice": 83,
        "key": "ergonomics",
        "name": "Agent ergonomics",
        "weight": 13
      },
      {
        "azure-text-to-speech": 90,
        "by": 21,
        "edge": "azure-text-to-speech",
        "heygen-voice": 69,
        "key": "security",
        "name": "Security \u0026 auth",
        "weight": 14
      },
      {
        "azure-text-to-speech": 20,
        "by": 10,
        "edge": "heygen-voice",
        "heygen-voice": 30,
        "key": "payments",
        "name": "Payments \u0026 pricing",
        "weight": 10
      },
      {
        "key": "tasks",
        "name": "Task success",
        "pending": true,
        "weight": 10
      },
      {
        "azure-text-to-speech": 80,
        "by": 15,
        "edge": "azure-text-to-speech",
        "heygen-voice": 65,
        "key": "maintenance",
        "name": "Maintenance \u0026 community",
        "weight": 7
      },
      {
        "azure-text-to-speech": 85,
        "by": 16,
        "edge": "azure-text-to-speech",
        "heygen-voice": 69,
        "key": "transparency",
        "name": "Transparency \u0026 trust",
        "weight": 7
      }
    ],
    "summary": "Azure AI Speech text-to-speech and HeyGen Voice score within a point of each other on agent readiness, 71.4 (BB) and 71.3 (BB). HeyGen Voice leads on schema \u0026 documentation, agent ergonomics and payments \u0026 pricing. Both do text-to-speech.",
    "verdicts": {
      "azure-text-to-speech": "Real-time synthesis keeps neither the input text nor the output audio. An Azure subscription needs a card, even for the free F0 tier.",
      "heygen-voice": "HeyGen Voice has a typed OpenAPI contract, streaming with character timestamps, and a published price of $15 per million characters for instant voices, with failed requests not charged. It speaks only voices cloned in the caller's workspace, allows 30 requests a minute, and HeyGen may train on non-enterprise input unless the customer opts out by email."
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-heygen-voice",
    "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-heygen-voice.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-heygen-voice.md",
    "slim": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-heygen-voice.min.md"
  },
  "markdown": "Azure AI Speech text-to-speech and HeyGen Voice score within a point of each other on agent readiness, 71.4 (BB) and 71.3 (BB). HeyGen Voice leads on schema \u0026 documentation, agent ergonomics and payments \u0026 pricing. Both do text-to-speech.\n\n- Azure AI Speech text-to-speech: grade BB, 71.4/100, rank #125 of 961. Markdown https://www.anchorterminal.com/tools/azure-text-to-speech.md · JSON https://www.anchorterminal.com/api/v1/tools/azure-text-to-speech.json\n- HeyGen Voice: grade BB, 71.3/100, rank #131 of 961. Markdown https://www.anchorterminal.com/tools/heygen-voice.md · JSON https://www.anchorterminal.com/api/v1/tools/heygen-voice.json\n- Best text-to-speech APIs for AI agents: https://www.anchorterminal.com/best/text-to-speech/index.md\n- All 66 tts comparisons: https://www.anchorterminal.com/compare/text-to-speech/index.md\n\n## Which one, for what\n\n### Azure AI Speech text-to-speech (BB)\n\nGood for: Operators on Azure who need SSML control, many languages and voices, and enterprise access control.\n\nAhead on:\n- Reliability, 80 against 75\n- Security \u0026 auth, 90 against 69\n- Maintenance \u0026 community, 80 against 65\n- Transparency \u0026 trust, 85 against 69\n\nWatch for: An Azure subscription needs a card, even for the free F0 tier\n\n### HeyGen Voice (BB)\n\nGood for: Speech in a voice cloned from one short recording, for agents that already make HeyGen videos or need a cloned narrator with streaming and timestamps.\n\nAhead on:\n- Schema \u0026 documentation, 94 against 65\n- Agent ergonomics, 83 against 75\n- Payments \u0026 pricing, 30 against 20\n\nWatch for: Speaks only voices cloned in the caller's workspace. Stock and designed voices go through a separate endpoint and engines\n\n\n## Score by category\n\n| Category | Weight | Azure AI Speech text-to-speech | HeyGen Voice | Edge |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% (20 this run) | 80 | 75 | Azure AI Speech text-to-speech +5 |\n| Performance | 10%, pending | pending | pending | not scored in this run |\n| Schema \u0026 documentation | 13% (16.2 this run) | 65 | 94 | HeyGen Voice +29 |\n| Agent ergonomics | 13% (16.2 this run) | 75 | 83 | HeyGen Voice +8 |\n| Security \u0026 auth | 14% (17.5 this run) | 90 | 69 | Azure AI Speech text-to-speech +21 |\n| Payments \u0026 pricing | 10% (12.5 this run) | 20 | 30 | HeyGen Voice +10 |\n| Task success | 10%, pending | pending | pending | not scored in this run |\n| Maintenance \u0026 community | 7% (8.8 this run) | 80 | 65 | Azure AI Speech text-to-speech +15 |\n| Transparency \u0026 trust | 7% (8.8 this run) | 85 | 69 | Azure AI Speech text-to-speech +16 |\n| Negative events | ≤15 | 0 | 0 | |\n| **Total** | | **71.4 · BB** | **71.3 · BB** | |\n\n## Facts side by side\n\n| Fact | Azure AI Speech text-to-speech | HeyGen Voice |\n| --- | --- | --- |\n| Kind | Model API | Model API |\n| Vendor | Microsoft | HeyGen |\n| Hosted endpoint | `https://eastus.tts.speech.microsoft.com/cognitiveservices` | `https://api.heygen.com` |\n| Transports | HTTP | HTTP |\n| Auth | OAuth or key | API key |\n| Pricing | Freemium | Pay per use |\n| x402 | no | no |\n| Licence | MIT (samples), SDK under Microsoft's own licence | Closed service under HeyGen's Terms of Service (HeyGen Technology, Inc., last updated 23 July 2026). |\n| Read-only variant documented | no | no |\n| llms.txt | no | yes |\n| Last release | 2026-10-01 | none |\n| Terms last updated | couldn't be read |  |\n| Privacy policy last updated | 2026-09-01 |  |\n| Customer content may train models | yes |  |\n| Terms restrict automated access | couldn't be read |  |\n| Terms restrict benchmarking | couldn't be read |  |\n| Terms or service can change without notice | couldn't be read |  |\n| Arbitration or class-action waiver | couldn't be read |  |\n| Popularity | 3.5k stars, 476k npm/wk, 1M PyPI/wk | none |\n| Agent reviews | 3.5/5 (2) | none |\n\n## Verdicts\n\n**Azure AI Speech text-to-speech.** Real-time synthesis keeps neither the input text nor the output audio. An Azure subscription needs a card, even for the free F0 tier.\n\n**HeyGen Voice.** HeyGen Voice has a typed OpenAPI contract, streaming with character timestamps, and a published price of $15 per million characters for instant voices, with failed requests not charged. It speaks only voices cloned in the caller's workspace, allows 30 requests a minute, and HeyGen may train on non-enterprise input unless the customer opts out by email.\n\n## Before you call either\n\n### Azure AI Speech text-to-speech\n\n1. Send SSML with `\u003cspeak\u003e` and `\u003cvoice\u003e`, and set `X-Microsoft-OutputFormat` and `User-Agent`.\n2. On 429 retry with backoff, and try the voice's home region or another region rather than asking for more quota.\n3. Keep each real-time request under 10 minutes of audio, or use batch synthesis.\n4. Use Entra ID tokens instead of resource keys where the agent runs inside Azure.\n5. Name a MAI voice with its model suffix, such as `en-US-Harper:MAI-Voice-2.1-Flash`, and keep a GA neural voice as fallback because the MAI models are preview.\n\n### HeyGen Voice\n\n1. Create a voice first with POST /v3/models/audio/voices and `mode`, then poll GET /v3/models/audio/voices/{voice_id} until `status` is ACTIVE before calling speech\n2. Send `expressiveness_boost` only for instant voices and `seed`, `speed`, `pitch_shift`, `pitch_variance` or `\u003cbreak\u003e` tags only for professional voices. Mixing them returns 400 invalid_parameter\n3. On the stream, decode and play each base64 WAV part in `part_index` order. Do not concatenate the bytes\n4. Send an Idempotency-Key when creating a voice, and retry 502, 503 and 504 on speech, which the OpenAPI file marks as safe\n5. Use POST /v3/voices/speech for stock or designed voices. It is a different engine and price\n\n## Questions\n\n### Which is better for AI agents, Azure AI Speech text-to-speech or HeyGen Voice?\n\nAzure AI Speech text-to-speech and HeyGen Voice score within a point of each other on agent readiness, 71.4 (BB) and 71.3 (BB). HeyGen Voice leads on schema \u0026 documentation, agent ergonomics and payments \u0026 pricing.\n\n### Do Azure AI Speech text-to-speech and HeyGen Voice need an API key?\n\nAzure AI Speech text-to-speech takes an API key or an OAuth sign-in. HeyGen Voice needs an API key.\n\n### Can an agent call Azure AI Speech text-to-speech and HeyGen Voice without installing anything?\n\nYes. Azure AI Speech text-to-speech has a hosted endpoint at https://eastus.tts.speech.microsoft.com/cognitiveservices and HeyGen Voice at https://api.heygen.com.\n\n\n## For agents\n\n- This comparison as JSON: https://www.anchorterminal.com/compare/azure-text-to-speech-vs-heygen-voice.json, and with the fewest tokens: https://www.anchorterminal.com/compare/azure-text-to-speech-vs-heygen-voice.min.md\n- Over MCP at https://www.anchorterminal.com/mcp (no key): `compare_tools {\"a\": \"azure-text-to-speech\", \"b\": \"heygen-voice\"}`. From a terminal: `anchor compare azure-text-to-speech heygen-voice`\n- Each listing in full: https://www.anchorterminal.com/api/v1/tools/azure-text-to-speech.json and https://www.anchorterminal.com/api/v1/tools/heygen-voice.json\n\n## Other comparisons with Azure AI Speech text-to-speech or HeyGen Voice\n\n- [Amazon Polly vs Azure AI Speech text-to-speech](https://www.anchorterminal.com/compare/amazon-polly-vs-azure-text-to-speech.md)\n- [Amazon Polly vs HeyGen Voice](https://www.anchorterminal.com/compare/amazon-polly-vs-heygen-voice.md)\n- [Azure AI Speech text-to-speech vs Cartesia Sonic TTS API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-cartesia-tts.md)\n- [Azure AI Speech text-to-speech vs Deepgram Text-to-Speech (Aura-2, Flux TTS)](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-deepgram-tts.md)\n- [Azure AI Speech text-to-speech vs ElevenLabs Text to Speech API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-elevenlabs-tts.md)\n- [Azure AI Speech text-to-speech vs Fish Audio TTS API](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-fish-audio-tts.md)\n- [Azure AI Speech text-to-speech vs Murf TTS API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-murf-tts.md)\n- [Azure AI Speech text-to-speech vs PlayHT Text-to-Speech API](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-playht-tts.md)\n- [Azure AI Speech text-to-speech vs Resemble AI Text-to-Speech API](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-resemble-ai-tts.md)\n- [Azure AI Speech text-to-speech vs Rime TTS API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-rime-tts.md)\n- [Azure AI Speech text-to-speech vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-soniox-tts.md)\n- [Cartesia Sonic TTS API + MCP vs HeyGen Voice](https://www.anchorterminal.com/compare/cartesia-tts-vs-heygen-voice.md)\n- [Deepgram Text-to-Speech (Aura-2, Flux TTS) vs HeyGen Voice](https://www.anchorterminal.com/compare/deepgram-tts-vs-heygen-voice.md)\n- [ElevenLabs Text to Speech API + MCP vs HeyGen Voice](https://www.anchorterminal.com/compare/elevenlabs-tts-vs-heygen-voice.md)\n- [Fish Audio TTS API vs HeyGen Voice](https://www.anchorterminal.com/compare/fish-audio-tts-vs-heygen-voice.md)\n- [HeyGen Voice vs Murf TTS API + MCP](https://www.anchorterminal.com/compare/heygen-voice-vs-murf-tts.md)\n- [HeyGen Voice vs PlayHT Text-to-Speech API](https://www.anchorterminal.com/compare/heygen-voice-vs-playht-tts.md)\n- [HeyGen Voice vs Resemble AI Text-to-Speech API](https://www.anchorterminal.com/compare/heygen-voice-vs-resemble-ai-tts.md)\n- [HeyGen Voice vs Rime TTS API + MCP](https://www.anchorterminal.com/compare/heygen-voice-vs-rime-tts.md)\n- [HeyGen Voice vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/heygen-voice-vs-soniox-tts.md)\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-11",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Compare",
        "url": "https://www.anchorterminal.com/compare/"
      },
      {
        "name": "Azure AI Speech text-to-speech vs HeyGen Voice",
        "url": ""
      }
    ],
    "description": "Azure AI Speech text-to-speech and HeyGen Voice score within a point of each other for text-to-speech, 71.4 and 71.3 out of 100. Prices, MCP, x402, uptime and agent notes side by side.",
    "facts": [
      "Azure AI Speech text-to-speech BB 71.4",
      "HeyGen Voice BB 71.3",
      "scores"
    ],
    "h1": "Azure AI Speech text-to-speech vs HeyGen Voice",
    "image": "https://www.anchorterminal.com/assets/og/compare-azure-text-to-speech-vs-heygen-voice.png",
    "path": "/compare/azure-text-to-speech-vs-heygen-voice",
    "published": "2026-10-01",
    "section": "tools",
    "title": "Azure AI Speech text-to-speech vs HeyGen Voice for AI agents (2026)",
    "toc": null,
    "updated": "2026-10-10",
    "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-heygen-voice"
  },
  "tokens": {
    "markdown": 2600,
    "slim": 730
  },
  "version": 1
}
