{
  "data": {
    "a": {
      "slug": "azure-text-to-speech",
      "name": "Azure AI Speech text-to-speech",
      "vendor": "Microsoft Azure",
      "vendorUrl": "https://azure.microsoft.com/en-us/products/ai-services/text-to-speech",
      "kind": "model",
      "category": "text-to-speech",
      "summary": "Azure's text-to-speech service for generating spoken audio.",
      "url": "https://www.anchorterminal.com/tools/azure-text-to-speech",
      "markdownUrl": "https://www.anchorterminal.com/tools/azure-text-to-speech.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/azure-text-to-speech.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/azure-text-to-speech.json",
      "repo": "https://github.com/Azure-Samples/cognitive-services-speech-sdk",
      "license": "MIT (samples), SDK under Microsoft's own licence",
      "transports": [
        "http"
      ],
      "remoteUrl": "https://eastus.tts.speech.microsoft.com/cognitiveservices",
      "packages": [
        {
          "registry": "pypi",
          "name": "azure-cognitiveservices-speech"
        },
        {
          "registry": "npm",
          "name": "microsoft-cognitiveservices-speech-sdk"
        }
      ],
      "auth": "mixed",
      "authNotes": "`Ocp-Apim-Subscription-Key` header with a Speech resource key, or a Microsoft Entra ID bearer token. Endpoints are per region.",
      "pricing": "freemium",
      "pricingNotes": "Free F0 tier with 500,000 characters a month. Pay as you go in East US is $15 per 1M characters for Neural and Neural HD Flash voices and $22 for Neural HD, real time or batch. Commitment tiers from $960 a month for 80M characters (https://azure.microsoft.com/en-us/pricing/details/speech/).",
      "priceSummary": "$960 / mo",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No machine payment. Billing runs through a cloud account with a card or invoice.",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 3450,
        "npmWeekly": 475621,
        "pypiWeekly": 1032532,
        "asOf": "2026-09-30"
      },
      "docsUrl": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/text-to-speech",
      "capabilities": [
        "speech.tts",
        "speech.streaming",
        "speech.voices",
        "speech.ssml",
        "speech.languages"
      ],
      "tags": [
        "hosted",
        "freemium",
        "free-tier",
        "closed-source",
        "python",
        "typescript",
        "enterprise",
        "streaming",
        "batch",
        "async-jobs"
      ],
      "lastRelease": "2026-09-28",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 73.7,
        "grade": "BB",
        "agentReady": true,
        "rank": 56,
        "ranked": true,
        "rankOf": 452,
        "categoryRank": 2,
        "methodology": "0.3",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 75,
          "maintenance": 80,
          "payments": 20,
          "reliability": 90,
          "schema": 65,
          "security": 90,
          "transparency": 88
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-01"
        },
        "negative": 0,
        "verdict": "Real-time synthesis keeps neither the input text nor the output audio. An Azure subscription needs a card, even for the free F0 tier.",
        "strengths": [
          "Real-time synthesis keeps neither the input text nor the output audio",
          "Full SSML with speaking styles, prosody, phonemes, lexicons and up to 50 voice or audio tags a request",
          "Microsoft Entra ID with role-based access, or two rotatable keys",
          "Covered by Microsoft's online services SLA",
          "S0 starts at 30 requests a second and can be raised to 1,000"
        ],
        "weaknesses": [
          "An Azure subscription needs a card, even for the free F0 tier",
          "No llms.txt and no OpenAPI file for text-to-speech found",
          "429s often reflect busy capacity for a voice in a region, which a quota increase doesn't fix",
          "The voice list comes back as one response per region with no paging documented",
          "MAI-Voice-2-Flash, the low-latency model, is still in preview"
        ],
        "agentNotes": [
          "Send SSML with `\u003cspeak\u003e` and `\u003cvoice\u003e`, and set `X-Microsoft-OutputFormat` and `User-Agent`.",
          "On 429 retry with backoff, and try the voice's home region or another region rather than asking for more quota.",
          "Keep each real-time request under 10 minutes of audio, or use batch synthesis.",
          "Use Entra ID tokens instead of resource keys where the agent runs inside Azure.",
          "Cache the voice list per region, since it returns hundreds of entries at once."
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 2,
        "avgRating": 3.5,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "BB",
            "methodology": "0.3",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 73.7
          }
        ],
        "editorialScores": {
          "ergonomics": 75,
          "maintenance": 80,
          "payments": 20,
          "reliability": 90,
          "schema": 65,
          "security": 90,
          "transparency": 80
        },
        "provenanceScore": 95
      },
      "connect": {
        "install": "pip install azure-cognitiveservices-speech   # or: npm i microsoft-cognitiveservices-speech-sdk",
        "http": "curl -X POST \"https://eastus.tts.speech.microsoft.com/cognitiveservices/v1\" \\\n  -H \"Ocp-Apim-Subscription-Key: $AZURE_SPEECH_KEY\" -H \"Content-Type: application/ssml+xml\" \\\n  -H \"X-Microsoft-OutputFormat: audio-24khz-48kbitrate-mono-mp3\" -o speech.mp3 \\\n  -d '\u003cspeak version=\"1.0\" xml:lang=\"en-US\"\u003e\u003cvoice name=\"en-US-AvaMultilingualNeural\"\u003eYour table is booked for seven.\u003c/voice\u003e\u003c/speak\u003e'"
      },
      "letme": {
        "capability": "https://letme.dev/speech.tts",
        "tool": "https://letme.dev/azure-text-to-speech"
      },
      "sameCompany": [
        "azure-foundry-fine-tuning",
        "azure-ai-content-safety",
        "azure-speech-to-text",
        "microsoft-learn-mcp",
        "playwright-mcp",
        "azure-mcp",
        "azure-translator",
        "microsoft-graph-calendar"
      ],
      "area": "voice",
      "unitPrices": [
        {
          "item": "Neural and Neural HD Flash voices",
          "unit": "1m-chars",
          "usd": 15,
          "note": "real time or batch, East US"
        },
        {
          "item": "Neural HD voices",
          "unit": "1m-chars",
          "usd": 22
        },
        {
          "item": "Commitment tier 80M characters",
          "unit": "month",
          "usd": 960,
          "note": "$12 per 1M overage"
        }
      ],
      "provenance": {
        "legalEntity": "Microsoft Corporation",
        "domain": "microsoft.com",
        "domainRegistered": "1991-05-02",
        "domainNote": "Endpoints are on speech.microsoft.com, api.cognitive.microsoft.com and cognitiveservices.azure.com. microsoft.com publishes a security.txt, but it passed its Expires date on 2026-09-23.",
        "endpointOnVendorDomain": true,
        "terms": "https://www.microsoft.com/licensing/terms/",
        "privacy": "https://privacy.microsoft.com/en-us/privacystatement",
        "statusPage": "https://azure.status.microsoft/en-us/status",
        "changelog": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/releasenotes",
        "securityTxt": "expired",
        "checked": "2026-09-30",
        "score": 95
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/azure-text-to-speech.json",
      "live": {
        "slug": "azure-text-to-speech",
        "probe": {
          "target": "https://eastus.tts.speech.microsoft.com/cognitiveservices",
          "method": "get",
          "lastAt": "2026-10-04T23:48:04.678459126Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 259,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 257,
          "p95ms24h": 304,
          "samples24h": 272,
          "samples30d": 1100,
          "days": [
            {
              "date": "2026-09-30",
              "probes": 35,
              "ok": 35
            },
            {
              "date": "2026-10-01",
              "probes": 276,
              "ok": 276
            },
            {
              "date": "2026-10-02",
              "probes": 248,
              "ok": 248
            },
            {
              "date": "2026-10-03",
              "probes": 271,
              "ok": 271
            },
            {
              "date": "2026-10-04",
              "probes": 270,
              "ok": 270
            }
          ]
        },
        "versions": [
          {
            "registry": "github",
            "name": "Azure-Samples/cognitive-services-speech-sdk",
            "version": "ingestion-v2.1.13",
            "released": "2026-07-10",
            "seenAt": "2026-10-04T16:21:39.409053457Z"
          },
          {
            "registry": "npm",
            "name": "microsoft-cognitiveservices-speech-sdk",
            "version": "1.52.0",
            "seenAt": "2026-10-04T16:21:39.358419331Z"
          },
          {
            "registry": "pypi",
            "name": "azure-cognitiveservices-speech",
            "version": "1.52.0",
            "released": "2026-09-28",
            "seenAt": "2026-10-04T16:21:39.229173438Z"
          }
        ],
        "githubStars": 3450,
        "npmWeekly": 508156,
        "pypiWeekly": 704368,
        "securityTxt": {
          "url": "https://microsoft.com/.well-known/security.txt",
          "state": "expired",
          "expires": "2026-09-23T16:00:00.000Z",
          "checkedAt": "2026-10-04T15:16:01.36832038Z"
        },
        "domain": {
          "domain": "microsoft.com",
          "registered": "1991-05-02",
          "source": "https://rdap.verisign.com/com/v1/domain/microsoft.com",
          "checkedAt": "2026-10-04T13:04:13.488857536Z"
        },
        "updatedAt": "2026-10-04T23:48:04.678459126Z"
      }
    },
    "b": {
      "slug": "deepgram-tts",
      "name": "Deepgram Text-to-Speech (Aura-2, Flux TTS)",
      "vendor": "Deepgram",
      "vendorUrl": "https://deepgram.com",
      "kind": "model",
      "category": "text-to-speech",
      "summary": "Deepgram's text-to-speech API for generating spoken audio.",
      "url": "https://www.anchorterminal.com/tools/deepgram-tts",
      "markdownUrl": "https://www.anchorterminal.com/tools/deepgram-tts.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/deepgram-tts.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/deepgram-tts.json",
      "repo": "https://github.com/deepgram/deepgram-python-sdk",
      "license": "MIT (SDKs)",
      "transports": [
        "http",
        "streamable-http",
        "stdio",
        "sse"
      ],
      "remoteUrl": "https://api.deepgram.com/v1",
      "packages": [
        {
          "registry": "npm",
          "name": "@deepgram/sdk"
        },
        {
          "registry": "pypi",
          "name": "deepgram-sdk"
        },
        {
          "registry": "pypi",
          "name": "deepctl"
        }
      ],
      "auth": "api-key",
      "authNotes": "`Authorization: Token \u003ckey\u003e` header on REST and WebSocket calls. Short-lived JWTs (30-second TTL) from the token endpoint for browsers. The `dg` CLI MCP server uses `dg login` credentials or `DEEPGRAM_API_KEY`. The docs MCP needs no key.",
      "pricing": "usage",
      "pricingNotes": "$200 free credit with no card, then pay as you go. Flux TTS $0.045 per 1,000 characters, Aura-2 $0.030, Aura-1 $0.015. Growth (from $4,000 a year prepaid) cuts these to $0.0405, $0.027 and $0.0135. A matching-credit offer on Flux TTS runs to 2026-12-31, capped at $500 (https://deepgram.com/pricing).",
      "priceSummary": "Pay per use",
      "where": "both",
      "x402": {
        "level": "no",
        "evidence": "No x402 or machine payment in the docs or pricing. Card or prepaid credits only (checked 2026-09-30).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 468,
        "npmWeekly": 1123798,
        "pypiWeekly": 805026,
        "asOf": "2026-09-30"
      },
      "docsUrl": "https://developers.deepgram.com/docs/tts-models-languages-overview",
      "llmsTxt": "https://developers.deepgram.com/llms.txt",
      "openapi": "https://developers.deepgram.com/openapi.json",
      "capabilities": [
        "speech.tts",
        "speech.streaming",
        "speech.voices",
        "speech.languages"
      ],
      "tags": [
        "hosted",
        "no-card",
        "closed-source",
        "python",
        "typescript",
        "openapi",
        "llms-txt",
        "mcp",
        "streaming",
        "batch",
        "enterprise",
        "self-hosted"
      ],
      "lastRelease": "2026-09-29",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 73,
        "grade": "BB",
        "agentReady": true,
        "rank": 63,
        "ranked": true,
        "rankOf": 452,
        "categoryRank": 4,
        "methodology": "0.3",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 82,
          "maintenance": 73,
          "payments": 40,
          "reliability": 70,
          "schema": 95,
          "security": 70,
          "transparency": 75
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "high",
          "date": "2026-10-01"
        },
        "negative": 0,
        "verdict": "OpenAPI 3.1 and AsyncAPI files, llms.txt and Markdown pages. Requests can be kept for training unless each one sets `mip_opt_out=true`.",
        "strengths": [
          "OpenAPI 3.1 and AsyncAPI files, llms.txt and Markdown pages",
          "Flux TTS's Interrupt event returns `text_spoken` and `text_remaining` on barge-in",
          "Keys carry roles and scopes, and browser tokens live 30 seconds",
          "Per-request logs through `GET /v1/projects/{project_id}/requests`, filterable by date and status",
          "$200 credit with no card, then Aura-2 at $0.030 per 1,000 characters"
        ],
        "weaknesses": [
          "Requests can be kept for training unless each one sets `mip_opt_out=true`",
          "Flux TTS is English only and capped at 5 concurrent streams in the EU, Australia and India below Enterprise",
          "No SSML, and Flux TTS strips other vendors' tags with a warning",
          "Elevated Flux TTS errors for about four hours on 25 September 2026",
          "Aura-2 REST requests stop at 2,000 characters"
        ],
        "agentNotes": [
          "Set `mip_opt_out=true` on every request if the text mustn't be kept for training.",
          "Split Aura-2 REST text under 2,000 characters or expect a 413.",
          "Pass `model` on `/v2/speak`, where it's required.",
          "Strip SSML before sending, since it's removed with an `INPUT_MARKUP_STRIPPED` warning.",
          "Back off exponentially on 429 and keep traffic in one project."
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 2,
        "avgRating": 3.5,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "high",
            "grade": "BB",
            "methodology": "0.3",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 73
          }
        ],
        "editorialScores": {
          "ergonomics": 82,
          "maintenance": 73,
          "payments": 40,
          "reliability": 70,
          "schema": 95,
          "security": 70,
          "transparency": 60
        },
        "provenanceScore": 90
      },
      "connect": {
        "http": "curl \"https://api.deepgram.com/v1/speak?model=aura-2-thalia-en\" \\\n  -H \"Authorization: Token $DEEPGRAM_API_KEY\" -H \"content-type: application/json\" \\\n  -d '{\"text\":\"Hello, how are you?\"}' -o hello.mp3",
        "claudeCode": "claude mcp add deepgram-docs --transport http https://api.dx.deepgram.com/kapa/mcp",
        "config": {
          "mcpServers": {
            "deepgram": {
              "args": [
                "mcp"
              ],
              "command": "dg",
              "env": {
                "DEEPGRAM_API_KEY": "${DEEPGRAM_API_KEY}"
              }
            }
          }
        }
      },
      "letme": {
        "capability": "https://letme.dev/speech.tts",
        "tool": "https://letme.dev/deepgram-tts"
      },
      "sameCompany": [
        "deepgram-stt",
        "deepgram-voice-agent"
      ],
      "area": "voice",
      "unitPrices": [
        {
          "item": "Flux TTS",
          "unit": "1m-chars",
          "usd": 45,
          "note": "pay as you go, $0.045 per 1,000 characters"
        },
        {
          "item": "Aura-2",
          "unit": "1m-chars",
          "usd": 30,
          "note": "pay as you go"
        },
        {
          "item": "Aura-1",
          "unit": "1m-chars",
          "usd": 15,
          "note": "pay as you go"
        },
        {
          "item": "Aura-2 on Growth",
          "unit": "1m-chars",
          "usd": 27,
          "note": "prepaid annual plan from $4,000"
        }
      ],
      "provenance": {
        "legalEntity": "Deepgram, Inc.",
        "domain": "deepgram.com",
        "domainRegistered": "2016-01-28",
        "endpointOnVendorDomain": true,
        "terms": "https://deepgram.com/terms",
        "privacy": "https://deepgram.com/privacy",
        "statusPage": "https://status.deepgram.com",
        "changelog": "https://developers.deepgram.com/changelog",
        "securityTxt": "none",
        "checked": "2026-09-30",
        "score": 90
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/deepgram-tts.json",
      "live": {
        "slug": "deepgram-tts",
        "probe": {
          "target": "https://api.deepgram.com/v1",
          "method": "get",
          "lastAt": "2026-10-04T23:48:07.226240112Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 310,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 347,
          "p95ms24h": 521,
          "samples24h": 272,
          "samples30d": 1100,
          "days": [
            {
              "date": "2026-09-30",
              "probes": 35,
              "ok": 35
            },
            {
              "date": "2026-10-01",
              "probes": 276,
              "ok": 276
            },
            {
              "date": "2026-10-02",
              "probes": 248,
              "ok": 248
            },
            {
              "date": "2026-10-03",
              "probes": 271,
              "ok": 271
            },
            {
              "date": "2026-10-04",
              "probes": 270,
              "ok": 270
            }
          ]
        },
        "vendorStatus": {
          "page": "https://status.deepgram.com",
          "indicator": "none",
          "summary": "All Systems Operational",
          "checkedAt": "2026-10-04T23:49:10.277586322Z"
        },
        "versions": [
          {
            "registry": "github",
            "name": "deepgram/deepgram-python-sdk",
            "version": "v7.12.0",
            "released": "2026-10-02",
            "seenAt": "2026-10-04T16:25:18.900755205Z"
          },
          {
            "registry": "npm",
            "name": "@deepgram/sdk",
            "version": "5.14.0",
            "seenAt": "2026-10-04T16:25:16.604084571Z"
          },
          {
            "registry": "pypi",
            "name": "deepctl",
            "version": "0.3.1",
            "released": "2026-09-29",
            "seenAt": "2026-10-04T16:25:17.004312351Z"
          },
          {
            "registry": "pypi",
            "name": "deepgram-sdk",
            "version": "7.12.0",
            "released": "2026-10-02",
            "seenAt": "2026-10-04T16:25:16.891210762Z"
          }
        ],
        "githubStars": 468,
        "npmWeekly": 1147181,
        "pypiWeekly": 806541,
        "securityTxt": {
          "url": "https://deepgram.com/.well-known/security.txt",
          "state": "none",
          "checkedAt": "2026-10-04T15:16:00.338324598Z"
        },
        "llmsTxt": {
          "url": "https://developers.deepgram.com/llms.txt",
          "ok": true,
          "status": 200,
          "checkedAt": "2026-10-04T15:17:31.890004033Z"
        },
        "domain": {
          "domain": "deepgram.com",
          "registered": "2016-01-28",
          "source": "https://rdap.verisign.com/com/v1/domain/deepgram.com",
          "checkedAt": "2026-10-04T13:03:28.939824686Z"
        },
        "updatedAt": "2026-10-04T23:49:10.277586322Z"
      }
    },
    "summary": "Azure AI Speech text-to-speech has a score of 73.7 (BB) against Deepgram Text-to-Speech (Aura-2, Flux TTS)'s 73 (BB). Both do speech tts. The largest gap is schema \u0026 documentation, 30 points."
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-deepgram-tts",
    "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-deepgram-tts.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-deepgram-tts.md",
    "slim": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-deepgram-tts.min.md"
  },
  "markdown": "Azure AI Speech text-to-speech has a score of 73.7 (BB) against Deepgram Text-to-Speech (Aura-2, Flux TTS)'s 73 (BB). Both do speech tts. The largest gap is schema \u0026 documentation, 30 points.\n\n- Azure AI Speech text-to-speech: grade BB, 73.7/100, rank #56 of 452. Markdown https://www.anchorterminal.com/tools/azure-text-to-speech.md · JSON https://www.anchorterminal.com/api/v1/tools/azure-text-to-speech.json\n- Deepgram Text-to-Speech (Aura-2, Flux TTS): grade BB, 73/100, rank #63 of 452. Markdown https://www.anchorterminal.com/tools/deepgram-tts.md · JSON https://www.anchorterminal.com/api/v1/tools/deepgram-tts.json\n\n## Which one, for what\n\nPick Azure AI Speech text-to-speech for reliability (+20), security \u0026 auth (+20), maintenance \u0026 community (+7), transparency \u0026 trust (+13).\n\nPick Deepgram Text-to-Speech (Aura-2, Flux TTS) for schema \u0026 documentation (+30), agent ergonomics (+7), payments \u0026 pricing (+20).\n\n## Score by category\n\n| Category | Weight | Azure AI Speech text-to-speech | Deepgram Text-to-Speech (Aura-2, Flux TTS) | Edge |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% (20 this run) | 90 | 70 | Azure AI Speech text-to-speech +20 |\n| Performance | 10%, pending | pending | pending | not scored in this run |\n| Schema \u0026 documentation | 13% (16.2 this run) | 65 | 95 | Deepgram Text-to-Speech (Aura-2, Flux TTS) +30 |\n| Agent ergonomics | 13% (16.2 this run) | 75 | 82 | Deepgram Text-to-Speech (Aura-2, Flux TTS) +7 |\n| Security \u0026 auth | 14% (17.5 this run) | 90 | 70 | Azure AI Speech text-to-speech +20 |\n| Payments \u0026 pricing | 10% (12.5 this run) | 20 | 40 | Deepgram Text-to-Speech (Aura-2, Flux TTS) +20 |\n| Task success | 10%, pending | pending | pending | not scored in this run |\n| Maintenance \u0026 community | 7% (8.8 this run) | 80 | 73 | Azure AI Speech text-to-speech +7 |\n| Transparency \u0026 trust | 7% (8.8 this run) | 88 | 75 | Azure AI Speech text-to-speech +13 |\n| Negative events | ≤15 | 0 | 0 | |\n| **Total** | | **73.7 · BB** | **73 · BB** | |\n\n## Facts side by side\n\n| Fact | Azure AI Speech text-to-speech | Deepgram Text-to-Speech (Aura-2, Flux TTS) |\n| --- | --- | --- |\n| Kind | Model API | Model API |\n| Vendor | Microsoft Azure | Deepgram |\n| Hosted endpoint | `https://eastus.tts.speech.microsoft.com/cognitiveservices` | `https://api.deepgram.com/v1` |\n| Transports | HTTP | HTTP, Streamable HTTP, stdio, SSE (legacy) |\n| Auth | OAuth or key | API key |\n| Pricing | Freemium | Pay per use |\n| x402 | no | no |\n| Licence | MIT (samples), SDK under Microsoft's own licence | MIT (SDKs) |\n| Tools exposed | none | none |\n| Context cost (tools/list) | n/a | n/a |\n| p95 latency | not measured yet | not measured yet |\n| Availability (30d) | not measured yet | not measured yet |\n| Read-only variant documented | no | no |\n| llms.txt | no | yes |\n| MCP registry | not listed | not listed |\n| Last release | 2026-09-28 | 2026-09-29 |\n| Popularity | 3.5k stars, 476k npm/wk, 1M PyPI/wk | 468 stars, 1.1M npm/wk, 805k PyPI/wk |\n| Agent reviews | 3.5/5 (2) | 3.5/5 (2) |\n\n## Verdicts\n\n**Azure AI Speech text-to-speech.** Real-time synthesis keeps neither the input text nor the output audio. An Azure subscription needs a card, even for the free F0 tier.\n\n**Deepgram Text-to-Speech (Aura-2, Flux TTS).** OpenAPI 3.1 and AsyncAPI files, llms.txt and Markdown pages. Requests can be kept for training unless each one sets `mip_opt_out=true`.\n\n## Before you call either\n\n### Azure AI Speech text-to-speech\n\n1. Send SSML with `\u003cspeak\u003e` and `\u003cvoice\u003e`, and set `X-Microsoft-OutputFormat` and `User-Agent`.\n2. On 429 retry with backoff, and try the voice's home region or another region rather than asking for more quota.\n3. Keep each real-time request under 10 minutes of audio, or use batch synthesis.\n4. Use Entra ID tokens instead of resource keys where the agent runs inside Azure.\n5. Cache the voice list per region, since it returns hundreds of entries at once.\n\n### Deepgram Text-to-Speech (Aura-2, Flux TTS)\n\n1. Set `mip_opt_out=true` on every request if the text mustn't be kept for training.\n2. Split Aura-2 REST text under 2,000 characters or expect a 413.\n3. Pass `model` on `/v2/speak`, where it's required.\n4. Strip SSML before sending, since it's removed with an `INPUT_MARKUP_STRIPPED` warning.\n5. Back off exponentially on 429 and keep traffic in one project.\n\n## Other comparisons with Azure AI Speech text-to-speech or Deepgram Text-to-Speech (Aura-2, Flux TTS)\n\n- [Amazon Polly vs Azure AI Speech text-to-speech](https://www.anchorterminal.com/compare/amazon-polly-vs-azure-text-to-speech.md)\n- [Amazon Polly vs Deepgram Text-to-Speech (Aura-2, Flux TTS)](https://www.anchorterminal.com/compare/amazon-polly-vs-deepgram-tts.md)\n- [Azure AI Speech text-to-speech vs Cartesia Sonic TTS API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-cartesia-tts.md)\n- [Azure AI Speech text-to-speech vs ElevenLabs Text to Speech API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-elevenlabs-tts.md)\n- [Azure AI Speech text-to-speech vs Murf TTS API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-murf-tts.md)\n- [Azure AI Speech text-to-speech vs PlayHT Text-to-Speech API](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-playht-tts.md)\n- [Azure AI Speech text-to-speech vs Resemble AI Text-to-Speech API](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-resemble-ai-tts.md)\n- [Azure AI Speech text-to-speech vs Rime TTS API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-rime-tts.md)\n- [Azure AI Speech text-to-speech vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-soniox-tts.md)\n- [Cartesia Sonic TTS API + MCP vs Deepgram Text-to-Speech (Aura-2, Flux TTS)](https://www.anchorterminal.com/compare/cartesia-tts-vs-deepgram-tts.md)\n- [Deepgram Text-to-Speech (Aura-2, Flux TTS) vs ElevenLabs Text to Speech API + MCP](https://www.anchorterminal.com/compare/deepgram-tts-vs-elevenlabs-tts.md)\n- [Deepgram Text-to-Speech (Aura-2, Flux TTS) vs Murf TTS API + MCP](https://www.anchorterminal.com/compare/deepgram-tts-vs-murf-tts.md)\n- [Deepgram Text-to-Speech (Aura-2, Flux TTS) vs PlayHT Text-to-Speech API](https://www.anchorterminal.com/compare/deepgram-tts-vs-playht-tts.md)\n- [Deepgram Text-to-Speech (Aura-2, Flux TTS) vs Resemble AI Text-to-Speech API](https://www.anchorterminal.com/compare/deepgram-tts-vs-resemble-ai-tts.md)\n- [Deepgram Text-to-Speech (Aura-2, Flux TTS) vs Rime TTS API + MCP](https://www.anchorterminal.com/compare/deepgram-tts-vs-rime-tts.md)\n- [Deepgram Text-to-Speech (Aura-2, Flux TTS) vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/deepgram-tts-vs-soniox-tts.md)\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-04",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.3",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Compare",
        "url": "https://www.anchorterminal.com/compare/"
      },
      {
        "name": "Azure AI Speech text-to-speech vs Deepgram Text-to-Speech (Aura-2, Flux TTS)",
        "url": ""
      }
    ],
    "description": "Azure AI Speech text-to-speech has a score of 73.7 (BB) against Deepgram Text-to-Speech (Aura-2, Flux TTS)'s 73 (BB). Both do speech tts. The largest gap is schema \u0026 documentation, 30 points. Category scores, facts, verdicts and agent notes side by side.",
    "facts": [
      "Azure AI Speech text-to-speech BB 73.7",
      "Deepgram Text-to-Speech (Aura-2, Flux TTS) BB 73",
      "scores"
    ],
    "h1": "Azure AI Speech text-to-speech vs Deepgram Text-to-Speech (Aura-2, Flux TTS)",
    "image": "https://www.anchorterminal.com/assets/og/compare-azure-text-to-speech-vs-deepgram-tts.png",
    "path": "/compare/azure-text-to-speech-vs-deepgram-tts",
    "published": "2026-10-01",
    "section": "tools",
    "title": "Azure AI Speech text-to-speech vs Deepgram Text-to-Speech (Aura-2…",
    "toc": null,
    "updated": "2026-10-04",
    "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-deepgram-tts"
  },
  "tokens": {
    "markdown": 1850,
    "slim": 380
  },
  "version": 1
}
