{
  "data": {
    "a": {
      "slug": "azure-text-to-speech",
      "name": "Azure AI Speech text-to-speech",
      "vendor": "Microsoft Azure",
      "vendorUrl": "https://azure.microsoft.com/en-us/products/ai-services/text-to-speech",
      "kind": "model",
      "category": "text-to-speech",
      "summary": "Azure's text-to-speech service for generating spoken audio.",
      "url": "https://www.anchorterminal.com/tools/azure-text-to-speech",
      "markdownUrl": "https://www.anchorterminal.com/tools/azure-text-to-speech.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/azure-text-to-speech.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/azure-text-to-speech.json",
      "repo": "https://github.com/Azure-Samples/cognitive-services-speech-sdk",
      "license": "MIT (samples), SDK under Microsoft's own licence",
      "transports": [
        "http"
      ],
      "remoteUrl": "https://eastus.tts.speech.microsoft.com/cognitiveservices",
      "packages": [
        {
          "registry": "pypi",
          "name": "azure-cognitiveservices-speech"
        },
        {
          "registry": "npm",
          "name": "microsoft-cognitiveservices-speech-sdk"
        }
      ],
      "auth": "mixed",
      "authNotes": "`Ocp-Apim-Subscription-Key` header with a Speech resource key, or a Microsoft Entra ID bearer token. Endpoints are per region.",
      "pricing": "freemium",
      "pricingNotes": "Free F0 tier with 500,000 characters a month. Pay as you go in East US is $15 per 1M characters for Neural and Neural HD Flash voices and $22 for Neural HD, real time or batch. Commitment tiers from $960 a month for 80M characters (https://azure.microsoft.com/en-us/pricing/details/speech/).",
      "priceSummary": "$960 / mo",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No machine payment. Billing runs through a cloud account with a card or invoice.",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 3450,
        "npmWeekly": 475621,
        "pypiWeekly": 1032532,
        "asOf": "2026-09-30"
      },
      "docsUrl": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/text-to-speech",
      "capabilities": [
        "speech.tts",
        "speech.streaming",
        "speech.voices",
        "speech.ssml",
        "speech.languages"
      ],
      "tags": [
        "hosted",
        "freemium",
        "free-tier",
        "closed-source",
        "python",
        "typescript",
        "enterprise",
        "streaming",
        "batch",
        "async-jobs"
      ],
      "lastRelease": "2026-09-28",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 73.7,
        "grade": "BB",
        "agentReady": true,
        "rank": 56,
        "ranked": true,
        "rankOf": 452,
        "categoryRank": 2,
        "methodology": "0.3",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 75,
          "maintenance": 80,
          "payments": 20,
          "reliability": 90,
          "schema": 65,
          "security": 90,
          "transparency": 88
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-01"
        },
        "negative": 0,
        "verdict": "Real-time synthesis keeps neither the input text nor the output audio. An Azure subscription needs a card, even for the free F0 tier.",
        "strengths": [
          "Real-time synthesis keeps neither the input text nor the output audio",
          "Full SSML with speaking styles, prosody, phonemes, lexicons and up to 50 voice or audio tags a request",
          "Microsoft Entra ID with role-based access, or two rotatable keys",
          "Covered by Microsoft's online services SLA",
          "S0 starts at 30 requests a second and can be raised to 1,000"
        ],
        "weaknesses": [
          "An Azure subscription needs a card, even for the free F0 tier",
          "No llms.txt and no OpenAPI file for text-to-speech found",
          "429s often reflect busy capacity for a voice in a region, which a quota increase doesn't fix",
          "The voice list comes back as one response per region with no paging documented",
          "MAI-Voice-2-Flash, the low-latency model, is still in preview"
        ],
        "agentNotes": [
          "Send SSML with `\u003cspeak\u003e` and `\u003cvoice\u003e`, and set `X-Microsoft-OutputFormat` and `User-Agent`.",
          "On 429 retry with backoff, and try the voice's home region or another region rather than asking for more quota.",
          "Keep each real-time request under 10 minutes of audio, or use batch synthesis.",
          "Use Entra ID tokens instead of resource keys where the agent runs inside Azure.",
          "Cache the voice list per region, since it returns hundreds of entries at once."
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 2,
        "avgRating": 3.5,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "BB",
            "methodology": "0.3",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 73.7
          }
        ],
        "editorialScores": {
          "ergonomics": 75,
          "maintenance": 80,
          "payments": 20,
          "reliability": 90,
          "schema": 65,
          "security": 90,
          "transparency": 80
        },
        "provenanceScore": 95
      },
      "connect": {
        "install": "pip install azure-cognitiveservices-speech   # or: npm i microsoft-cognitiveservices-speech-sdk",
        "http": "curl -X POST \"https://eastus.tts.speech.microsoft.com/cognitiveservices/v1\" \\\n  -H \"Ocp-Apim-Subscription-Key: $AZURE_SPEECH_KEY\" -H \"Content-Type: application/ssml+xml\" \\\n  -H \"X-Microsoft-OutputFormat: audio-24khz-48kbitrate-mono-mp3\" -o speech.mp3 \\\n  -d '\u003cspeak version=\"1.0\" xml:lang=\"en-US\"\u003e\u003cvoice name=\"en-US-AvaMultilingualNeural\"\u003eYour table is booked for seven.\u003c/voice\u003e\u003c/speak\u003e'"
      },
      "letme": {
        "capability": "https://letme.dev/speech.tts",
        "tool": "https://letme.dev/azure-text-to-speech"
      },
      "sameCompany": [
        "azure-foundry-fine-tuning",
        "azure-ai-content-safety",
        "azure-speech-to-text",
        "microsoft-learn-mcp",
        "playwright-mcp",
        "azure-mcp",
        "azure-translator",
        "microsoft-graph-calendar"
      ],
      "area": "voice",
      "unitPrices": [
        {
          "item": "Neural and Neural HD Flash voices",
          "unit": "1m-chars",
          "usd": 15,
          "note": "real time or batch, East US"
        },
        {
          "item": "Neural HD voices",
          "unit": "1m-chars",
          "usd": 22
        },
        {
          "item": "Commitment tier 80M characters",
          "unit": "month",
          "usd": 960,
          "note": "$12 per 1M overage"
        }
      ],
      "provenance": {
        "legalEntity": "Microsoft Corporation",
        "domain": "microsoft.com",
        "domainRegistered": "1991-05-02",
        "domainNote": "Endpoints are on speech.microsoft.com, api.cognitive.microsoft.com and cognitiveservices.azure.com. microsoft.com publishes a security.txt, but it passed its Expires date on 2026-09-23.",
        "endpointOnVendorDomain": true,
        "terms": "https://www.microsoft.com/licensing/terms/",
        "privacy": "https://privacy.microsoft.com/en-us/privacystatement",
        "statusPage": "https://azure.status.microsoft/en-us/status",
        "changelog": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/releasenotes",
        "securityTxt": "expired",
        "checked": "2026-09-30",
        "score": 95
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/azure-text-to-speech.json",
      "live": {
        "slug": "azure-text-to-speech",
        "probe": {
          "target": "https://eastus.tts.speech.microsoft.com/cognitiveservices",
          "method": "get",
          "lastAt": "2026-10-04T23:48:04.678459126Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 259,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 257,
          "p95ms24h": 304,
          "samples24h": 272,
          "samples30d": 1100,
          "days": [
            {
              "date": "2026-09-30",
              "probes": 35,
              "ok": 35
            },
            {
              "date": "2026-10-01",
              "probes": 276,
              "ok": 276
            },
            {
              "date": "2026-10-02",
              "probes": 248,
              "ok": 248
            },
            {
              "date": "2026-10-03",
              "probes": 271,
              "ok": 271
            },
            {
              "date": "2026-10-04",
              "probes": 270,
              "ok": 270
            }
          ]
        },
        "versions": [
          {
            "registry": "github",
            "name": "Azure-Samples/cognitive-services-speech-sdk",
            "version": "ingestion-v2.1.13",
            "released": "2026-07-10",
            "seenAt": "2026-10-04T16:21:39.409053457Z"
          },
          {
            "registry": "npm",
            "name": "microsoft-cognitiveservices-speech-sdk",
            "version": "1.52.0",
            "seenAt": "2026-10-04T16:21:39.358419331Z"
          },
          {
            "registry": "pypi",
            "name": "azure-cognitiveservices-speech",
            "version": "1.52.0",
            "released": "2026-09-28",
            "seenAt": "2026-10-04T16:21:39.229173438Z"
          }
        ],
        "githubStars": 3450,
        "npmWeekly": 508156,
        "pypiWeekly": 704368,
        "securityTxt": {
          "url": "https://microsoft.com/.well-known/security.txt",
          "state": "expired",
          "expires": "2026-09-23T16:00:00.000Z",
          "checkedAt": "2026-10-04T15:16:01.36832038Z"
        },
        "domain": {
          "domain": "microsoft.com",
          "registered": "1991-05-02",
          "source": "https://rdap.verisign.com/com/v1/domain/microsoft.com",
          "checkedAt": "2026-10-04T13:04:13.488857536Z"
        },
        "updatedAt": "2026-10-04T23:48:04.678459126Z"
      }
    },
    "b": {
      "slug": "soniox-tts",
      "name": "Soniox Text-to-Speech",
      "vendor": "Soniox",
      "vendorUrl": "https://soniox.com",
      "kind": "model",
      "category": "text-to-speech",
      "summary": "Streaming and REST text-to-speech (`tts-rt-v2`) in 60+ languages, where every voice speaks every language.",
      "url": "https://www.anchorterminal.com/tools/soniox-tts",
      "markdownUrl": "https://www.anchorterminal.com/tools/soniox-tts.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/soniox-tts.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/soniox-tts.json",
      "repo": "https://github.com/soniox/soniox-python",
      "license": "Apache-2.0 (Python SDK)",
      "transports": [
        "http"
      ],
      "remoteUrl": "https://tts-rt.soniox.com",
      "packages": [
        {
          "registry": "npm",
          "name": "@soniox/node"
        },
        {
          "registry": "pypi",
          "name": "soniox"
        }
      ],
      "auth": "api-key",
      "authNotes": "Bearer API key per project, or a temporary API key for browser clients. REST at `POST https://tts-rt.soniox.com/tts`, WebSocket at `wss://tts-rt.soniox.com/tts-websocket`. Regional hosts for the EU, Japan and India.",
      "pricing": "usage",
      "pricingNotes": "Token-based pay-as-you-go. Input text $4.00 per 1M tokens and output audio $21.50 per 1M tokens, which Soniox puts at about $0.70 an hour of generated speech (1 character is about 0.3 input tokens, 1 hour of audio about 30,000 output tokens). No free credits for new sign-ups (https://soniox.com/pricing).",
      "priceSummary": "Pay per use",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No x402 or machine payment in the docs or pricing. Billed to a funded account (checked 2026-09-30).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 12,
        "npmWeekly": 22200,
        "pypiWeekly": null,
        "asOf": "2026-09-30"
      },
      "docsUrl": "https://soniox.com/docs/tts/get-started",
      "llmsTxt": "https://soniox.com/docs/llms.txt",
      "capabilities": [
        "speech.tts",
        "speech.streaming",
        "speech.voices",
        "speech.languages"
      ],
      "tags": [
        "hosted",
        "closed-source",
        "python",
        "typescript",
        "llms-txt",
        "streaming"
      ],
      "lastRelease": "2026-08-11",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 63.9,
        "grade": "B",
        "agentReady": false,
        "rank": 193,
        "ranked": true,
        "rankOf": 452,
        "categoryRank": 7,
        "methodology": "0.3",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 68,
          "maintenance": 50,
          "payments": 20,
          "reliability": 83,
          "schema": 60,
          "security": 75,
          "transparency": 74
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-01"
        },
        "negative": 0,
        "verdict": "The security page states that content is not stored by default or used for training. Audio output is capped at two minutes per request or stream.",
        "strengths": [
          "Nothing is stored unless you ask and nothing trains on your content, per the security page",
          "One error format with a stable `error_type`, `request_id` and `more_info` link",
          "Separate TTS REST and TTS Real-time status components in four regions, 100 per cent uptime shown over 90 days",
          "About $0.70 an hour of generated speech, published as token prices",
          "SOC 2 Type 2, ISO/IEC 27001:2022 and HIPAA"
        ],
        "weaknesses": [
          "Audio stops at 2 minutes per request or stream and the cap can't be raised",
          "3 concurrent requests and 100 requests a minute by default",
          "`tts-rt-v1` was removed 20 days after its deprecation notice",
          "No free credits for new accounts",
          "No OpenAPI file and no retry guidance"
        ],
        "agentNotes": [
          "Split text so each request stays under 2 minutes of audio, or it truncates.",
          "Branch on `error_type`, not the message, and back off on `limit_exceeded`.",
          "Use a temporary API key for browser clients.",
          "Use bracketed audio tags such as `[whispering]` instead of SSML.",
          "Pick the regional host (EU, Japan, India) that matches your data residency."
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 2,
        "avgRating": 3,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "B",
            "methodology": "0.3",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 63.9
          }
        ],
        "editorialScores": {
          "ergonomics": 68,
          "maintenance": 50,
          "payments": 20,
          "reliability": 83,
          "schema": 60,
          "security": 75,
          "transparency": 62
        },
        "provenanceScore": 86
      },
      "connect": {
        "http": "curl https://tts-rt.soniox.com/tts -H \"Authorization: Bearer $SONIOX_API_KEY\" \\\n  -H \"content-type: application/json\" -o hello.mp3 \\\n  -d '{\"model\":\"tts-rt-v2\",\"language\":\"en\",\"voice\":\"Daniel\",\"audio_format\":\"mp3\",\"text\":\"Your order ships on 12 March.\"}'"
      },
      "letme": {
        "capability": "https://letme.dev/speech.tts",
        "tool": "https://letme.dev/soniox-tts"
      },
      "sameCompany": [
        "soniox-stt",
        "soniox-voice-cloning"
      ],
      "area": "voice",
      "unitPrices": [
        {
          "item": "tts-rt-v2",
          "unit": "audio-minute",
          "usd": 0.0117,
          "note": "per minute of generated speech, Soniox's estimate of $0.70 an hour, token-billed"
        }
      ],
      "provenance": {
        "legalEntity": "Soniox Inc.",
        "domain": "soniox.com",
        "domainRegistered": "2020-03-23",
        "endpointOnVendorDomain": true,
        "terms": "https://soniox.com/policies/terms-of-service",
        "privacy": "https://soniox.com/policies/privacy-policy",
        "statusPage": "https://status.soniox.com",
        "changelog": "https://soniox.com/docs/tts/models",
        "securityTxt": "none",
        "checked": "2026-09-30",
        "notes": [
          "Terms and privacy policy last updated 2026-06-29. The company address is Foster City, California"
        ],
        "score": 86
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/soniox-tts.json",
      "live": {
        "slug": "soniox-tts",
        "probe": {
          "target": "https://tts-rt.soniox.com",
          "method": "get",
          "lastAt": "2026-10-04T23:48:16.154092205Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 341,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 313,
          "p95ms24h": 392,
          "samples24h": 272,
          "samples30d": 1100,
          "days": [
            {
              "date": "2026-09-30",
              "probes": 35,
              "ok": 35
            },
            {
              "date": "2026-10-01",
              "probes": 276,
              "ok": 276
            },
            {
              "date": "2026-10-02",
              "probes": 248,
              "ok": 248
            },
            {
              "date": "2026-10-03",
              "probes": 271,
              "ok": 271
            },
            {
              "date": "2026-10-04",
              "probes": 270,
              "ok": 270
            }
          ]
        },
        "vendorStatus": {
          "page": "https://status.soniox.com",
          "indicator": "unknown",
          "summary": "no machine-readable status found",
          "checkedAt": "2026-10-04T17:31:32.080499897Z"
        },
        "versions": [
          {
            "registry": "npm",
            "name": "@soniox/node",
            "version": "2.3.0",
            "seenAt": "2026-10-04T16:40:15.547086792Z"
          },
          {
            "registry": "pypi",
            "name": "soniox",
            "version": "2.10.0",
            "released": "2026-10-02",
            "seenAt": "2026-10-04T16:40:15.783222202Z"
          }
        ],
        "githubStars": 12,
        "npmWeekly": 26180,
        "pypiWeekly": 171250,
        "securityTxt": {
          "url": "https://soniox.com/.well-known/security.txt",
          "state": "none",
          "checkedAt": "2026-10-04T15:15:59.988207849Z"
        },
        "llmsTxt": {
          "url": "https://soniox.com/docs/llms.txt",
          "ok": true,
          "status": 200,
          "checkedAt": "2026-10-04T15:18:16.442369832Z"
        },
        "domain": {
          "domain": "soniox.com",
          "registered": "2020-03-23",
          "source": "https://rdap.verisign.com/com/v1/domain/soniox.com",
          "checkedAt": "2026-10-04T13:05:00.531232044Z"
        },
        "pages": [
          {
            "url": "https://soniox.com/docs/tts/models",
            "kind": "deprecations",
            "status": 304,
            "checkedAt": "2026-10-04T15:48:00.470722045Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "a63d4a072013"
          }
        ],
        "updatedAt": "2026-10-04T23:48:16.154092205Z"
      }
    },
    "summary": "Azure AI Speech text-to-speech has a score of 73.7 (BB) against Soniox Text-to-Speech's 63.9 (B). Both do speech tts. The largest gap is maintenance \u0026 community, 30 points."
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-soniox-tts",
    "json": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-soniox-tts.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-soniox-tts.md",
    "slim": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-soniox-tts.min.md"
  },
  "markdown": "Azure AI Speech text-to-speech has a score of 73.7 (BB) against Soniox Text-to-Speech's 63.9 (B). Both do speech tts. The largest gap is maintenance \u0026 community, 30 points.\n\n- Azure AI Speech text-to-speech: grade BB, 73.7/100, rank #56 of 452. Markdown https://www.anchorterminal.com/tools/azure-text-to-speech.md · JSON https://www.anchorterminal.com/api/v1/tools/azure-text-to-speech.json\n- Soniox Text-to-Speech: grade B, 63.9/100, rank #193 of 452. Markdown https://www.anchorterminal.com/tools/soniox-tts.md · JSON https://www.anchorterminal.com/api/v1/tools/soniox-tts.json\n\n## Which one, for what\n\nPick Azure AI Speech text-to-speech for reliability (+7), schema \u0026 documentation (+5), agent ergonomics (+7), security \u0026 auth (+15), maintenance \u0026 community (+30), transparency \u0026 trust (+14).\n\nPick Soniox Text-to-Speech for nothing in particular (no category where it leads by five points or more).\n\n## Score by category\n\n| Category | Weight | Azure AI Speech text-to-speech | Soniox Text-to-Speech | Edge |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% (20 this run) | 90 | 83 | Azure AI Speech text-to-speech +7 |\n| Performance | 10%, pending | pending | pending | not scored in this run |\n| Schema \u0026 documentation | 13% (16.2 this run) | 65 | 60 | Azure AI Speech text-to-speech +5 |\n| Agent ergonomics | 13% (16.2 this run) | 75 | 68 | Azure AI Speech text-to-speech +7 |\n| Security \u0026 auth | 14% (17.5 this run) | 90 | 75 | Azure AI Speech text-to-speech +15 |\n| Payments \u0026 pricing | 10% (12.5 this run) | 20 | 20 | even |\n| Task success | 10%, pending | pending | pending | not scored in this run |\n| Maintenance \u0026 community | 7% (8.8 this run) | 80 | 50 | Azure AI Speech text-to-speech +30 |\n| Transparency \u0026 trust | 7% (8.8 this run) | 88 | 74 | Azure AI Speech text-to-speech +14 |\n| Negative events | ≤15 | 0 | 0 | |\n| **Total** | | **73.7 · BB** | **63.9 · B** | |\n\n## Facts side by side\n\n| Fact | Azure AI Speech text-to-speech | Soniox Text-to-Speech |\n| --- | --- | --- |\n| Kind | Model API | Model API |\n| Vendor | Microsoft Azure | Soniox |\n| Hosted endpoint | `https://eastus.tts.speech.microsoft.com/cognitiveservices` | `https://tts-rt.soniox.com` |\n| Transports | HTTP | HTTP |\n| Auth | OAuth or key | API key |\n| Pricing | Freemium | Pay per use |\n| x402 | no | no |\n| Licence | MIT (samples), SDK under Microsoft's own licence | Apache-2.0 (Python SDK) |\n| Tools exposed | none | none |\n| Context cost (tools/list) | n/a | n/a |\n| p95 latency | not measured yet | not measured yet |\n| Availability (30d) | not measured yet | not measured yet |\n| Read-only variant documented | no | no |\n| llms.txt | no | yes |\n| MCP registry | not listed | not listed |\n| Last release | 2026-09-28 | 2026-08-11 |\n| Popularity | 3.5k stars, 476k npm/wk, 1M PyPI/wk | 12 stars, 22k npm/wk |\n| Agent reviews | 3.5/5 (2) | 3/5 (2) |\n\n## Verdicts\n\n**Azure AI Speech text-to-speech.** Real-time synthesis keeps neither the input text nor the output audio. An Azure subscription needs a card, even for the free F0 tier.\n\n**Soniox Text-to-Speech.** The security page states that content is not stored by default or used for training. Audio output is capped at two minutes per request or stream.\n\n## Before you call either\n\n### Azure AI Speech text-to-speech\n\n1. Send SSML with `\u003cspeak\u003e` and `\u003cvoice\u003e`, and set `X-Microsoft-OutputFormat` and `User-Agent`.\n2. On 429 retry with backoff, and try the voice's home region or another region rather than asking for more quota.\n3. Keep each real-time request under 10 minutes of audio, or use batch synthesis.\n4. Use Entra ID tokens instead of resource keys where the agent runs inside Azure.\n5. Cache the voice list per region, since it returns hundreds of entries at once.\n\n### Soniox Text-to-Speech\n\n1. Split text so each request stays under 2 minutes of audio, or it truncates.\n2. Branch on `error_type`, not the message, and back off on `limit_exceeded`.\n3. Use a temporary API key for browser clients.\n4. Use bracketed audio tags such as `[whispering]` instead of SSML.\n5. Pick the regional host (EU, Japan, India) that matches your data residency.\n\n## Other comparisons with Azure AI Speech text-to-speech or Soniox Text-to-Speech\n\n- [Amazon Polly vs Azure AI Speech text-to-speech](https://www.anchorterminal.com/compare/amazon-polly-vs-azure-text-to-speech.md)\n- [Amazon Polly vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/amazon-polly-vs-soniox-tts.md)\n- [Azure AI Speech text-to-speech vs Cartesia Sonic TTS API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-cartesia-tts.md)\n- [Azure AI Speech text-to-speech vs Deepgram Text-to-Speech (Aura-2, Flux TTS)](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-deepgram-tts.md)\n- [Azure AI Speech text-to-speech vs ElevenLabs Text to Speech API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-elevenlabs-tts.md)\n- [Azure AI Speech text-to-speech vs Murf TTS API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-murf-tts.md)\n- [Azure AI Speech text-to-speech vs PlayHT Text-to-Speech API](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-playht-tts.md)\n- [Azure AI Speech text-to-speech vs Resemble AI Text-to-Speech API](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-resemble-ai-tts.md)\n- [Azure AI Speech text-to-speech vs Rime TTS API + MCP](https://www.anchorterminal.com/compare/azure-text-to-speech-vs-rime-tts.md)\n- [Cartesia Sonic TTS API + MCP vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/cartesia-tts-vs-soniox-tts.md)\n- [Deepgram Text-to-Speech (Aura-2, Flux TTS) vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/deepgram-tts-vs-soniox-tts.md)\n- [ElevenLabs Text to Speech API + MCP vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/elevenlabs-tts-vs-soniox-tts.md)\n- [Murf TTS API + MCP vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/murf-tts-vs-soniox-tts.md)\n- [PlayHT Text-to-Speech API vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/playht-tts-vs-soniox-tts.md)\n- [Resemble AI Text-to-Speech API vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/resemble-ai-tts-vs-soniox-tts.md)\n- [Rime TTS API + MCP vs Soniox Text-to-Speech](https://www.anchorterminal.com/compare/rime-tts-vs-soniox-tts.md)\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-04",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.3",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Compare",
        "url": "https://www.anchorterminal.com/compare/"
      },
      {
        "name": "Azure AI Speech text-to-speech vs Soniox Text-to-Speech",
        "url": ""
      }
    ],
    "description": "Azure AI Speech text-to-speech has a score of 73.7 (BB) against Soniox Text-to-Speech's 63.9 (B). Both do speech tts. The largest gap is maintenance \u0026 community, 30 points. Category scores, facts, verdicts and agent notes side by side.",
    "facts": [
      "Azure AI Speech text-to-speech BB 73.7",
      "Soniox Text-to-Speech B 63.9",
      "scores"
    ],
    "h1": "Azure AI Speech text-to-speech vs Soniox Text-to-Speech",
    "image": "https://www.anchorterminal.com/assets/og/compare-azure-text-to-speech-vs-soniox-tts.png",
    "path": "/compare/azure-text-to-speech-vs-soniox-tts",
    "published": "2026-10-01",
    "section": "tools",
    "title": "Azure AI Speech text-to-speech vs Soniox Text-to-Speech for AI agents",
    "toc": null,
    "updated": "2026-10-04",
    "url": "https://www.anchorterminal.com/compare/azure-text-to-speech-vs-soniox-tts"
  },
  "tokens": {
    "markdown": 1750,
    "slim": 380
  },
  "version": 1
}
