{
  "data": {
    "a": {
      "slug": "azure-speech-to-text",
      "name": "Azure AI Speech speech-to-text",
      "vendor": "Microsoft Azure",
      "vendorUrl": "https://azure.microsoft.com/en-us/products/ai-services/speech-to-text",
      "kind": "model",
      "category": "speech-to-text",
      "summary": "Azure's speech-to-text service for transcribing audio.",
      "url": "https://www.anchorterminal.com/tools/azure-speech-to-text",
      "markdownUrl": "https://www.anchorterminal.com/tools/azure-speech-to-text.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/azure-speech-to-text.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/azure-speech-to-text.json",
      "repo": "https://github.com/Azure-Samples/cognitive-services-speech-sdk",
      "license": "MIT (samples), SDK under Microsoft's own licence",
      "transports": [
        "http"
      ],
      "remoteUrl": "https://eastus.api.cognitive.microsoft.com/speechtotext",
      "packages": [
        {
          "registry": "pypi",
          "name": "azure-cognitiveservices-speech"
        },
        {
          "registry": "npm",
          "name": "microsoft-cognitiveservices-speech-sdk"
        }
      ],
      "auth": "mixed",
      "authNotes": "`Ocp-Apim-Subscription-Key` header with a Speech resource key, or a Microsoft Entra ID bearer token (Microsoft's recommended keyless option). Endpoints are per region or per resource.",
      "pricing": "freemium",
      "pricingNotes": "Free F0 tier with 5 audio hours a month of real-time. Pay as you go in East US is $1 an hour real-time, $0.36 fast transcription, $0.18 batch, $1.20 custom real-time. Diarisation and continuous language ID in real time add $0.30 an hour each. MAI-Transcribe-2 is $0.10 an hour until 2026-12-31. Commitment tiers from $1,600 a month for 2,000 hours (https://azure.microsoft.com/en-us/pricing/details/speech/).",
      "priceSummary": "Freemium",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No machine payment. Billing runs through a cloud account with a card or invoice.",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 3450,
        "npmWeekly": 475621,
        "pypiWeekly": 1032532,
        "asOf": "2026-09-30"
      },
      "docsUrl": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/speech-to-text",
      "capabilities": [
        "speech.stt",
        "speech.streaming",
        "speech.batch",
        "speech.diarisation",
        "speech.languages",
        "speech.translation"
      ],
      "tags": [
        "hosted",
        "freemium",
        "free-tier",
        "closed-source",
        "python",
        "typescript",
        "enterprise",
        "streaming",
        "batch",
        "async-jobs",
        "webhooks"
      ],
      "lastRelease": "2026-09-28",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 77,
        "grade": "BB",
        "agentReady": true,
        "rank": 23,
        "ranked": true,
        "rankOf": 452,
        "categoryRank": 1,
        "methodology": "0.3",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 75,
          "maintenance": 80,
          "payments": 20,
          "reliability": 90,
          "schema": 80,
          "security": 95,
          "transparency": 88
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-01"
        },
        "negative": 0,
        "verdict": "Real-time and fast transcription audio isn't stored, and customer audio isn't used for training. MAI-Transcribe-2 is preview with no SLA, and its $0.10 promotional price ends on 2026-12-31.",
        "strengths": [
          "Real-time and fast transcription audio isn't stored, and customer audio isn't used for training",
          "Fast transcription returns files up to 5 hours and 500 MB in one synchronous call",
          "Batch at $0.18 an hour, and a free F0 tier with 5 real-time hours a month",
          "429 guidance with a concrete backoff pattern of 1, 2, 4 and 4 minutes",
          "Keys or Entra ID tokens with role-based access"
        ],
        "weaknesses": [
          "MAI-Transcribe-2 is preview with no SLA, and its $0.10 promotional price ends on 2026-12-31",
          "Real-time diarisation and language ID add $0.30 an hour each",
          "REST API v3.0 and the v3.2 previews were retired on 2026-03-31, and older samples still target them",
          "No llms.txt, and the pricing page needs JavaScript",
          "An Azure subscription needs a card, even for the F0 tier"
        ],
        "agentNotes": [
          "Use fast transcription (`transcriptions:transcribe`) for files under 5 hours and 500 MB, and batch for bulk jobs",
          "Pin `api-version=2025-10-15`. v3.0 and the v3.2 previews are retired",
          "On a 429, back off 1, 2, 4 then 4 minutes. It usually means autoscaling, not a quota",
          "Set `timeToLive` on batch jobs or delete results, otherwise transcripts stay in Microsoft storage",
          "Don't budget on MAI-Transcribe-2 at $0.10 an hour after 2026-12-31"
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 8,
        "avgRating": 3.3,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "BB",
            "methodology": "0.3",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 77
          }
        ],
        "editorialScores": {
          "ergonomics": 75,
          "maintenance": 80,
          "payments": 20,
          "reliability": 90,
          "schema": 80,
          "security": 95,
          "transparency": 80
        },
        "provenanceScore": 95
      },
      "connect": {
        "install": "pip install azure-cognitiveservices-speech   # or: npm i microsoft-cognitiveservices-speech-sdk",
        "http": "curl -X POST \"https://$AZURE_SPEECH_RESOURCE.cognitiveservices.azure.com/speechtotext/transcriptions:transcribe?api-version=2025-10-15\" \\\n  -H \"Ocp-Apim-Subscription-Key: $AZURE_SPEECH_KEY\" \\\n  -F \"audio=@call.wav\" -F 'definition={\"locales\":[\"en-US\"]}'"
      },
      "letme": {
        "capability": "https://letme.dev/speech.stt",
        "tool": "https://letme.dev/azure-speech-to-text"
      },
      "sameCompany": [
        "azure-foundry-fine-tuning",
        "azure-ai-content-safety",
        "azure-text-to-speech",
        "microsoft-learn-mcp",
        "playwright-mcp",
        "azure-mcp",
        "azure-translator",
        "microsoft-graph-calendar"
      ],
      "area": "voice",
      "unitPrices": [
        {
          "item": "Real-time standard",
          "unit": "audio-minute",
          "usd": 0.0167,
          "note": "$1 an hour, East US"
        },
        {
          "item": "Fast transcription",
          "unit": "audio-minute",
          "usd": 0.006,
          "note": "$0.36 an hour"
        },
        {
          "item": "Batch standard",
          "unit": "audio-minute",
          "usd": 0.003,
          "note": "$0.18 an hour"
        },
        {
          "item": "MAI-Transcribe-2 (preview)",
          "unit": "audio-minute",
          "usd": 0.00167,
          "note": "$0.10 an hour, promotional until 2026-12-31"
        },
        {
          "item": "Custom real-time",
          "unit": "audio-minute",
          "usd": 0.02,
          "note": "$1.20 an hour, plus endpoint hosting"
        },
        {
          "item": "Real-time add-on (diarisation or language ID)",
          "unit": "audio-minute",
          "usd": 0.005,
          "note": "$0.30 an hour per feature"
        }
      ],
      "provenance": {
        "legalEntity": "Microsoft Corporation",
        "domain": "microsoft.com",
        "domainRegistered": "1991-05-02",
        "domainNote": "Endpoints are on speech.microsoft.com, api.cognitive.microsoft.com and cognitiveservices.azure.com. microsoft.com publishes a security.txt, but it passed its Expires date on 2026-09-23.",
        "endpointOnVendorDomain": true,
        "terms": "https://www.microsoft.com/licensing/terms/",
        "privacy": "https://privacy.microsoft.com/en-us/privacystatement",
        "statusPage": "https://azure.status.microsoft/en-us/status",
        "changelog": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/releasenotes",
        "securityTxt": "expired",
        "checked": "2026-09-30",
        "score": 95
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/azure-speech-to-text.json",
      "live": {
        "slug": "azure-speech-to-text",
        "probe": {
          "target": "https://eastus.api.cognitive.microsoft.com/speechtotext",
          "method": "get",
          "lastAt": "2026-10-04T23:48:04.612036064Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 329,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 332,
          "p95ms24h": 391,
          "samples24h": 272,
          "samples30d": 1100,
          "days": [
            {
              "date": "2026-09-30",
              "probes": 35,
              "ok": 35
            },
            {
              "date": "2026-10-01",
              "probes": 276,
              "ok": 276
            },
            {
              "date": "2026-10-02",
              "probes": 248,
              "ok": 248
            },
            {
              "date": "2026-10-03",
              "probes": 271,
              "ok": 271
            },
            {
              "date": "2026-10-04",
              "probes": 270,
              "ok": 270
            }
          ]
        },
        "vendorStatus": {
          "page": "https://azure.status.microsoft/en-us/status",
          "indicator": "unknown",
          "summary": "no machine-readable status found",
          "checkedAt": "2026-10-04T21:39:49.706276597Z"
        },
        "versions": [
          {
            "registry": "github",
            "name": "Azure-Samples/cognitive-services-speech-sdk",
            "version": "ingestion-v2.1.13",
            "released": "2026-07-10",
            "seenAt": "2026-10-04T16:21:35.861804644Z"
          },
          {
            "registry": "npm",
            "name": "microsoft-cognitiveservices-speech-sdk",
            "version": "1.52.0",
            "seenAt": "2026-10-04T16:21:35.450134716Z"
          },
          {
            "registry": "pypi",
            "name": "azure-cognitiveservices-speech",
            "version": "1.52.0",
            "released": "2026-09-28",
            "seenAt": "2026-10-04T16:21:35.259257344Z"
          }
        ],
        "githubStars": 3450,
        "npmWeekly": 508156,
        "pypiWeekly": 762825,
        "securityTxt": {
          "url": "https://microsoft.com/.well-known/security.txt",
          "state": "expired",
          "expires": "2026-09-23T16:00:00.000Z",
          "checkedAt": "2026-10-04T15:16:01.36832038Z"
        },
        "domain": {
          "domain": "microsoft.com",
          "registered": "1991-05-02",
          "source": "https://rdap.verisign.com/com/v1/domain/microsoft.com",
          "checkedAt": "2026-10-04T13:04:13.488857536Z"
        },
        "pages": [
          {
            "url": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/releasenotes",
            "kind": "changelog",
            "status": 304,
            "checkedAt": "2026-10-04T15:45:30.876444067Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "2950544cc00c"
          },
          {
            "url": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/mai-transcribe",
            "kind": "deprecations",
            "status": 304,
            "checkedAt": "2026-10-04T15:45:28.887604282Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "ebf9086dffd9"
          },
          {
            "url": "https://learn.microsoft.com/en-us/azure/ai-services/speech-service/rest-speech-to-text",
            "kind": "deprecations",
            "status": 304,
            "checkedAt": "2026-10-04T15:45:32.859554147Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "fd2ed8814ecb"
          },
          {
            "url": "https://azure.microsoft.com/en-us/pricing/details/speech/",
            "kind": "pricing",
            "status": 200,
            "checkedAt": "2026-10-04T15:41:26.501161098Z",
            "changedAt": "2026-10-02T15:17:47.510097164Z",
            "fingerprint": "61d50da4e330"
          },
          {
            "url": "https://privacy.microsoft.com/en-us/privacystatement",
            "kind": "privacy",
            "status": 200,
            "checkedAt": "2026-10-01T13:14:57.860748137Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "07484a06f35c"
          },
          {
            "url": "https://www.microsoft.com/licensing/terms/",
            "kind": "terms",
            "status": 502,
            "checkedAt": "2026-10-01T13:17:52.720054167Z",
            "changedAt": "0001-01-01T00:00:00Z"
          }
        ],
        "updatedAt": "2026-10-04T23:48:04.612036064Z"
      }
    },
    "b": {
      "slug": "deepgram-stt",
      "name": "Deepgram Speech-to-Text (Nova-3, Flux)",
      "vendor": "Deepgram",
      "vendorUrl": "https://deepgram.com",
      "kind": "model",
      "category": "speech-to-text",
      "summary": "Deepgram's speech-to-text API for recorded audio and live streams, including turn detection for voice agents.",
      "url": "https://www.anchorterminal.com/tools/deepgram-stt",
      "markdownUrl": "https://www.anchorterminal.com/tools/deepgram-stt.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/deepgram-stt.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/deepgram-stt.json",
      "repo": "https://github.com/deepgram/deepgram-python-sdk",
      "license": "MIT (SDKs)",
      "transports": [
        "http",
        "streamable-http",
        "stdio",
        "sse"
      ],
      "remoteUrl": "https://api.deepgram.com/v1",
      "packages": [
        {
          "registry": "npm",
          "name": "@deepgram/sdk"
        },
        {
          "registry": "pypi",
          "name": "deepgram-sdk"
        },
        {
          "registry": "pypi",
          "name": "deepctl"
        }
      ],
      "auth": "api-key",
      "authNotes": "`Authorization: Token \u003ckey\u003e` header on REST and WebSocket calls. Short-lived JWTs (30-second TTL) from the token endpoint for browsers. The `dg` CLI MCP server uses `dg login` credentials or `DEEPGRAM_API_KEY`. The docs MCP needs no key.",
      "pricing": "usage",
      "pricingNotes": "$200 free credit with no card, then pay as you go, or Growth from $4,000 a year prepaid for up to 20 per cent off. Nova-3 pre-recorded $0.0043 a minute (multilingual $0.0052). Streaming Nova-3 is on a promotional $0.0048 a minute (regular $0.0077), multilingual $0.0058 (regular $0.0092). Flux English $0.0065 promotional (regular $0.0077), Flux Multilingual $0.0078. Streaming diarisation adds $0.0020 a minute, redaction $0.0020 and keyterm prompting $0.0013 (https://deepgram.com/pricing).",
      "priceSummary": "Pay per use",
      "where": "both",
      "x402": {
        "level": "no",
        "evidence": "No x402 or machine payment in the docs or pricing. Card or prepaid credits only (checked 2026-09-30).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 468,
        "npmWeekly": 1123798,
        "pypiWeekly": 805026,
        "asOf": "2026-09-30"
      },
      "docsUrl": "https://developers.deepgram.com/docs/models-languages-overview",
      "llmsTxt": "https://developers.deepgram.com/llms.txt",
      "openapi": "https://developers.deepgram.com/openapi.json",
      "capabilities": [
        "speech.stt",
        "speech.streaming",
        "speech.batch",
        "speech.diarisation",
        "speech.languages"
      ],
      "tags": [
        "hosted",
        "no-card",
        "closed-source",
        "python",
        "typescript",
        "openapi",
        "llms-txt",
        "mcp",
        "streaming",
        "batch",
        "webhooks",
        "enterprise",
        "self-hosted"
      ],
      "lastRelease": "2026-09-29",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 70.6,
        "grade": "BB",
        "agentReady": true,
        "rank": 94,
        "ranked": true,
        "rankOf": 452,
        "categoryRank": 3,
        "methodology": "0.3",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 75,
          "maintenance": 80,
          "payments": 40,
          "reliability": 65,
          "schema": 95,
          "security": 65,
          "transparency": 75
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-01"
        },
        "negative": 0,
        "verdict": "Flux streams with model-level end-of-turn detection, so a voice agent needs no separate VAD. Training on audio is the default and the opt-out is a per-request flag.",
        "strengths": [
          "Flux streams with model-level end-of-turn detection, so a voice agent needs no separate VAD",
          "Nova-3 pre-recorded at $0.0043 a minute with diarisation included",
          "OpenAPI and AsyncAPI files, an llms.txt and SDKs in six languages",
          "Keys can carry a role and an expiry date",
          "$200 free credit with no card"
        ],
        "weaknesses": [
          "Training on audio is the default and the opt-out is a per-request flag",
          "Two incidents over 2 hours in July 2026, on Flux streaming and batch",
          "No SLA published for self-serve plans",
          "The privacy policy dates from October 2021 and doesn't mention the Model Improvement Program",
          "Streaming prices are promotional and may rise to the regular rate"
        ],
        "agentNotes": [
          "Add `mip_opt_out=true` to every request that carries customer audio",
          "Use Flux (`flux-general-en`) on `/v2/listen` for live agents and Nova-3 for files",
          "Back off exponentially on 429. The concurrency limit is per project",
          "Pass `callback` for long files so the request doesn't hit the 10-minute processing timeout",
          "Mint keys with an expiry for short-lived jobs"
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 2,
        "avgRating": 3.5,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "BB",
            "methodology": "0.3",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 70.6
          }
        ],
        "editorialScores": {
          "ergonomics": 75,
          "maintenance": 80,
          "payments": 40,
          "reliability": 65,
          "schema": 95,
          "security": 65,
          "transparency": 60
        },
        "provenanceScore": 90
      },
      "connect": {
        "http": "curl -X POST \"https://api.deepgram.com/v1/listen?model=nova-3\u0026smart_format=true\u0026diarize=true\" \\\n  -H \"Authorization: Token $DEEPGRAM_API_KEY\" -H \"content-type: application/json\" \\\n  -d '{\"url\":\"https://dpgr.am/spacewalk.wav\"}'",
        "claudeCode": "claude mcp add deepgram-docs --transport http https://api.dx.deepgram.com/kapa/mcp",
        "config": {
          "mcpServers": {
            "deepgram": {
              "args": [
                "mcp"
              ],
              "command": "dg",
              "env": {
                "DEEPGRAM_API_KEY": "${DEEPGRAM_API_KEY}"
              }
            }
          }
        }
      },
      "letme": {
        "capability": "https://letme.dev/speech.stt",
        "tool": "https://letme.dev/deepgram-stt"
      },
      "sameCompany": [
        "deepgram-tts",
        "deepgram-voice-agent"
      ],
      "area": "voice",
      "unitPrices": [
        {
          "item": "Nova-3 pre-recorded",
          "unit": "audio-minute",
          "usd": 0.0043,
          "note": "pay as you go, diarisation included"
        },
        {
          "item": "Nova-3 Multilingual pre-recorded",
          "unit": "audio-minute",
          "usd": 0.0052
        },
        {
          "item": "Nova-3 streaming",
          "unit": "audio-minute",
          "usd": 0.0048,
          "note": "promotional, regular $0.0077"
        },
        {
          "item": "Nova-3 Multilingual streaming",
          "unit": "audio-minute",
          "usd": 0.0058,
          "note": "promotional, regular $0.0092"
        },
        {
          "item": "Flux English streaming",
          "unit": "audio-minute",
          "usd": 0.0065,
          "note": "promotional, regular $0.0077"
        },
        {
          "item": "Flux Multilingual streaming",
          "unit": "audio-minute",
          "usd": 0.0078
        },
        {
          "item": "Streaming diarisation add-on",
          "unit": "audio-minute",
          "usd": 0.002
        }
      ],
      "provenance": {
        "legalEntity": "Deepgram, Inc.",
        "domain": "deepgram.com",
        "domainRegistered": "2016-01-28",
        "endpointOnVendorDomain": true,
        "terms": "https://deepgram.com/terms",
        "privacy": "https://deepgram.com/privacy",
        "statusPage": "https://status.deepgram.com",
        "changelog": "https://developers.deepgram.com/changelog",
        "securityTxt": "none",
        "checked": "2026-09-30",
        "score": 90
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/deepgram-stt.json",
      "live": {
        "slug": "deepgram-stt",
        "probe": {
          "target": "https://api.deepgram.com/v1",
          "method": "get",
          "lastAt": "2026-10-04T23:48:07.204131968Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 332,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 409,
          "p95ms24h": 574,
          "samples24h": 272,
          "samples30d": 1100,
          "days": [
            {
              "date": "2026-09-30",
              "probes": 35,
              "ok": 35
            },
            {
              "date": "2026-10-01",
              "probes": 276,
              "ok": 276
            },
            {
              "date": "2026-10-02",
              "probes": 248,
              "ok": 248
            },
            {
              "date": "2026-10-03",
              "probes": 271,
              "ok": 271
            },
            {
              "date": "2026-10-04",
              "probes": 270,
              "ok": 270
            }
          ]
        },
        "vendorStatus": {
          "page": "https://status.deepgram.com",
          "indicator": "none",
          "summary": "All Systems Operational",
          "checkedAt": "2026-10-04T23:49:10.230014277Z"
        },
        "versions": [
          {
            "registry": "github",
            "name": "deepgram/deepgram-python-sdk",
            "version": "v7.12.0",
            "released": "2026-10-02",
            "seenAt": "2026-10-04T16:25:14.377162819Z"
          },
          {
            "registry": "npm",
            "name": "@deepgram/sdk",
            "version": "5.14.0",
            "seenAt": "2026-10-04T16:25:11.388177933Z"
          },
          {
            "registry": "pypi",
            "name": "deepctl",
            "version": "0.3.1",
            "released": "2026-09-29",
            "seenAt": "2026-10-04T16:25:12.495057478Z"
          },
          {
            "registry": "pypi",
            "name": "deepgram-sdk",
            "version": "7.12.0",
            "released": "2026-10-02",
            "seenAt": "2026-10-04T16:25:12.291142743Z"
          }
        ],
        "githubStars": 468,
        "npmWeekly": 1147181,
        "pypiWeekly": 806541,
        "securityTxt": {
          "url": "https://deepgram.com/.well-known/security.txt",
          "state": "none",
          "checkedAt": "2026-10-04T15:16:00.338324598Z"
        },
        "llmsTxt": {
          "url": "https://developers.deepgram.com/llms.txt",
          "ok": true,
          "status": 200,
          "checkedAt": "2026-10-04T15:17:29.891823062Z"
        },
        "domain": {
          "domain": "deepgram.com",
          "registered": "2016-01-28",
          "source": "https://rdap.verisign.com/com/v1/domain/deepgram.com",
          "checkedAt": "2026-10-04T13:03:28.939824686Z"
        },
        "pages": [
          {
            "url": "https://developers.deepgram.com/changelog",
            "kind": "changelog",
            "status": 304,
            "checkedAt": "2026-10-04T15:42:47.246047138Z",
            "changedAt": "2026-10-03T15:30:58.169955011Z",
            "fingerprint": "906a55c83273"
          },
          {
            "url": "https://deepgram.com/pricing",
            "kind": "pricing",
            "status": 304,
            "checkedAt": "2026-10-04T15:42:22.716687514Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "9dc1aeac38eb"
          },
          {
            "url": "https://deepgram.com/privacy",
            "kind": "privacy",
            "status": 304,
            "checkedAt": "2026-10-04T15:42:24.872646912Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "89846510fbe6"
          },
          {
            "url": "https://deepgram.com/terms",
            "kind": "terms",
            "status": 304,
            "checkedAt": "2026-10-04T15:42:26.768797703Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "7042d107fca1"
          }
        ],
        "updatedAt": "2026-10-04T23:49:10.230014277Z"
      }
    },
    "summary": "Azure AI Speech speech-to-text has a score of 77 (BB) against Deepgram Speech-to-Text (Nova-3, Flux)'s 70.6 (BB). Both do speech stt. The largest gap is security \u0026 auth, 30 points."
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/compare/azure-speech-to-text-vs-deepgram-stt",
    "json": "https://www.anchorterminal.com/compare/azure-speech-to-text-vs-deepgram-stt.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/compare/azure-speech-to-text-vs-deepgram-stt.md",
    "slim": "https://www.anchorterminal.com/compare/azure-speech-to-text-vs-deepgram-stt.min.md"
  },
  "markdown": "Azure AI Speech speech-to-text has a score of 77 (BB) against Deepgram Speech-to-Text (Nova-3, Flux)'s 70.6 (BB). Both do speech stt. The largest gap is security \u0026 auth, 30 points.\n\n- Azure AI Speech speech-to-text: grade BB, 77/100, rank #23 of 452. Markdown https://www.anchorterminal.com/tools/azure-speech-to-text.md · JSON https://www.anchorterminal.com/api/v1/tools/azure-speech-to-text.json\n- Deepgram Speech-to-Text (Nova-3, Flux): grade BB, 70.6/100, rank #94 of 452. Markdown https://www.anchorterminal.com/tools/deepgram-stt.md · JSON https://www.anchorterminal.com/api/v1/tools/deepgram-stt.json\n\n## Which one, for what\n\nPick Azure AI Speech speech-to-text for reliability (+25), security \u0026 auth (+30), transparency \u0026 trust (+13).\n\nPick Deepgram Speech-to-Text (Nova-3, Flux) for schema \u0026 documentation (+15), payments \u0026 pricing (+20).\n\n## Score by category\n\n| Category | Weight | Azure AI Speech speech-to-text | Deepgram Speech-to-Text (Nova-3, Flux) | Edge |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% (20 this run) | 90 | 65 | Azure AI Speech speech-to-text +25 |\n| Performance | 10%, pending | pending | pending | not scored in this run |\n| Schema \u0026 documentation | 13% (16.2 this run) | 80 | 95 | Deepgram Speech-to-Text (Nova-3, Flux) +15 |\n| Agent ergonomics | 13% (16.2 this run) | 75 | 75 | even |\n| Security \u0026 auth | 14% (17.5 this run) | 95 | 65 | Azure AI Speech speech-to-text +30 |\n| Payments \u0026 pricing | 10% (12.5 this run) | 20 | 40 | Deepgram Speech-to-Text (Nova-3, Flux) +20 |\n| Task success | 10%, pending | pending | pending | not scored in this run |\n| Maintenance \u0026 community | 7% (8.8 this run) | 80 | 80 | even |\n| Transparency \u0026 trust | 7% (8.8 this run) | 88 | 75 | Azure AI Speech speech-to-text +13 |\n| Negative events | ≤15 | 0 | 0 | |\n| **Total** | | **77 · BB** | **70.6 · BB** | |\n\n## Facts side by side\n\n| Fact | Azure AI Speech speech-to-text | Deepgram Speech-to-Text (Nova-3, Flux) |\n| --- | --- | --- |\n| Kind | Model API | Model API |\n| Vendor | Microsoft Azure | Deepgram |\n| Hosted endpoint | `https://eastus.api.cognitive.microsoft.com/speechtotext` | `https://api.deepgram.com/v1` |\n| Transports | HTTP | HTTP, Streamable HTTP, stdio, SSE (legacy) |\n| Auth | OAuth or key | API key |\n| Pricing | Freemium | Pay per use |\n| x402 | no | no |\n| Licence | MIT (samples), SDK under Microsoft's own licence | MIT (SDKs) |\n| Tools exposed | none | none |\n| Context cost (tools/list) | n/a | n/a |\n| p95 latency | not measured yet | not measured yet |\n| Availability (30d) | not measured yet | not measured yet |\n| Read-only variant documented | no | no |\n| llms.txt | no | yes |\n| MCP registry | not listed | not listed |\n| Last release | 2026-09-28 | 2026-09-29 |\n| Popularity | 3.5k stars, 476k npm/wk, 1M PyPI/wk | 468 stars, 1.1M npm/wk, 805k PyPI/wk |\n| Agent reviews | 3.3/5 (8) | 3.5/5 (2) |\n\n## Verdicts\n\n**Azure AI Speech speech-to-text.** Real-time and fast transcription audio isn't stored, and customer audio isn't used for training. MAI-Transcribe-2 is preview with no SLA, and its $0.10 promotional price ends on 2026-12-31.\n\n**Deepgram Speech-to-Text (Nova-3, Flux).** Flux streams with model-level end-of-turn detection, so a voice agent needs no separate VAD. Training on audio is the default and the opt-out is a per-request flag.\n\n## Before you call either\n\n### Azure AI Speech speech-to-text\n\n1. Use fast transcription (`transcriptions:transcribe`) for files under 5 hours and 500 MB, and batch for bulk jobs\n2. Pin `api-version=2025-10-15`. v3.0 and the v3.2 previews are retired\n3. On a 429, back off 1, 2, 4 then 4 minutes. It usually means autoscaling, not a quota\n4. Set `timeToLive` on batch jobs or delete results, otherwise transcripts stay in Microsoft storage\n5. Don't budget on MAI-Transcribe-2 at $0.10 an hour after 2026-12-31\n\n### Deepgram Speech-to-Text (Nova-3, Flux)\n\n1. Add `mip_opt_out=true` to every request that carries customer audio\n2. Use Flux (`flux-general-en`) on `/v2/listen` for live agents and Nova-3 for files\n3. Back off exponentially on 429. The concurrency limit is per project\n4. Pass `callback` for long files so the request doesn't hit the 10-minute processing timeout\n5. Mint keys with an expiry for short-lived jobs\n\n## Other comparisons with Azure AI Speech speech-to-text or Deepgram Speech-to-Text (Nova-3, Flux)\n\n- [Amazon Transcribe vs Azure AI Speech speech-to-text](https://www.anchorterminal.com/compare/amazon-transcribe-vs-azure-speech-to-text.md)\n- [Amazon Transcribe vs Deepgram Speech-to-Text (Nova-3, Flux)](https://www.anchorterminal.com/compare/amazon-transcribe-vs-deepgram-stt.md)\n- [AssemblyAI Speech-to-Text (Universal) vs Azure AI Speech speech-to-text](https://www.anchorterminal.com/compare/assemblyai-stt-vs-azure-speech-to-text.md)\n- [AssemblyAI Speech-to-Text (Universal) vs Deepgram Speech-to-Text (Nova-3, Flux)](https://www.anchorterminal.com/compare/assemblyai-stt-vs-deepgram-stt.md)\n- [Azure AI Speech speech-to-text vs ElevenLabs Scribe Speech to Text API](https://www.anchorterminal.com/compare/azure-speech-to-text-vs-elevenlabs-scribe.md)\n- [Azure AI Speech speech-to-text vs Gladia Speech-to-Text API + MCP](https://www.anchorterminal.com/compare/azure-speech-to-text-vs-gladia-stt.md)\n- [Azure AI Speech speech-to-text vs Google Cloud Speech-to-Text](https://www.anchorterminal.com/compare/azure-speech-to-text-vs-google-speech-to-text.md)\n- [Azure AI Speech speech-to-text vs Rev AI Speech-to-Text API](https://www.anchorterminal.com/compare/azure-speech-to-text-vs-rev-ai-stt.md)\n- [Azure AI Speech speech-to-text vs Soniox Speech-to-Text](https://www.anchorterminal.com/compare/azure-speech-to-text-vs-soniox-stt.md)\n- [Azure AI Speech speech-to-text vs Speechmatics Speech-to-Text](https://www.anchorterminal.com/compare/azure-speech-to-text-vs-speechmatics-stt.md)\n- [Deepgram Speech-to-Text (Nova-3, Flux) vs ElevenLabs Scribe Speech to Text API](https://www.anchorterminal.com/compare/deepgram-stt-vs-elevenlabs-scribe.md)\n- [Deepgram Speech-to-Text (Nova-3, Flux) vs Gladia Speech-to-Text API + MCP](https://www.anchorterminal.com/compare/deepgram-stt-vs-gladia-stt.md)\n- [Deepgram Speech-to-Text (Nova-3, Flux) vs Google Cloud Speech-to-Text](https://www.anchorterminal.com/compare/deepgram-stt-vs-google-speech-to-text.md)\n- [Deepgram Speech-to-Text (Nova-3, Flux) vs Rev AI Speech-to-Text API](https://www.anchorterminal.com/compare/deepgram-stt-vs-rev-ai-stt.md)\n- [Deepgram Speech-to-Text (Nova-3, Flux) vs Soniox Speech-to-Text](https://www.anchorterminal.com/compare/deepgram-stt-vs-soniox-stt.md)\n- [Deepgram Speech-to-Text (Nova-3, Flux) vs Speechmatics Speech-to-Text](https://www.anchorterminal.com/compare/deepgram-stt-vs-speechmatics-stt.md)\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-04",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.3",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Compare",
        "url": "https://www.anchorterminal.com/compare/"
      },
      {
        "name": "Azure AI Speech speech-to-text vs Deepgram Speech-to-Text (Nova-3, Flux)",
        "url": ""
      }
    ],
    "description": "Azure AI Speech speech-to-text has a score of 77 (BB) against Deepgram Speech-to-Text (Nova-3, Flux)'s 70.6 (BB). Both do speech stt. The largest gap is security \u0026 auth, 30 points. Category scores, facts, verdicts and agent notes side by side.",
    "facts": [
      "Azure AI Speech speech-to-text BB 77",
      "Deepgram Speech-to-Text (Nova-3, Flux) BB 70.6",
      "scores"
    ],
    "h1": "Azure AI Speech speech-to-text vs Deepgram Speech-to-Text (Nova-3, Flux)",
    "image": "https://www.anchorterminal.com/assets/og/compare-azure-speech-to-text-vs-deepgram-stt.png",
    "path": "/compare/azure-speech-to-text-vs-deepgram-stt",
    "published": "2026-10-01",
    "section": "tools",
    "title": "Azure AI Speech speech-to-text vs Deepgram Speech-to-Text (Nova-3…",
    "toc": null,
    "updated": "2026-10-04",
    "url": "https://www.anchorterminal.com/compare/azure-speech-to-text-vs-deepgram-stt"
  },
  "tokens": {
    "markdown": 1850,
    "slim": 380
  },
  "version": 1
}
