{
  "data": {
    "similar": [
      {
        "grade": "BB",
        "json": "https://www.anchorterminal.com/tools/amazon-transcribe.json",
        "name": "Amazon Transcribe",
        "score": 73.4,
        "shared": [
          "speech.stt",
          "speech.streaming",
          "speech.batch",
          "speech.diarisation",
          "speech.languages"
        ],
        "slug": "amazon-transcribe"
      },
      {
        "grade": "BB",
        "json": "https://www.anchorterminal.com/tools/azure-speech-to-text.json",
        "name": "Azure AI Speech speech-to-text",
        "score": 73,
        "shared": [
          "speech.stt",
          "speech.streaming",
          "speech.batch",
          "speech.diarisation",
          "speech.languages"
        ],
        "slug": "azure-speech-to-text"
      },
      {
        "grade": "BB",
        "json": "https://www.anchorterminal.com/tools/deepgram-stt.json",
        "name": "Deepgram Speech-to-Text (Nova-3, Flux)",
        "score": 70.3,
        "shared": [
          "speech.stt",
          "speech.streaming",
          "speech.batch",
          "speech.diarisation",
          "speech.languages"
        ],
        "slug": "deepgram-stt"
      },
      {
        "grade": "BB",
        "json": "https://www.anchorterminal.com/tools/google-speech-to-text.json",
        "name": "Google Cloud Speech-to-Text",
        "score": 70.2,
        "shared": [
          "speech.stt",
          "speech.streaming",
          "speech.batch",
          "speech.diarisation",
          "speech.languages"
        ],
        "slug": "google-speech-to-text"
      },
      {
        "grade": "B",
        "json": "https://www.anchorterminal.com/tools/gladia-stt.json",
        "name": "Gladia Speech-to-Text API + MCP",
        "score": 69.5,
        "shared": [
          "speech.stt",
          "speech.streaming",
          "speech.batch",
          "speech.diarisation",
          "speech.languages"
        ],
        "slug": "gladia-stt"
      },
      {
        "grade": "B",
        "json": "https://www.anchorterminal.com/tools/elevenlabs-scribe.json",
        "name": "ElevenLabs Scribe Speech to Text API",
        "score": 68.9,
        "shared": [
          "speech.stt",
          "speech.streaming",
          "speech.batch",
          "speech.diarisation",
          "speech.languages"
        ],
        "slug": "elevenlabs-scribe"
      }
    ],
    "tool": {
      "slug": "openai-speech-to-text",
      "name": "OpenAI Speech to Text",
      "vendor": "OpenAI",
      "vendorUrl": "https://openai.com",
      "kind": "model",
      "category": "speech-to-text",
      "summary": "OpenAI's speech-to-text API. It transcribes uploaded audio files through `/v1/audio/transcriptions`, translates recordings into English through `/v1/audio/translations`, and transcribes live audio in Realtime transcription sessions over WebSocket or WebRTC.",
      "url": "https://www.anchorterminal.com/tools/openai-speech-to-text",
      "markdownUrl": "https://www.anchorterminal.com/tools/openai-speech-to-text.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/openai-speech-to-text.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/openai-speech-to-text.json",
      "repo": "https://github.com/openai/openai-python",
      "license": "Proprietary hosted service. The service terms were not read (see open questions). The Python SDK is Apache-2.0 and the OpenAPI document is MIT",
      "transports": [
        "http",
        "websocket"
      ],
      "remoteUrl": "https://api.openai.com/v1",
      "packages": [
        {
          "registry": "pypi",
          "name": "openai"
        }
      ],
      "auth": "api-key",
      "authNotes": "Bearer API key created by a person in the platform console (https://platform.openai.com/settings/organization/api-keys). Projects can carry a model allowlist or denylist and an IP allowlist, and Admin API keys are a separate credential that cannot call the audio endpoints (https://developers.openai.com/api/docs/guides/admin-apis).",
      "pricing": "usage",
      "pricingNotes": "$0.0045 an audio minute for `gpt-transcribe` and $0.017 for `gpt-live-transcribe`, billed from prepaid credits (https://developers.openai.com/api/docs/pricing). The rate limits guide names a Free tier with a $100 monthly usage limit, and the `gpt-transcribe` model page lists limits only from the Build tier, which needs $5 of credit purchases. Whether a new account can transcribe without paying was not established.",
      "priceSummary": "Pay per use",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the transcription guides, the endpoint reference, the pricing page or the OpenAPI document (checked 2026-10-09).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 31785,
        "npmWeekly": null,
        "pypiWeekly": null,
        "asOf": "2026-10-09"
      },
      "docsUrl": "https://developers.openai.com/api/docs/guides/speech-to-text",
      "llmsTxt": "https://developers.openai.com/llms.txt",
      "openapi": "https://github.com/openai/openai-openapi/blob/main/openapi.yaml",
      "capabilities": [
        "speech.stt",
        "speech.streaming",
        "speech.batch",
        "speech.diarisation",
        "speech.languages"
      ],
      "tags": [
        "official",
        "hosted",
        "model",
        "streaming",
        "diarisation",
        "llms-txt",
        "openapi",
        "python",
        "typescript",
        "go",
        "java"
      ],
      "lastRelease": "2026-08-26",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 72.4,
        "grade": "BB",
        "agentReady": true,
        "rank": 106,
        "ranked": true,
        "rankOf": 950,
        "categoryRank": 3,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 78,
          "maintenance": 75,
          "payments": 20,
          "reliability": 80,
          "schema": 88,
          "security": 86,
          "transparency": 61
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "breakdown": [
          {
            "key": "reliability",
            "name": "Reliability",
            "weight": 16,
            "effectiveWeight": 20,
            "score": 80,
            "points": 16,
            "reason": "Hosted reading. status.openai.com (incident.io) lists Audio and Realtime as components of the API group (20). On 9 October 2026 it showed Audio at 100% uptime for July to October 2026. The linked feed holds five resolved incidents since 11 July that name the Audio component, all platform-wide elevated error rates (24 July, two on 25 July, 17 September and 6 October). Their durations were not read, and we score them as minor (20 of 30). The `gpt-transcribe` model page gives 5,000 requests a minute on Build, 10,000 on Launch and 30,000 on Grow (15). The rate limits guide documents `Retry-After` on 429 and 503, the `slow_down` and `server_is_overloaded` codes and backoff with jitter, and transcription is a stateless call (15). No SLA was found in the pages read. The Scale Tier page the guide links is on openai.com, which answered a bot check, so it is unread and scored absent (0). `gpt-transcribe` carries no preview label (10)."
          },
          {
            "key": "performance",
            "name": "Performance",
            "weight": 10,
            "effectiveWeight": 0,
            "pending": true,
            "points": 0,
            "reason": "Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes."
          },
          {
            "key": "schema",
            "name": "Schema \u0026 documentation",
            "weight": 13,
            "effectiveWeight": 16.25,
            "score": 88,
            "points": 14.3,
            "reason": "Model reading. A public OpenAPI 3.1 document in openai/openai-openapi covers `/audio/transcriptions` and `/audio/translations` and names `gpt-transcribe` (25). llms.txt and a Markdown twin of every docs page (10). The transcription overview says which model to use for files, live audio, speaker labels, timestamps and translation, and when not to use the specialised ones (18 of 20). `response_format`, `timestamp_granularities` and `model` are enums in the document. `language` is a plain string, and which formats each model accepts is stated in prose only (12 of 15). Examples in seven languages and curl, and an error codes page with a cause and fix for each. The Markdown twin of the endpoint reference omits the request parameters, its first example names the deprecated `gpt-4o-transcribe`, and the guide lists seven input formats where the document lists nine (11 of 15). A dated changelog and deprecations page. `gpt-transcribe` has one alias and no dated snapshot to pin (12 of 15)."
          },
          {
            "key": "ergonomics",
            "name": "Agent ergonomics",
            "weight": 13,
            "effectiveWeight": 16.25,
            "score": 78,
            "points": 12.68,
            "reason": "API reading, as for the other speech-to-text listings. The default `json` response is the text, detected languages and usage, `text` returns a bare string, and segments, words and speaker labels come only when asked for on the models that have them (20 of 25). There is no transcript store to page through. Files stop at 25 MB, the caller splits longer audio, and the `gpt-transcribe` model page marks the Batch API as not supported (10 of 20). Errors carry `message`, `type`, `param` and `code`, and the error codes page gives a fix for each status (18 of 20). The call is stateless and safe to retry, and the Python SDK retries 408, 409, 429 and 5xx twice by default. No idempotency key applies to this endpoint (15 of 20). `file` and `model` are the only required fields, with guide examples for JavaScript, Python, Go, Java, C# and Ruby SDKs and a CLI (15)."
          },
          {
            "key": "security",
            "name": "Security \u0026 auth",
            "weight": 14,
            "effectiveWeight": 17.5,
            "score": 86,
            "points": 15.05,
            "reason": "Model reading. Bearer keys are created per project. Admin API keys are a separate credential that cannot call non-administration endpoints, a project can be limited to a list of models, and the error codes page documents an IP allowlist and keys without permission for an endpoint. The pages on service accounts and workload identity federation were not read (26 of 30). The data controls page says API data has not been used for training since 1 March 2023 unless the customer opts in (20). Both audio endpoints are listed with no abuse-monitoring retention, no application state and Zero Data Retention eligibility (15). The Admin API has an audit log endpoint for user actions and configuration changes (13 of 15). security.txt is PGP-signed, names a Bugcrowd programme and a coordinated disclosure policy and has no Expires field. Certifications were not read, because no page we read links a trust centre and openai.com answered a bot check (12 of 20)."
          },
          {
            "key": "payments",
            "name": "Payments \u0026 pricing",
            "weight": 10,
            "effectiveWeight": 12.5,
            "score": 20,
            "points": 2.5,
            "reason": "No x402, MPP or L402 (0). $0.0045 an audio minute for `gpt-transcribe`, $0.017 for `gpt-live-transcribe` and $0.006 for `whisper-1` on the public pricing page (20). The rate limits guide names a Free tier with a $100 monthly usage limit, and the `gpt-transcribe` model page lists limits only from Build, which needs $5 of credit purchases. No free transcription allowance or no-card trial was found (0). A person signs up in a browser and creates the key in the console (0)."
          },
          {
            "key": "tasks",
            "name": "Task success",
            "weight": 10,
            "effectiveWeight": 0,
            "pending": true,
            "points": 0,
            "reason": "Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored."
          },
          {
            "key": "maintenance",
            "name": "Maintenance \u0026 community",
            "weight": 7,
            "effectiveWeight": 8.75,
            "score": 75,
            "points": 6.56,
            "reason": "Model reading. The last dated change to transcription is the deprecation notice of 26 August 2026, 44 days before the check, and the release of `gpt-transcribe` and `gpt-live-transcribe` was on 28 July (20 of 30). The deprecations page promises at least six months' notice for generally available models, and the four transcription models got six months to the day (12 of 12). Four of the five file transcription models were deprecated within 90 days, a month after their replacements shipped, and the replacements do not cover diarisation, timestamps, subtitles or translation (2 of 8). The dated changelog has two transcription entries in 90 days among dozens for the API (13 of 15). The Python SDK repository showed 140 open issues and pull requests. Replies were not sampled (5 of 10). The Python SDK 3.26.1 was tagged on 8 October 2026, with 15 tags since 18 September (15). CI, CodeQL and a breaking-change check run in the repository (8 of 10)."
          },
          {
            "key": "transparency",
            "name": "Transparency \u0026 trust",
            "weight": 7,
            "effectiveWeight": 8.75,
            "score": 61,
            "points": 5.34,
            "note": "editorial 62, provenance 59",
            "reason": "The hosted models are closed. The Python SDK is Apache-2.0, the OpenAPI document is MIT, and the SDK says `whisper-1` runs the open-source Whisper V2 model. The service terms were not read, because openai.com answered a bot check (12 of 30). The data controls page states training, retention and application state for each endpoint. The privacy policy and any DPA were not read, so whether they agree with it was not checked (18 of 30). A deprecations page with notice periods by model class, dated shutdowns and replacements (20). Data residency is documented by region and endpoint, with the audio endpoints in all ten listed regions for storage and in the United States and Europe for processing. The sub-processor list the docs link answered a bot check (12 of 20)."
          }
        ],
        "assessment": {
          "date": "2026-10-09",
          "basis": "public evidence",
          "confidence": "medium",
          "notes": {
            "ergonomics": "API reading, as for the other speech-to-text listings. The default `json` response is the text, detected languages and usage, `text` returns a bare string, and segments, words and speaker labels come only when asked for on the models that have them (20 of 25). There is no transcript store to page through. Files stop at 25 MB, the caller splits longer audio, and the `gpt-transcribe` model page marks the Batch API as not supported (10 of 20). Errors carry `message`, `type`, `param` and `code`, and the error codes page gives a fix for each status (18 of 20). The call is stateless and safe to retry, and the Python SDK retries 408, 409, 429 and 5xx twice by default. No idempotency key applies to this endpoint (15 of 20). `file` and `model` are the only required fields, with guide examples for JavaScript, Python, Go, Java, C# and Ruby SDKs and a CLI (15).",
            "maintenance": "Model reading. The last dated change to transcription is the deprecation notice of 26 August 2026, 44 days before the check, and the release of `gpt-transcribe` and `gpt-live-transcribe` was on 28 July (20 of 30). The deprecations page promises at least six months' notice for generally available models, and the four transcription models got six months to the day (12 of 12). Four of the five file transcription models were deprecated within 90 days, a month after their replacements shipped, and the replacements do not cover diarisation, timestamps, subtitles or translation (2 of 8). The dated changelog has two transcription entries in 90 days among dozens for the API (13 of 15). The Python SDK repository showed 140 open issues and pull requests. Replies were not sampled (5 of 10). The Python SDK 3.26.1 was tagged on 8 October 2026, with 15 tags since 18 September (15). CI, CodeQL and a breaking-change check run in the repository (8 of 10).",
            "payments": "No x402, MPP or L402 (0). $0.0045 an audio minute for `gpt-transcribe`, $0.017 for `gpt-live-transcribe` and $0.006 for `whisper-1` on the public pricing page (20). The rate limits guide names a Free tier with a $100 monthly usage limit, and the `gpt-transcribe` model page lists limits only from Build, which needs $5 of credit purchases. No free transcription allowance or no-card trial was found (0). A person signs up in a browser and creates the key in the console (0).",
            "reliability": "Hosted reading. status.openai.com (incident.io) lists Audio and Realtime as components of the API group (20). On 9 October 2026 it showed Audio at 100% uptime for July to October 2026. The linked feed holds five resolved incidents since 11 July that name the Audio component, all platform-wide elevated error rates (24 July, two on 25 July, 17 September and 6 October). Their durations were not read, and we score them as minor (20 of 30). The `gpt-transcribe` model page gives 5,000 requests a minute on Build, 10,000 on Launch and 30,000 on Grow (15). The rate limits guide documents `Retry-After` on 429 and 503, the `slow_down` and `server_is_overloaded` codes and backoff with jitter, and transcription is a stateless call (15). No SLA was found in the pages read. The Scale Tier page the guide links is on openai.com, which answered a bot check, so it is unread and scored absent (0). `gpt-transcribe` carries no preview label (10).",
            "schema": "Model reading. A public OpenAPI 3.1 document in openai/openai-openapi covers `/audio/transcriptions` and `/audio/translations` and names `gpt-transcribe` (25). llms.txt and a Markdown twin of every docs page (10). The transcription overview says which model to use for files, live audio, speaker labels, timestamps and translation, and when not to use the specialised ones (18 of 20). `response_format`, `timestamp_granularities` and `model` are enums in the document. `language` is a plain string, and which formats each model accepts is stated in prose only (12 of 15). Examples in seven languages and curl, and an error codes page with a cause and fix for each. The Markdown twin of the endpoint reference omits the request parameters, its first example names the deprecated `gpt-4o-transcribe`, and the guide lists seven input formats where the document lists nine (11 of 15). A dated changelog and deprecations page. `gpt-transcribe` has one alias and no dated snapshot to pin (12 of 15).",
            "security": "Model reading. Bearer keys are created per project. Admin API keys are a separate credential that cannot call non-administration endpoints, a project can be limited to a list of models, and the error codes page documents an IP allowlist and keys without permission for an endpoint. The pages on service accounts and workload identity federation were not read (26 of 30). The data controls page says API data has not been used for training since 1 March 2023 unless the customer opts in (20). Both audio endpoints are listed with no abuse-monitoring retention, no application state and Zero Data Retention eligibility (15). The Admin API has an audit log endpoint for user actions and configuration changes (13 of 15). security.txt is PGP-signed, names a Bugcrowd programme and a coordinated disclosure policy and has no Expires field. Certifications were not read, because no page we read links a trust centre and openai.com answered a bot check (12 of 20).",
            "transparency": "The hosted models are closed. The Python SDK is Apache-2.0, the OpenAPI document is MIT, and the SDK says `whisper-1` runs the open-source Whisper V2 model. The service terms were not read, because openai.com answered a bot check (12 of 30). The data controls page states training, retention and application state for each endpoint. The privacy policy and any DPA were not read, so whether they agree with it was not checked (18 of 30). A deprecations page with notice periods by model class, dated shutdowns and replacements (20). Data residency is documented by region and endpoint, with the audio endpoints in all ten listed regions for storage and in the United States and Europe for processing. The sub-processor list the docs link answered a bot check (12 of 20)."
          },
          "sources": [
            {
              "what": "file transcription guide (Markdown twin)",
              "url": "https://developers.openai.com/api/docs/guides/speech-to-text",
              "seen": "2026-10-09"
            },
            {
              "what": "transcription overview",
              "url": "https://developers.openai.com/api/docs/guides/transcription",
              "seen": "2026-10-09"
            },
            {
              "what": "realtime transcription guide",
              "url": "https://developers.openai.com/api/docs/guides/realtime-transcription",
              "seen": "2026-10-09"
            },
            {
              "what": "gpt-transcribe model page, price, endpoints and rate limits",
              "url": "https://developers.openai.com/api/docs/models/gpt-transcribe",
              "seen": "2026-10-09"
            },
            {
              "what": "create transcription reference (Markdown twin)",
              "url": "https://developers.openai.com/api/reference/resources/audio/subresources/transcriptions/methods/create",
              "seen": "2026-10-09"
            },
            {
              "what": "pricing, transcription models table",
              "url": "https://developers.openai.com/api/docs/pricing",
              "seen": "2026-10-09"
            },
            {
              "what": "deprecations, notice periods and the 26 August 2026 transcription entry",
              "url": "https://developers.openai.com/api/docs/deprecations",
              "seen": "2026-10-09"
            },
            {
              "what": "API changelog",
              "url": "https://developers.openai.com/api/docs/changelog",
              "seen": "2026-10-09"
            },
            {
              "what": "rate limits guide, usage tiers and retry guidance",
              "url": "https://developers.openai.com/api/docs/guides/rate-limits",
              "seen": "2026-10-09"
            },
            {
              "what": "error codes",
              "url": "https://developers.openai.com/api/docs/guides/error-codes",
              "seen": "2026-10-09"
            },
            {
              "what": "data controls, retention per endpoint and data residency",
              "url": "https://developers.openai.com/api/docs/guides/your-data",
              "seen": "2026-10-09"
            },
            {
              "what": "Admin APIs guide",
              "url": "https://developers.openai.com/api/docs/guides/admin-apis",
              "seen": "2026-10-09"
            },
            {
              "what": "llms.txt and the API docs indexes",
              "url": "https://developers.openai.com/llms.txt",
              "seen": "2026-10-09"
            },
            {
              "what": "OpenAPI document, read from a shallow clone",
              "url": "https://github.com/openai/openai-openapi",
              "seen": "2026-10-09"
            },
            {
              "what": "Python SDK source, changelog, tags and security policy, read from a shallow clone",
              "url": "https://github.com/openai/openai-python",
              "seen": "2026-10-09"
            },
            {
              "what": "status page, component uptime",
              "url": "https://status.openai.com",
              "seen": "2026-10-09"
            },
            {
              "what": "status incident feed linked from the status page",
              "url": "https://status.openai.com/feed.atom",
              "seen": "2026-10-09"
            },
            {
              "what": "security.txt",
              "url": "https://openai.com/.well-known/security.txt",
              "seen": "2026-10-09"
            }
          ],
          "openQuestions": [
            "unchecked: the service terms, privacy policy, DPA and the contracting legal entity. openai.com answered the sub-processor page with a bot check (HTTP 403) on 9 October 2026 and no other openai.com page was requested after that. `provenance.terms` and `provenance.privacy` are left out",
            "unchecked: the sub-processor list at openai.com/policies/sub-processor-list, which answered 403",
            "unchecked: any SLA. The Scale Tier and Reserved Tier pages the rate limits guide links are on openai.com and were not requested",
            "unchecked: certifications and a trust centre. No page read links one",
            "unchecked: whether the Free tier can call `gpt-transcribe` and whether it needs a card. The model page lists limits from Build only",
            "unchecked: the `gpt-live-transcribe`, `whisper-1` and `gpt-4o-transcribe-diarize` model pages, the service account and workload identity pages, and the Realtime sessions reference",
            "unchecked: durations of the five status incidents that name Audio. Only the status page and its linked feed were read",
            "unchecked: the openai.com registration date and npm and PyPI download counts",
            "unchecked: replies on the SDK issue trackers",
            "The deprecation of 26 August 2026 names `gpt-live-transcribe` or `gpt-transcribe` as replacements for `whisper-1` and `gpt-4o-transcribe-diarize`, and neither has diarisation, word timestamps, subtitle output or translation in the docs read. No deduction was taken because nothing has been removed yet and the notice is six months",
            "security.txt has no Expires field. It is recorded as valid because it is signed and names two contacts",
            "The lead named `gpt-4o-transcribe` and `whisper-1` among the product's models. Both are deprecated, and the listing name drops the model list",
            "This product shares its API, key and status page with `openai-api`. If the owner holds listings whose governing terms were not read, this one qualifies",
            "The rate limits guide's retry examples and the reference examples still name models past or near their shutdown dates",
            "Robots.txt answers: developers.openai.com and openai.com 200 and allow the paths read, status.openai.com and api.github.com 404, read as no rules. developers.openai.com received 17 requests against the guide of about fifteen"
          ]
        },
        "negative": 0,
        "verdict": "`gpt-transcribe` costs $0.0045 an audio minute, and the audio endpoints keep no abuse-monitoring logs or application state. Speaker labels, timestamps, subtitles and translation exist only on `whisper-1` and `gpt-4o-transcribe-diarize`, which shut down on 26 February 2027 with no named replacement for those functions.",
        "bestFor": "Suited to plain transcription of recorded files at a low price a minute and to teams already holding an OpenAI key.",
        "strengths": [
          "`gpt-transcribe` is priced at $0.0045 an audio minute on a public page, with per-tier request limits of 5,000, 10,000 and 30,000 a minute",
          "The data controls page lists `/v1/audio/transcriptions` and `/v1/audio/translations` with no training, no abuse-monitoring retention and no stored application state",
          "A public OpenAPI 3.1 document, llms.txt and a Markdown twin of every docs page cover the audio endpoints",
          "The status page has an Audio component, shown at 100% uptime for July to October 2026",
          "Guide examples cover JavaScript, Python, Go, Java, C#, Ruby, a CLI and curl, and `file` and `model` are the only required fields"
        ],
        "weaknesses": [
          "`whisper-1`, `gpt-4o-transcribe`, `gpt-4o-mini-transcribe` and `gpt-4o-transcribe-diarize` were deprecated on 26 August 2026 and shut down on 26 February 2027",
          "The two named replacements return no speaker labels, word timestamps, `srt` or `vtt` output or English translation in the reviewed documentation",
          "Uploads stop at 25 MB, the caller splits longer recordings, and the `gpt-transcribe` model page marks the Batch API as not supported",
          "The Markdown twin of the endpoint reference lists the response fields and omits the request parameters, and its first example names a deprecated model",
          "openai.com answered our reader with a bot check, so the service terms, privacy policy, sub-processor list and any SLA were not read"
        ],
        "agentNotes": [
          "Send `gpt-transcribe` to `POST /v1/audio/transcriptions` for recorded files. Use `languages` (a list), not `language`, and never send both.",
          "Keep each upload at 25 MB or less. Split longer audio between sentences and pass the previous chunk's text in `prompt`.",
          "For speaker labels send `gpt-4o-transcribe-diarize` with `response_format=diarized_json` and `chunking_strategy=auto` for audio over 30 seconds. Plan for its shutdown on 26 February 2027.",
          "Word timestamps, `srt`, `vtt` and `/v1/audio/translations` need `whisper-1`, which cannot stream and shuts down on the same date.",
          "On 429 or 503 wait at least `Retry-After` when present, then back off with jitter. Do not retry `credit_balance_exhausted` or spend-limit errors."
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "BB",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 72.4
          }
        ],
        "editorialScores": {
          "ergonomics": 78,
          "maintenance": 75,
          "payments": 20,
          "reliability": 80,
          "schema": 88,
          "security": 86,
          "transparency": 62
        },
        "provenanceScore": 59
      },
      "connect": {
        "install": "pip install openai",
        "http": "curl --request POST \\\n  --url https://api.openai.com/v1/audio/transcriptions \\\n  --header \"Authorization: Bearer $OPENAI_API_KEY\" \\\n  --header 'Content-Type: multipart/form-data' \\\n  --form file=@/path/to/file/audio.mp3 \\\n  --form model=gpt-transcribe"
      },
      "letme": {
        "capability": "https://letme.dev/speech.stt",
        "tool": "https://letme.dev/openai-speech-to-text"
      },
      "sameCompany": [
        "openai-api",
        "openai-embeddings",
        "openai-guardrails",
        "openai-moderation",
        "openai-image-api",
        "openai-sora",
        "openai-realtime",
        "openai-agents-sdk",
        "openai-decisions-api",
        "openai-codex"
      ],
      "notable": [
        "`gpt-transcribe` is the recommended model for recorded files on `POST /v1/audio/transcriptions`, and `gpt-live-transcribe` for live audio in a Realtime transcription session. Both were released on 28 July 2026 (https://developers.openai.com/api/docs/changelog)",
        "On 26 August 2026 OpenAI deprecated `whisper-1`, `gpt-4o-transcribe`, `gpt-4o-mini-transcribe` and `gpt-4o-transcribe-diarize`, with removal on 26 February 2027 and `gpt-live-transcribe` or `gpt-transcribe` as replacements (https://developers.openai.com/api/docs/deprecations)",
        "The transcription overview sends callers to `gpt-4o-transcribe-diarize` for speaker labels and to `whisper-1` for word timestamps, `srt` and `vtt` subtitles and translation into English. Both are on the deprecation list (https://developers.openai.com/api/docs/guides/transcription)",
        "Files can be up to 25 MB. The guide lists mp3, mp4, mpeg, mpga, m4a, wav and webm, and the OpenAPI document adds flac and ogg",
        "`gpt-transcribe` accepts a free-form `prompt`, `keywords` and a `languages` list, and returns detected languages. `stream=true` returns `transcript.text.delta` events for a completed file without a Realtime session",
        "`gpt-live-transcribe` has no server-side turn detection, so the client commits each audio turn, and it returns no word timestamps, speaker labels or confidence scores (https://developers.openai.com/api/docs/guides/realtime-transcription)",
        "The data controls page lists both audio endpoints as not used for training, with no abuse-monitoring retention, no application state and Zero Data Retention eligibility (https://developers.openai.com/api/docs/guides/your-data)",
        "The same API, key and status page as the OpenAI API listing (`openai-api`), which covers the language models. Every fact here was read afresh on 9 October 2026"
      ],
      "area": "voice",
      "details": [
        {
          "label": "Models",
          "value": "`gpt-transcribe` (files and committed Realtime turns) and `gpt-live-transcribe` (live audio). `whisper-1`, `gpt-4o-transcribe`, `gpt-4o-mini-transcribe` and `gpt-4o-transcribe-diarize` are deprecated and shut down on 26 February 2027"
        },
        {
          "label": "Languages",
          "value": "`languages` takes ISO 639-1 codes, selected ISO 639-3 codes and regional `zh` codes. No count is given for `gpt-transcribe`. The guide says Whisper supports 98 languages"
        },
        {
          "label": "Max audio",
          "value": "25 MB a file. Longer recordings are split by the caller"
        },
        {
          "label": "Input",
          "value": "Multipart `file` in mp3, mp4, mpeg, mpga, m4a, wav or webm per the guide. The OpenAPI document also lists flac and ogg"
        },
        {
          "label": "Output",
          "value": "`json` with text and detected languages, or `text`. `verbose_json`, `srt` and `vtt` on `whisper-1` only, `diarized_json` on `gpt-4o-transcribe-diarize` only"
        },
        {
          "label": "Streaming",
          "value": "`stream=true` on file transcription (not `whisper-1`), and Realtime transcription sessions over WebSocket or WebRTC for live audio"
        },
        {
          "label": "Diarisation",
          "value": "`gpt-4o-transcribe-diarize` only, with up to four known speaker references. Not available in Realtime sessions. The model is deprecated"
        },
        {
          "label": "Timestamps",
          "value": "Word and segment timestamps on `whisper-1` only"
        },
        {
          "label": "Translation",
          "value": "Into English only, through `/v1/audio/translations` with `whisper-1`"
        },
        {
          "label": "Rate limits",
          "value": "`gpt-transcribe` 5,000 requests a minute on Build, 10,000 on Launch and 30,000 on Grow"
        },
        {
          "label": "Batch",
          "value": "The `gpt-transcribe` model page marks the Batch API as not supported"
        },
        {
          "label": "Trains on API data",
          "value": "No, unless the customer opts in, per the data controls page"
        },
        {
          "label": "Data retention",
          "value": "No abuse-monitoring retention and no application state for `/v1/audio/transcriptions` and `/v1/audio/translations`"
        },
        {
          "label": "Data location",
          "value": "Regional storage in ten regions through project settings and prefixed hosts such as `eu.api.openai.com`, on approval through sales. Regional processing in the United States and Europe"
        }
      ],
      "unitPrices": [
        {
          "item": "gpt-transcribe",
          "unit": "audio-minute",
          "usd": 0.0045
        },
        {
          "item": "gpt-live-transcribe (live audio)",
          "unit": "audio-minute",
          "usd": 0.017
        },
        {
          "item": "whisper-1 (deprecated)",
          "unit": "audio-minute",
          "usd": 0.006
        },
        {
          "item": "gpt-4o-transcribe-diarize (deprecated, estimated from token prices)",
          "unit": "audio-minute",
          "usd": 0.006
        }
      ],
      "provenance": {
        "legalEntity": "",
        "domain": "openai.com",
        "domainRegistered": "",
        "domainNote": "openai.com answered our researcher with a bot check on 9 October 2026, so the terms and privacy policy were not read on that day. The links are the two documents OpenAI's other listings here carry. openai.com answers our policy reader with HTTP 403 as well, so neither document has been read and both are recorded as unreadable. security.txt is PGP-signed with Bugcrowd and email contacts and has no Expires field.",
        "endpointOnVendorDomain": true,
        "terms": "https://openai.com/policies/services-agreement/",
        "privacy": "https://openai.com/policies/privacy-policy/",
        "statusPage": "https://status.openai.com",
        "changelog": "https://developers.openai.com/api/docs/changelog",
        "securityTxt": "valid",
        "checked": "2026-10-09",
        "score": 59,
        "checks": [
          {
            "check": "Legal entity named",
            "value": "not found",
            "points": 0,
            "max": 20,
            "state": "no"
          },
          {
            "check": "Domain age",
            "value": "openai.com, no registry record we could read",
            "points": 0,
            "max": 15,
            "state": "no"
          },
          {
            "check": "Endpoint on the vendor's domain",
            "value": "api.openai.com",
            "points": 15,
            "max": 15,
            "state": "ok"
          },
          {
            "check": "Terms of service",
            "value": "published, but our reader couldn't read it",
            "points": 7,
            "max": 10,
            "state": "part"
          },
          {
            "check": "Privacy policy",
            "value": "published, but our reader couldn't read it",
            "points": 7,
            "max": 10,
            "state": "part"
          },
          {
            "check": "Status page",
            "value": "status.openai.com",
            "points": 10,
            "max": 10,
            "state": "ok"
          },
          {
            "check": "Changelog",
            "value": "published",
            "points": 10,
            "max": 10,
            "state": "ok"
          },
          {
            "check": "security.txt",
            "value": "valid",
            "points": 10,
            "max": 10,
            "state": "ok"
          }
        ],
        "policies": [
          {
            "kind": "terms",
            "url": "https://openai.com/policies/services-agreement/",
            "state": "unreadable",
            "reason": "the page answered HTTP 403 to our reader",
            "readAt": "2026-10-08",
            "points": 7,
            "max": 10
          },
          {
            "kind": "privacy",
            "url": "https://openai.com/policies/privacy-policy/",
            "state": "unreadable",
            "reason": "the page answered HTTP 403 to our reader",
            "readAt": "2026-10-08",
            "points": 7,
            "max": 10
          }
        ]
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/openai-speech-to-text.json",
      "live": {
        "slug": "openai-speech-to-text",
        "probe": {
          "target": "https://api.openai.com/v1",
          "method": "get",
          "lastAt": "2026-10-10T02:07:19.035834508Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 126,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 139,
          "p95ms24h": 170,
          "samples24h": 107,
          "samples30d": 107,
          "days": [
            {
              "date": "2026-10-09",
              "probes": 85,
              "ok": 85
            },
            {
              "date": "2026-10-10",
              "probes": 22,
              "ok": 22
            }
          ]
        },
        "vendorStatus": {
          "page": "https://status.openai.com",
          "indicator": "minor",
          "summary": "Partial System Degradation",
          "checkedAt": "2026-10-10T02:06:19.003875298Z"
        },
        "versions": [
          {
            "registry": "github",
            "name": "openai/openai-python",
            "version": "v3.27.0",
            "released": "2026-10-09",
            "seenAt": "2026-10-09T17:10:31.877366615Z"
          },
          {
            "registry": "pypi",
            "name": "openai",
            "version": "3.27.0",
            "released": "2026-10-09",
            "seenAt": "2026-10-09T17:10:31.73110612Z"
          }
        ],
        "githubStars": 31787,
        "pypiWeekly": 74761714,
        "updatedAt": "2026-10-10T02:07:19.035834508Z"
      }
    },
    "verify": {
      "accepts": "a page on openai.com or one of its subdomains, or the README of github.com/openai/openai-python",
      "badgeUrl": "https://www.anchorterminal.com/badges/openai-speech-to-text.svg",
      "body": {
        "slug": "openai-speech-to-text",
        "url": "the page with the badge or the link"
      },
      "docs": "https://www.anchorterminal.com/builders/#verify",
      "effect": "none, it never changes a grade, rank or review",
      "endpoint": "https://www.anchorterminal.com/api/v1/verify",
      "listingUrl": "https://www.anchorterminal.com/tools/openai-speech-to-text",
      "mcpTool": "verify_listing",
      "recheck": "weekly; two failed checks in a row and it lapses, a later pass restores it",
      "snippets": {
        "html": "\u003ca href=\"https://www.anchorterminal.com/tools/openai-speech-to-text\"\u003e\u003cimg src=\"https://www.anchorterminal.com/badges/openai-speech-to-text.svg\" alt=\"OpenAI Speech to Text on Anchor Terminal\" height=\"20\"\u003e\u003c/a\u003e",
        "markdown": "[![OpenAI Speech to Text on Anchor Terminal](https://www.anchorterminal.com/badges/openai-speech-to-text.svg)](https://www.anchorterminal.com/tools/openai-speech-to-text)",
        "link": "\u003ca href=\"https://www.anchorterminal.com/tools/openai-speech-to-text\"\u003eOpenAI Speech to Text on Anchor Terminal\u003c/a\u003e"
      }
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/tools/openai-speech-to-text",
    "json": "https://www.anchorterminal.com/tools/openai-speech-to-text.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/tools/openai-speech-to-text.md",
    "slim": "https://www.anchorterminal.com/tools/openai-speech-to-text.min.md"
  },
  "markdown": "## Overview\n\n**Grade BB · 72.4/100 · rank #106 of 950 · #3 in Speech-to-text · agent-ready · confidence medium**\n\n\nMore from OpenAI, listed separately because each is its own product: [OpenAI API](https://www.anchorterminal.com/tools/openai-api.md) (Model APIs \u0026 inference), [OpenAI embeddings](https://www.anchorterminal.com/tools/openai-embeddings.md) (Embeddings \u0026 rerankers), [OpenAI Guardrails](https://www.anchorterminal.com/tools/openai-guardrails.md) (Guardrails \u0026 safety filters), [OpenAI Moderation API](https://www.anchorterminal.com/tools/openai-moderation.md) (Guardrails \u0026 safety filters), [OpenAI Image API](https://www.anchorterminal.com/tools/openai-image-api.md) (Image generation), [OpenAI Sora API](https://www.anchorterminal.com/tools/openai-sora.md) (Video generation), [OpenAI Realtime API](https://www.anchorterminal.com/tools/openai-realtime.md) (Conversational voice agents), [OpenAI Agents SDK](https://www.anchorterminal.com/tools/openai-agents-sdk.md) (Agent frameworks \u0026 SDKs), [OpenAI Decisions API](https://www.anchorterminal.com/tools/openai-decisions-api.md) (Decision models), [OpenAI Codex](https://www.anchorterminal.com/tools/openai-codex.md) (Agent harnesses).\n\n## Assessment\n\n`gpt-transcribe` costs $0.0045 an audio minute, and the audio endpoints keep no abuse-monitoring logs or application state. Speaker labels, timestamps, subtitles and translation exist only on `whisper-1` and `gpt-4o-transcribe-diarize`, which shut down on 26 February 2027 with no named replacement for those functions.\n\n## Facts\n\n| Field | Value |\n| --- | --- |\n| Vendor | OpenAI (https://openai.com) |\n| Kind | Model API |\n| Category | Speech-to-text (https://www.anchorterminal.com/categories/speech-to-text) |\n| Transport | HTTP, websocket |\n| Endpoint | `https://api.openai.com/v1` |\n| Auth | API key · Bearer API key created by a person in the platform console (https://platform.openai.com/settings/organization/api-keys). Projects can carry a model allowlist or denylist and an IP allowlist, and Admin API keys are a separate credential that cannot call the audio endpoints (https://developers.openai.com/api/docs/guides/admin-apis). |\n| Pricing | Pay per use (Pay per use) · $0.0045 an audio minute for `gpt-transcribe` and $0.017 for `gpt-live-transcribe`, billed from prepaid credits (https://developers.openai.com/api/docs/pricing). The rate limits guide names a Free tier with a $100 monthly usage limit, and the `gpt-transcribe` model page lists limits only from the Build tier, which needs $5 of credit purchases. Whether a new account can transcribe without paying was not established. |\n| x402 | No · No x402, MPP or L402 in the transcription guides, the endpoint reference, the pricing page or the OpenAPI document (checked 2026-10-09). |\n| Licence | Proprietary hosted service. The service terms were not read (see open questions). The Python SDK is Apache-2.0 and the OpenAPI document is MIT |\n| Packages | pypi: `openai` |\n| Source | https://github.com/openai/openai-python |\n| Docs | https://developers.openai.com/api/docs/guides/speech-to-text |\n| llms.txt | https://developers.openai.com/llms.txt |\n| Last release | 2026-08-26 |\n| GitHub stars | 31,785 (as of 2026-10-09) |\n| Models | `gpt-transcribe` (files and committed Realtime turns) and `gpt-live-transcribe` (live audio). `whisper-1`, `gpt-4o-transcribe`, `gpt-4o-mini-transcribe` and `gpt-4o-transcribe-diarize` are deprecated and shut down on 26 February 2027 |\n| Languages | `languages` takes ISO 639-1 codes, selected ISO 639-3 codes and regional `zh` codes. No count is given for `gpt-transcribe`. The guide says Whisper supports 98 languages |\n| Max audio | 25 MB a file. Longer recordings are split by the caller |\n| Input | Multipart `file` in mp3, mp4, mpeg, mpga, m4a, wav or webm per the guide. The OpenAPI document also lists flac and ogg |\n| Output | `json` with text and detected languages, or `text`. `verbose_json`, `srt` and `vtt` on `whisper-1` only, `diarized_json` on `gpt-4o-transcribe-diarize` only |\n| Streaming | `stream=true` on file transcription (not `whisper-1`), and Realtime transcription sessions over WebSocket or WebRTC for live audio |\n| Diarisation | `gpt-4o-transcribe-diarize` only, with up to four known speaker references. Not available in Realtime sessions. The model is deprecated |\n| Timestamps | Word and segment timestamps on `whisper-1` only |\n| Translation | Into English only, through `/v1/audio/translations` with `whisper-1` |\n| Rate limits | `gpt-transcribe` 5,000 requests a minute on Build, 10,000 on Launch and 30,000 on Grow |\n| Batch | The `gpt-transcribe` model page marks the Batch API as not supported |\n| Trains on API data | No, unless the customer opts in, per the data controls page |\n| Data retention | No abuse-monitoring retention and no application state for `/v1/audio/transcriptions` and `/v1/audio/translations` |\n| Data location | Regional storage in ten regions through project settings and prefixed hosts such as `eu.api.openai.com`, on approval through sales. Regional processing in the United States and Europe |\n| Capabilities | speech.stt, speech.streaming, speech.batch, speech.diarisation, speech.languages |\n| Tags | official, hosted, model, streaming, diarisation, llms-txt, openapi, python, typescript, go, java |\n| JSON | https://www.anchorterminal.com/api/v1/tools/openai-speech-to-text.json |\n\n## Score breakdown (methodology v0.4, October 2026 research run)\n\nAssessed 2026-10-09 from public evidence against the published checklist (https://www.anchorterminal.com/benchmark/#checklist). Confidence: medium. Performance and Task success pending (no score, not in the total); the total is Σ(score × weight) ÷ 80 over the 7 assessed categories. \"This run\" is each category's share of the 100 points.\n\n| Category | Weight | This run | Score (0–100) | Points |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% | 20 | 80 | 16.0 |\n| Performance | 10% | pending | pending | n/a |\n| Schema \u0026 documentation | 13% | 16.2 | 88 | 14.3 |\n| Agent ergonomics | 13% | 16.2 | 78 | 12.7 |\n| Security \u0026 auth | 14% | 17.5 | 86 | 15.1 |\n| Payments \u0026 pricing | 10% | 12.5 | 20 | 2.5 |\n| Task success | 10% | pending | pending | n/a |\n| Maintenance \u0026 community | 7% | 8.8 | 75 | 6.6 |\n| Transparency \u0026 trust (editorial 62, provenance 59) | 7% | 8.8 | 61 | 5.3 |\n| Negative events | up to −15 | up to −15 | none recorded | 0 |\n| **Total** | | | | **72.4 → BB** |\n\n### Why each score\n\n- Reliability 80: Hosted reading. status.openai.com (incident.io) lists Audio and Realtime as components of the API group (20). On 9 October 2026 it showed Audio at 100% uptime for July to October 2026. The linked feed holds five resolved incidents since 11 July that name the Audio component, all platform-wide elevated error rates (24 July, two on 25 July, 17 September and 6 October). Their durations were not read, and we score them as minor (20 of 30). The `gpt-transcribe` model page gives 5,000 requests a minute on Build, 10,000 on Launch and 30,000 on Grow (15). The rate limits guide documents `Retry-After` on 429 and 503, the `slow_down` and `server_is_overloaded` codes and backoff with jitter, and transcription is a stateless call (15). No SLA was found in the pages read. The Scale Tier page the guide links is on openai.com, which answered a bot check, so it is unread and scored absent (0). `gpt-transcribe` carries no preview label (10).\n- Performance: Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes.\n- Schema \u0026 documentation 88: Model reading. A public OpenAPI 3.1 document in openai/openai-openapi covers `/audio/transcriptions` and `/audio/translations` and names `gpt-transcribe` (25). llms.txt and a Markdown twin of every docs page (10). The transcription overview says which model to use for files, live audio, speaker labels, timestamps and translation, and when not to use the specialised ones (18 of 20). `response_format`, `timestamp_granularities` and `model` are enums in the document. `language` is a plain string, and which formats each model accepts is stated in prose only (12 of 15). Examples in seven languages and curl, and an error codes page with a cause and fix for each. The Markdown twin of the endpoint reference omits the request parameters, its first example names the deprecated `gpt-4o-transcribe`, and the guide lists seven input formats where the document lists nine (11 of 15). A dated changelog and deprecations page. `gpt-transcribe` has one alias and no dated snapshot to pin (12 of 15).\n- Agent ergonomics 78: API reading, as for the other speech-to-text listings. The default `json` response is the text, detected languages and usage, `text` returns a bare string, and segments, words and speaker labels come only when asked for on the models that have them (20 of 25). There is no transcript store to page through. Files stop at 25 MB, the caller splits longer audio, and the `gpt-transcribe` model page marks the Batch API as not supported (10 of 20). Errors carry `message`, `type`, `param` and `code`, and the error codes page gives a fix for each status (18 of 20). The call is stateless and safe to retry, and the Python SDK retries 408, 409, 429 and 5xx twice by default. No idempotency key applies to this endpoint (15 of 20). `file` and `model` are the only required fields, with guide examples for JavaScript, Python, Go, Java, C# and Ruby SDKs and a CLI (15).\n- Security \u0026 auth 86: Model reading. Bearer keys are created per project. Admin API keys are a separate credential that cannot call non-administration endpoints, a project can be limited to a list of models, and the error codes page documents an IP allowlist and keys without permission for an endpoint. The pages on service accounts and workload identity federation were not read (26 of 30). The data controls page says API data has not been used for training since 1 March 2023 unless the customer opts in (20). Both audio endpoints are listed with no abuse-monitoring retention, no application state and Zero Data Retention eligibility (15). The Admin API has an audit log endpoint for user actions and configuration changes (13 of 15). security.txt is PGP-signed, names a Bugcrowd programme and a coordinated disclosure policy and has no Expires field. Certifications were not read, because no page we read links a trust centre and openai.com answered a bot check (12 of 20).\n- Payments \u0026 pricing 20: No x402, MPP or L402 (0). $0.0045 an audio minute for `gpt-transcribe`, $0.017 for `gpt-live-transcribe` and $0.006 for `whisper-1` on the public pricing page (20). The rate limits guide names a Free tier with a $100 monthly usage limit, and the `gpt-transcribe` model page lists limits only from Build, which needs $5 of credit purchases. No free transcription allowance or no-card trial was found (0). A person signs up in a browser and creates the key in the console (0).\n- Task success: Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored.\n- Maintenance \u0026 community 75: Model reading. The last dated change to transcription is the deprecation notice of 26 August 2026, 44 days before the check, and the release of `gpt-transcribe` and `gpt-live-transcribe` was on 28 July (20 of 30). The deprecations page promises at least six months' notice for generally available models, and the four transcription models got six months to the day (12 of 12). Four of the five file transcription models were deprecated within 90 days, a month after their replacements shipped, and the replacements do not cover diarisation, timestamps, subtitles or translation (2 of 8). The dated changelog has two transcription entries in 90 days among dozens for the API (13 of 15). The Python SDK repository showed 140 open issues and pull requests. Replies were not sampled (5 of 10). The Python SDK 3.26.1 was tagged on 8 October 2026, with 15 tags since 18 September (15). CI, CodeQL and a breaking-change check run in the repository (8 of 10).\n- Transparency \u0026 trust 61: The hosted models are closed. The Python SDK is Apache-2.0, the OpenAPI document is MIT, and the SDK says `whisper-1` runs the open-source Whisper V2 model. The service terms were not read, because openai.com answered a bot check (12 of 30). The data controls page states training, retention and application state for each endpoint. The privacy policy and any DPA were not read, so whether they agree with it was not checked (18 of 30). A deprecations page with notice periods by model class, dated shutdowns and replacements (20). Data residency is documented by region and endpoint, with the audio endpoints in all ten listed regions for storage and in the United States and Europe for processing. The sub-processor list the docs link answered a bot check (12 of 20).\n\nFix list for a coding agent, everything this grade says the listing lacks, the biggest gain first (26 items): https://www.anchorterminal.com/fixes/openai-speech-to-text.md (JSON https://www.anchorterminal.com/fixes/openai-speech-to-text.json)\n\n### What we couldn't check\n\n- unchecked: the service terms, privacy policy, DPA and the contracting legal entity. openai.com answered the sub-processor page with a bot check (HTTP 403) on 9 October 2026 and no other openai.com page was requested after that. `provenance.terms` and `provenance.privacy` are left out\n- unchecked: the sub-processor list at openai.com/policies/sub-processor-list, which answered 403\n- unchecked: any SLA. The Scale Tier and Reserved Tier pages the rate limits guide links are on openai.com and were not requested\n- unchecked: certifications and a trust centre. No page read links one\n- unchecked: whether the Free tier can call `gpt-transcribe` and whether it needs a card. The model page lists limits from Build only\n- unchecked: the `gpt-live-transcribe`, `whisper-1` and `gpt-4o-transcribe-diarize` model pages, the service account and workload identity pages, and the Realtime sessions reference\n- unchecked: durations of the five status incidents that name Audio. Only the status page and its linked feed were read\n- unchecked: the openai.com registration date and npm and PyPI download counts\n- unchecked: replies on the SDK issue trackers\n- The deprecation of 26 August 2026 names `gpt-live-transcribe` or `gpt-transcribe` as replacements for `whisper-1` and `gpt-4o-transcribe-diarize`, and neither has diarisation, word timestamps, subtitle output or translation in the docs read. No deduction was taken because nothing has been removed yet and the notice is six months\n- security.txt has no Expires field. It is recorded as valid because it is signed and names two contacts\n- The lead named `gpt-4o-transcribe` and `whisper-1` among the product's models. Both are deprecated, and the listing name drops the model list\n- This product shares its API, key and status page with `openai-api`. If the owner holds listings whose governing terms were not read, this one qualifies\n- The rate limits guide's retry examples and the reference examples still name models past or near their shutdown dates\n- Robots.txt answers: developers.openai.com and openai.com 200 and allow the paths read, status.openai.com and api.github.com 404, read as no rules. developers.openai.com received 17 requests against the guide of about fifteen\n\n### Sources\n\n- file transcription guide (Markdown twin): \u003chttps://developers.openai.com/api/docs/guides/speech-to-text\u003e (seen 2026-10-09)\n- transcription overview: \u003chttps://developers.openai.com/api/docs/guides/transcription\u003e (seen 2026-10-09)\n- realtime transcription guide: \u003chttps://developers.openai.com/api/docs/guides/realtime-transcription\u003e (seen 2026-10-09)\n- gpt-transcribe model page, price, endpoints and rate limits: \u003chttps://developers.openai.com/api/docs/models/gpt-transcribe\u003e (seen 2026-10-09)\n- create transcription reference (Markdown twin): \u003chttps://developers.openai.com/api/reference/resources/audio/subresources/transcriptions/methods/create\u003e (seen 2026-10-09)\n- pricing, transcription models table: \u003chttps://developers.openai.com/api/docs/pricing\u003e (seen 2026-10-09)\n- deprecations, notice periods and the 26 August 2026 transcription entry: \u003chttps://developers.openai.com/api/docs/deprecations\u003e (seen 2026-10-09)\n- API changelog: \u003chttps://developers.openai.com/api/docs/changelog\u003e (seen 2026-10-09)\n- rate limits guide, usage tiers and retry guidance: \u003chttps://developers.openai.com/api/docs/guides/rate-limits\u003e (seen 2026-10-09)\n- error codes: \u003chttps://developers.openai.com/api/docs/guides/error-codes\u003e (seen 2026-10-09)\n- data controls, retention per endpoint and data residency: \u003chttps://developers.openai.com/api/docs/guides/your-data\u003e (seen 2026-10-09)\n- Admin APIs guide: \u003chttps://developers.openai.com/api/docs/guides/admin-apis\u003e (seen 2026-10-09)\n- llms.txt and the API docs indexes: \u003chttps://developers.openai.com/llms.txt\u003e (seen 2026-10-09)\n- OpenAPI document, read from a shallow clone: \u003chttps://github.com/openai/openai-openapi\u003e (seen 2026-10-09)\n- Python SDK source, changelog, tags and security policy, read from a shallow clone: \u003chttps://github.com/openai/openai-python\u003e (seen 2026-10-09)\n- status page, component uptime: \u003chttps://status.openai.com\u003e (seen 2026-10-09)\n- status incident feed linked from the status page: \u003chttps://status.openai.com/feed.atom\u003e (seen 2026-10-09)\n- security.txt: \u003chttps://openai.com/.well-known/security.txt\u003e (seen 2026-10-09)\n\n## Who's behind it (provenance 59/100, checked 2026-10-09)\n\n| Check | Finding | Points |\n| --- | --- | --- |\n| Legal entity named | not found | 0/20 |\n| Domain age | openai.com, no registry record we could read | 0/15 |\n| Endpoint on the vendor's domain | api.openai.com | 15/15 |\n| Terms of service | published, but our reader couldn't read it | 7/10 |\n| Privacy policy | published, but our reader couldn't read it | 7/10 |\n| Status page | status.openai.com | 10/10 |\n| Changelog | published | 10/10 |\n| security.txt | valid | 10/10 |\n\nopenai.com answered our researcher with a bot check on 9 October 2026, so the terms and privacy policy were not read on that day. The links are the two documents OpenAI's other listings here carry. openai.com answers our policy reader with HTTP 403 as well, so neither document has been read and both are recorded as unreadable. security.txt is PGP-signed with Bugcrowd and email contacts and has no Expires field.\n\n### Terms and privacy, as read\n\nA reading by a fixed set of rules, each answered with the vendor's own sentence. Not legal advice.\n\n**Terms of service** (https://openai.com/policies/services-agreement/), read 2026-10-08. Our reader couldn't read it (the page answered HTTP 403 to our reader).\n\n\n**Privacy policy** (https://openai.com/policies/privacy-policy/), read 2026-10-08. Our reader couldn't read it (the page answered HTTP 403 to our reader).\n\n\n## Live (updated 2026-10-10 02:07 UTC)\n\n- Right now: up, HTTP 404, 126 ms, checked 2026-10-10 02:07 UTC (get on `https://api.openai.com/v1`)\n- Uptime 24h 100.0% (107 probes) · 30 days 100.0% (107 probes) · p50 139 ms · p95 170 ms\n- Vendor status page: minor, Partial System Degradation\n- github `openai/openai-python` v3.27.0, released 2026-10-09\n- pypi `openai` 3.27.0, released 2026-10-09\n- Always current: https://www.anchorterminal.com/api/v1/live/openai-speech-to-text.json\n\n## Probe metrics\n\nNot measured yet. Our benchmark probes haven't run, so there's no availability, latency or error rate from a run and Performance is pending. Live uptime, where we poll the endpoint, is under Live and doesn't change the score.\n\n## Prices\n\n| Item | Price | Unit | Note |\n| --- | --- | --- | --- |\n| gpt-transcribe | $0.0045 | per minute of audio |  |\n| gpt-live-transcribe (live audio) | $0.017 | per minute of audio |  |\n| whisper-1 (deprecated) | $0.006 | per minute of audio |  |\n| gpt-4o-transcribe-diarize (deprecated, estimated from token prices) | $0.006 | per minute of audio |  |\n\nAcross all listings: https://www.anchorterminal.com/prices/index.md\n\n## Strengths\n\n- `gpt-transcribe` is priced at $0.0045 an audio minute on a public page, with per-tier request limits of 5,000, 10,000 and 30,000 a minute\n- The data controls page lists `/v1/audio/transcriptions` and `/v1/audio/translations` with no training, no abuse-monitoring retention and no stored application state\n- A public OpenAPI 3.1 document, llms.txt and a Markdown twin of every docs page cover the audio endpoints\n- The status page has an Audio component, shown at 100% uptime for July to October 2026\n- Guide examples cover JavaScript, Python, Go, Java, C#, Ruby, a CLI and curl, and `file` and `model` are the only required fields\n\n## Weaknesses\n\n- `whisper-1`, `gpt-4o-transcribe`, `gpt-4o-mini-transcribe` and `gpt-4o-transcribe-diarize` were deprecated on 26 August 2026 and shut down on 26 February 2027\n- The two named replacements return no speaker labels, word timestamps, `srt` or `vtt` output or English translation in the reviewed documentation\n- Uploads stop at 25 MB, the caller splits longer recordings, and the `gpt-transcribe` model page marks the Batch API as not supported\n- The Markdown twin of the endpoint reference lists the response fields and omits the request parameters, and its first example names a deprecated model\n- openai.com answered our reader with a bot check, so the service terms, privacy policy, sub-processor list and any SLA were not read\n\n## Before you call it (notes for agents)\n\n1. Send `gpt-transcribe` to `POST /v1/audio/transcriptions` for recorded files. Use `languages` (a list), not `language`, and never send both.\n2. Keep each upload at 25 MB or less. Split longer audio between sentences and pass the previous chunk's text in `prompt`.\n3. For speaker labels send `gpt-4o-transcribe-diarize` with `response_format=diarized_json` and `chunking_strategy=auto` for audio over 30 seconds. Plan for its shutdown on 26 February 2027.\n4. Word timestamps, `srt`, `vtt` and `/v1/audio/translations` need `whisper-1`, which cannot stream and shuts down on the same date.\n5. On 429 or 503 wait at least `Retry-After` when present, then back off with jitter. Do not retry `credit_balance_exhausted` or spend-limit errors.\n\n## Connect\n\nInstall:\n\n```bash\npip install openai\n```\n\nFirst request:\n\n```bash\ncurl --request POST \\\n  --url https://api.openai.com/v1/audio/transcriptions \\\n  --header \"Authorization: Bearer $OPENAI_API_KEY\" \\\n  --header 'Content-Type: multipart/form-data' \\\n  --form file=@/path/to/file/audio.mp3 \\\n  --form model=gpt-transcribe\n```\n\n## Similar tools\n\nRanked by shared capabilities, then score. Same-category tools with no shared capability key are listed last.\n\n| Tool | Grade | Score | Rank | Shared capabilities | x402 | Markdown |\n| --- | --- | --- | --- | --- | --- | --- |\n| Amazon Transcribe | BB | 73.4 | 89 | speech.stt, speech.streaming, speech.batch, speech.diarisation, speech.languages | no | https://www.anchorterminal.com/tools/amazon-transcribe.md |\n| Azure AI Speech speech-to-text | BB | 73 | 93 | speech.stt, speech.streaming, speech.batch, speech.diarisation, speech.languages | no | https://www.anchorterminal.com/tools/azure-speech-to-text.md |\n| Deepgram Speech-to-Text (Nova-3, Flux) | BB | 70.3 | 157 | speech.stt, speech.streaming, speech.batch, speech.diarisation, speech.languages | no | https://www.anchorterminal.com/tools/deepgram-stt.md |\n| Google Cloud Speech-to-Text | BB | 70.2 | 159 | speech.stt, speech.streaming, speech.batch, speech.diarisation, speech.languages | no | https://www.anchorterminal.com/tools/google-speech-to-text.md |\n| Gladia Speech-to-Text API + MCP | B | 69.5 | 180 | speech.stt, speech.streaming, speech.batch, speech.diarisation, speech.languages | no | https://www.anchorterminal.com/tools/gladia-stt.md |\n| ElevenLabs Scribe Speech to Text API | B | 68.9 | 200 | speech.stt, speech.streaming, speech.batch, speech.diarisation, speech.languages | no | https://www.anchorterminal.com/tools/elevenlabs-scribe.md |\n\n## Panel reviews (0)\n\nReviewed by the Anchor panel (https://www.anchorterminal.com/reviewers/index.md): .\n\nDesk reviews, written from public documentation, pricing, terms, source and status history on 1 October 2026. No calls made. For a desk review, the outcome says whether the reviewer's questions could be answered from public material: success, partial or failure. How reviews work: https://www.anchorterminal.com/reviews/how-it-works.md\n\n## Notable\n\n- `gpt-transcribe` is the recommended model for recorded files on `POST /v1/audio/transcriptions`, and `gpt-live-transcribe` for live audio in a Realtime transcription session. Both were released on 28 July 2026 (source: \u003chttps://developers.openai.com/api/docs/changelog\u003e)\n- On 26 August 2026 OpenAI deprecated `whisper-1`, `gpt-4o-transcribe`, `gpt-4o-mini-transcribe` and `gpt-4o-transcribe-diarize`, with removal on 26 February 2027 and `gpt-live-transcribe` or `gpt-transcribe` as replacements (source: \u003chttps://developers.openai.com/api/docs/deprecations\u003e)\n- The transcription overview sends callers to `gpt-4o-transcribe-diarize` for speaker labels and to `whisper-1` for word timestamps, `srt` and `vtt` subtitles and translation into English. Both are on the deprecation list (source: \u003chttps://developers.openai.com/api/docs/guides/transcription\u003e)\n- Files can be up to 25 MB. The guide lists mp3, mp4, mpeg, mpga, m4a, wav and webm, and the OpenAPI document adds flac and ogg\n- `gpt-transcribe` accepts a free-form `prompt`, `keywords` and a `languages` list, and returns detected languages. `stream=true` returns `transcript.text.delta` events for a completed file without a Realtime session\n- `gpt-live-transcribe` has no server-side turn detection, so the client commits each audio turn, and it returns no word timestamps, speaker labels or confidence scores (source: \u003chttps://developers.openai.com/api/docs/guides/realtime-transcription\u003e)\n- The data controls page lists both audio endpoints as not used for training, with no abuse-monitoring retention, no application state and Zero Data Retention eligibility (source: \u003chttps://developers.openai.com/api/docs/guides/your-data\u003e)\n- The same API, key and status page as the OpenAI API listing (`openai-api`), which covers the language models. Every fact here was read afresh on 9 October 2026\n\n- #3 of 14 in Best speech-to-text APIs for AI agents: https://www.anchorterminal.com/best/speech-to-text/index.md\n- All 91 stt comparisons: https://www.anchorterminal.com/compare/speech-to-text/index.md\n\n## Compare\n\n- [Amazon Transcribe vs OpenAI Speech to Text](https://www.anchorterminal.com/compare/amazon-transcribe-vs-openai-speech-to-text.md): BB 73.4 vs BB 72.4\n- [AssemblyAI Speech-to-Text (Universal) vs OpenAI Speech to Text](https://www.anchorterminal.com/compare/assemblyai-stt-vs-openai-speech-to-text.md): B 66.8 vs BB 72.4\n- [Azure AI Speech speech-to-text vs OpenAI Speech to Text](https://www.anchorterminal.com/compare/azure-speech-to-text-vs-openai-speech-to-text.md): BB 73 vs BB 72.4\n- [Cartesia Ink vs OpenAI Speech to Text](https://www.anchorterminal.com/compare/cartesia-ink-stt-vs-openai-speech-to-text.md): B 67.9 vs BB 72.4\n- [Deepgram Speech-to-Text (Nova-3, Flux) vs OpenAI Speech to Text](https://www.anchorterminal.com/compare/deepgram-stt-vs-openai-speech-to-text.md): BB 70.3 vs BB 72.4\n- [ElevenLabs Scribe Speech to Text API vs OpenAI Speech to Text](https://www.anchorterminal.com/compare/elevenlabs-scribe-vs-openai-speech-to-text.md): B 68.9 vs BB 72.4\n- [Gladia Speech-to-Text API + MCP vs OpenAI Speech to Text](https://www.anchorterminal.com/compare/gladia-stt-vs-openai-speech-to-text.md): B 69.5 vs BB 72.4\n- [Google Cloud Speech-to-Text vs OpenAI Speech to Text](https://www.anchorterminal.com/compare/google-speech-to-text-vs-openai-speech-to-text.md): BB 70.2 vs BB 72.4\n- [Groq Speech-to-Text vs OpenAI Speech to Text](https://www.anchorterminal.com/compare/groq-speech-to-text-vs-openai-speech-to-text.md): BB 71.8 vs BB 72.4\n- [Mistral Voxtral Transcribe vs OpenAI Speech to Text](https://www.anchorterminal.com/compare/mistral-voxtral-transcribe-vs-openai-speech-to-text.md): B 64.1 vs BB 72.4\n- [OpenAI Speech to Text vs Rev AI Speech-to-Text API](https://www.anchorterminal.com/compare/openai-speech-to-text-vs-rev-ai-stt.md): BB 72.4 vs C 57.8\n- [OpenAI Speech to Text vs Soniox Speech-to-Text](https://www.anchorterminal.com/compare/openai-speech-to-text-vs-soniox-stt.md): BB 72.4 vs C 58.2\n- [OpenAI Speech to Text vs Speechmatics Speech-to-Text](https://www.anchorterminal.com/compare/openai-speech-to-text-vs-speechmatics-stt.md): BB 72.4 vs B 67.1\n\n## Verify this listing\n\nFor the vendor. The badge or a plain link to this page verifies the listing, from a page on openai.com or one of its subdomains, or the README of github.com/openai/openai-python. It shows the listing is the vendor's and that the vendor knows it's here, and it never changes a grade, rank or review. The vendor sends the page's address to `POST https://www.anchorterminal.com/api/v1/verify` as `{\"slug\": \"openai-speech-to-text\", \"url\": \"…\"}`, or calls the `verify_listing` tool at https://www.anchorterminal.com/mcp. We fetch the page once, then again every week; two failed checks in a row and the verification lapses, and a later pass restores it. What we check: https://www.anchorterminal.com/builders/index.md#verify\n\nHTML badge:\n\n```html\n\u003ca href=\"https://www.anchorterminal.com/tools/openai-speech-to-text\"\u003e\u003cimg src=\"https://www.anchorterminal.com/badges/openai-speech-to-text.svg\" alt=\"OpenAI Speech to Text on Anchor Terminal\" height=\"20\"\u003e\u003c/a\u003e\n```\n\nMarkdown badge, for a README:\n\n```markdown\n[![OpenAI Speech to Text on Anchor Terminal](https://www.anchorterminal.com/badges/openai-speech-to-text.svg)](https://www.anchorterminal.com/tools/openai-speech-to-text)\n```\n\nPlain link:\n\n```html\n\u003ca href=\"https://www.anchorterminal.com/tools/openai-speech-to-text\"\u003eOpenAI Speech to Text on Anchor Terminal\u003c/a\u003e\n```\n\n## Share this listing\n\nFor the vendor. Sharing assets for social media, two PNGs of 1200 × 630 that say OpenAI Speech to Text is listed on Anchor Terminal, with the vendor's logo and this page's address and no grade or score.\n\n- Dark: https://www.anchorterminal.com/assets/share/openai-speech-to-text-dark.png\n- Light: https://www.anchorterminal.com/assets/share/openai-speech-to-text-light.png\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-10",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Terminal",
        "url": "https://www.anchorterminal.com/tools/"
      },
      {
        "name": "Speech-to-text",
        "url": "https://www.anchorterminal.com/categories/speech-to-text"
      },
      {
        "name": "OpenAI Speech to Text",
        "url": ""
      }
    ],
    "description": "OpenAI's speech-to-text API. It transcribes uploaded audio files through /v1/audio/transcriptions, translates recordings into English through /v1/audio/translations, and transcribes live audio in Realtime transcription sessions over WebSocket or WebRTC.",
    "facts": [
      "rank #106 of 950",
      "API key auth",
      "0 desk reviews"
    ],
    "h1": "OpenAI Speech to Text",
    "image": "https://www.anchorterminal.com/assets/og/tools-openai-speech-to-text.png",
    "path": "/tools/openai-speech-to-text",
    "published": "2026-10-01",
    "section": "tools",
    "title": "OpenAI Speech to Text review: pricing, alternatives, grade BB",
    "toc": null,
    "updated": "2026-10-10",
    "url": "https://www.anchorterminal.com/tools/openai-speech-to-text"
  },
  "tokens": {
    "markdown": 7800,
    "slim": 1730
  },
  "version": 1
}
