{
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-10",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "tool": {
    "slug": "openai-speech-to-text",
    "name": "OpenAI Speech to Text",
    "vendor": "OpenAI",
    "vendorUrl": "https://openai.com",
    "kind": "model",
    "category": "speech-to-text",
    "summary": "OpenAI's speech-to-text API. It transcribes uploaded audio files through `/v1/audio/transcriptions`, translates recordings into English through `/v1/audio/translations`, and transcribes live audio in Realtime transcription sessions over WebSocket or WebRTC.",
    "url": "https://www.anchorterminal.com/tools/openai-speech-to-text",
    "markdownUrl": "https://www.anchorterminal.com/tools/openai-speech-to-text.md",
    "slimMarkdownUrl": "https://www.anchorterminal.com/tools/openai-speech-to-text.min.md",
    "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/openai-speech-to-text.json",
    "repo": "https://github.com/openai/openai-python",
    "license": "Proprietary hosted service. The service terms were not read (see open questions). The Python SDK is Apache-2.0 and the OpenAPI document is MIT",
    "transports": [
      "http",
      "websocket"
    ],
    "remoteUrl": "https://api.openai.com/v1",
    "packages": [
      {
        "registry": "pypi",
        "name": "openai"
      }
    ],
    "auth": "api-key",
    "authNotes": "Bearer API key created by a person in the platform console (https://platform.openai.com/settings/organization/api-keys). Projects can carry a model allowlist or denylist and an IP allowlist, and Admin API keys are a separate credential that cannot call the audio endpoints (https://developers.openai.com/api/docs/guides/admin-apis).",
    "pricing": "usage",
    "pricingNotes": "$0.0045 an audio minute for `gpt-transcribe` and $0.017 for `gpt-live-transcribe`, billed from prepaid credits (https://developers.openai.com/api/docs/pricing). The rate limits guide names a Free tier with a $100 monthly usage limit, and the `gpt-transcribe` model page lists limits only from the Build tier, which needs $5 of credit purchases. Whether a new account can transcribe without paying was not established.",
    "priceSummary": "Pay per use",
    "where": "hosted",
    "x402": {
      "level": "no",
      "evidence": "No x402, MPP or L402 in the transcription guides, the endpoint reference, the pricing page or the OpenAPI document (checked 2026-10-09).",
      "endpoints": []
    },
    "toolCount": null,
    "popularity": {
      "githubStars": 31785,
      "npmWeekly": null,
      "pypiWeekly": null,
      "asOf": "2026-10-09"
    },
    "docsUrl": "https://developers.openai.com/api/docs/guides/speech-to-text",
    "llmsTxt": "https://developers.openai.com/llms.txt",
    "openapi": "https://github.com/openai/openai-openapi/blob/main/openapi.yaml",
    "capabilities": [
      "speech.stt",
      "speech.streaming",
      "speech.batch",
      "speech.diarisation",
      "speech.languages"
    ],
    "tags": [
      "official",
      "hosted",
      "model",
      "streaming",
      "diarisation",
      "llms-txt",
      "openapi",
      "python",
      "typescript",
      "go",
      "java"
    ],
    "lastRelease": "2026-08-26",
    "graded": true,
    "anchor": {
      "graded": true,
      "score": 72.4,
      "grade": "BB",
      "agentReady": true,
      "rank": 106,
      "ranked": true,
      "rankOf": 950,
      "categoryRank": 3,
      "methodology": "0.4",
      "run": "2026-10-01",
      "scores": {
        "ergonomics": 78,
        "maintenance": 75,
        "payments": 20,
        "reliability": 80,
        "schema": 88,
        "security": 86,
        "transparency": 61
      },
      "pending": [
        "performance",
        "tasks"
      ],
      "breakdown": [
        {
          "key": "reliability",
          "name": "Reliability",
          "weight": 16,
          "effectiveWeight": 20,
          "score": 80,
          "points": 16,
          "reason": "Hosted reading. status.openai.com (incident.io) lists Audio and Realtime as components of the API group (20). On 9 October 2026 it showed Audio at 100% uptime for July to October 2026. The linked feed holds five resolved incidents since 11 July that name the Audio component, all platform-wide elevated error rates (24 July, two on 25 July, 17 September and 6 October). Their durations were not read, and we score them as minor (20 of 30). The `gpt-transcribe` model page gives 5,000 requests a minute on Build, 10,000 on Launch and 30,000 on Grow (15). The rate limits guide documents `Retry-After` on 429 and 503, the `slow_down` and `server_is_overloaded` codes and backoff with jitter, and transcription is a stateless call (15). No SLA was found in the pages read. The Scale Tier page the guide links is on openai.com, which answered a bot check, so it is unread and scored absent (0). `gpt-transcribe` carries no preview label (10)."
        },
        {
          "key": "performance",
          "name": "Performance",
          "weight": 10,
          "effectiveWeight": 0,
          "pending": true,
          "points": 0,
          "reason": "Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes."
        },
        {
          "key": "schema",
          "name": "Schema \u0026 documentation",
          "weight": 13,
          "effectiveWeight": 16.25,
          "score": 88,
          "points": 14.3,
          "reason": "Model reading. A public OpenAPI 3.1 document in openai/openai-openapi covers `/audio/transcriptions` and `/audio/translations` and names `gpt-transcribe` (25). llms.txt and a Markdown twin of every docs page (10). The transcription overview says which model to use for files, live audio, speaker labels, timestamps and translation, and when not to use the specialised ones (18 of 20). `response_format`, `timestamp_granularities` and `model` are enums in the document. `language` is a plain string, and which formats each model accepts is stated in prose only (12 of 15). Examples in seven languages and curl, and an error codes page with a cause and fix for each. The Markdown twin of the endpoint reference omits the request parameters, its first example names the deprecated `gpt-4o-transcribe`, and the guide lists seven input formats where the document lists nine (11 of 15). A dated changelog and deprecations page. `gpt-transcribe` has one alias and no dated snapshot to pin (12 of 15)."
        },
        {
          "key": "ergonomics",
          "name": "Agent ergonomics",
          "weight": 13,
          "effectiveWeight": 16.25,
          "score": 78,
          "points": 12.68,
          "reason": "API reading, as for the other speech-to-text listings. The default `json` response is the text, detected languages and usage, `text` returns a bare string, and segments, words and speaker labels come only when asked for on the models that have them (20 of 25). There is no transcript store to page through. Files stop at 25 MB, the caller splits longer audio, and the `gpt-transcribe` model page marks the Batch API as not supported (10 of 20). Errors carry `message`, `type`, `param` and `code`, and the error codes page gives a fix for each status (18 of 20). The call is stateless and safe to retry, and the Python SDK retries 408, 409, 429 and 5xx twice by default. No idempotency key applies to this endpoint (15 of 20). `file` and `model` are the only required fields, with guide examples for JavaScript, Python, Go, Java, C# and Ruby SDKs and a CLI (15)."
        },
        {
          "key": "security",
          "name": "Security \u0026 auth",
          "weight": 14,
          "effectiveWeight": 17.5,
          "score": 86,
          "points": 15.05,
          "reason": "Model reading. Bearer keys are created per project. Admin API keys are a separate credential that cannot call non-administration endpoints, a project can be limited to a list of models, and the error codes page documents an IP allowlist and keys without permission for an endpoint. The pages on service accounts and workload identity federation were not read (26 of 30). The data controls page says API data has not been used for training since 1 March 2023 unless the customer opts in (20). Both audio endpoints are listed with no abuse-monitoring retention, no application state and Zero Data Retention eligibility (15). The Admin API has an audit log endpoint for user actions and configuration changes (13 of 15). security.txt is PGP-signed, names a Bugcrowd programme and a coordinated disclosure policy and has no Expires field. Certifications were not read, because no page we read links a trust centre and openai.com answered a bot check (12 of 20)."
        },
        {
          "key": "payments",
          "name": "Payments \u0026 pricing",
          "weight": 10,
          "effectiveWeight": 12.5,
          "score": 20,
          "points": 2.5,
          "reason": "No x402, MPP or L402 (0). $0.0045 an audio minute for `gpt-transcribe`, $0.017 for `gpt-live-transcribe` and $0.006 for `whisper-1` on the public pricing page (20). The rate limits guide names a Free tier with a $100 monthly usage limit, and the `gpt-transcribe` model page lists limits only from Build, which needs $5 of credit purchases. No free transcription allowance or no-card trial was found (0). A person signs up in a browser and creates the key in the console (0)."
        },
        {
          "key": "tasks",
          "name": "Task success",
          "weight": 10,
          "effectiveWeight": 0,
          "pending": true,
          "points": 0,
          "reason": "Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored."
        },
        {
          "key": "maintenance",
          "name": "Maintenance \u0026 community",
          "weight": 7,
          "effectiveWeight": 8.75,
          "score": 75,
          "points": 6.56,
          "reason": "Model reading. The last dated change to transcription is the deprecation notice of 26 August 2026, 44 days before the check, and the release of `gpt-transcribe` and `gpt-live-transcribe` was on 28 July (20 of 30). The deprecations page promises at least six months' notice for generally available models, and the four transcription models got six months to the day (12 of 12). Four of the five file transcription models were deprecated within 90 days, a month after their replacements shipped, and the replacements do not cover diarisation, timestamps, subtitles or translation (2 of 8). The dated changelog has two transcription entries in 90 days among dozens for the API (13 of 15). The Python SDK repository showed 140 open issues and pull requests. Replies were not sampled (5 of 10). The Python SDK 3.26.1 was tagged on 8 October 2026, with 15 tags since 18 September (15). CI, CodeQL and a breaking-change check run in the repository (8 of 10)."
        },
        {
          "key": "transparency",
          "name": "Transparency \u0026 trust",
          "weight": 7,
          "effectiveWeight": 8.75,
          "score": 61,
          "points": 5.34,
          "note": "editorial 62, provenance 59",
          "reason": "The hosted models are closed. The Python SDK is Apache-2.0, the OpenAPI document is MIT, and the SDK says `whisper-1` runs the open-source Whisper V2 model. The service terms were not read, because openai.com answered a bot check (12 of 30). The data controls page states training, retention and application state for each endpoint. The privacy policy and any DPA were not read, so whether they agree with it was not checked (18 of 30). A deprecations page with notice periods by model class, dated shutdowns and replacements (20). Data residency is documented by region and endpoint, with the audio endpoints in all ten listed regions for storage and in the United States and Europe for processing. The sub-processor list the docs link answered a bot check (12 of 20)."
        }
      ],
      "assessment": {
        "date": "2026-10-09",
        "basis": "public evidence",
        "confidence": "medium",
        "notes": {
          "ergonomics": "API reading, as for the other speech-to-text listings. The default `json` response is the text, detected languages and usage, `text` returns a bare string, and segments, words and speaker labels come only when asked for on the models that have them (20 of 25). There is no transcript store to page through. Files stop at 25 MB, the caller splits longer audio, and the `gpt-transcribe` model page marks the Batch API as not supported (10 of 20). Errors carry `message`, `type`, `param` and `code`, and the error codes page gives a fix for each status (18 of 20). The call is stateless and safe to retry, and the Python SDK retries 408, 409, 429 and 5xx twice by default. No idempotency key applies to this endpoint (15 of 20). `file` and `model` are the only required fields, with guide examples for JavaScript, Python, Go, Java, C# and Ruby SDKs and a CLI (15).",
          "maintenance": "Model reading. The last dated change to transcription is the deprecation notice of 26 August 2026, 44 days before the check, and the release of `gpt-transcribe` and `gpt-live-transcribe` was on 28 July (20 of 30). The deprecations page promises at least six months' notice for generally available models, and the four transcription models got six months to the day (12 of 12). Four of the five file transcription models were deprecated within 90 days, a month after their replacements shipped, and the replacements do not cover diarisation, timestamps, subtitles or translation (2 of 8). The dated changelog has two transcription entries in 90 days among dozens for the API (13 of 15). The Python SDK repository showed 140 open issues and pull requests. Replies were not sampled (5 of 10). The Python SDK 3.26.1 was tagged on 8 October 2026, with 15 tags since 18 September (15). CI, CodeQL and a breaking-change check run in the repository (8 of 10).",
          "payments": "No x402, MPP or L402 (0). $0.0045 an audio minute for `gpt-transcribe`, $0.017 for `gpt-live-transcribe` and $0.006 for `whisper-1` on the public pricing page (20). The rate limits guide names a Free tier with a $100 monthly usage limit, and the `gpt-transcribe` model page lists limits only from Build, which needs $5 of credit purchases. No free transcription allowance or no-card trial was found (0). A person signs up in a browser and creates the key in the console (0).",
          "reliability": "Hosted reading. status.openai.com (incident.io) lists Audio and Realtime as components of the API group (20). On 9 October 2026 it showed Audio at 100% uptime for July to October 2026. The linked feed holds five resolved incidents since 11 July that name the Audio component, all platform-wide elevated error rates (24 July, two on 25 July, 17 September and 6 October). Their durations were not read, and we score them as minor (20 of 30). The `gpt-transcribe` model page gives 5,000 requests a minute on Build, 10,000 on Launch and 30,000 on Grow (15). The rate limits guide documents `Retry-After` on 429 and 503, the `slow_down` and `server_is_overloaded` codes and backoff with jitter, and transcription is a stateless call (15). No SLA was found in the pages read. The Scale Tier page the guide links is on openai.com, which answered a bot check, so it is unread and scored absent (0). `gpt-transcribe` carries no preview label (10).",
          "schema": "Model reading. A public OpenAPI 3.1 document in openai/openai-openapi covers `/audio/transcriptions` and `/audio/translations` and names `gpt-transcribe` (25). llms.txt and a Markdown twin of every docs page (10). The transcription overview says which model to use for files, live audio, speaker labels, timestamps and translation, and when not to use the specialised ones (18 of 20). `response_format`, `timestamp_granularities` and `model` are enums in the document. `language` is a plain string, and which formats each model accepts is stated in prose only (12 of 15). Examples in seven languages and curl, and an error codes page with a cause and fix for each. The Markdown twin of the endpoint reference omits the request parameters, its first example names the deprecated `gpt-4o-transcribe`, and the guide lists seven input formats where the document lists nine (11 of 15). A dated changelog and deprecations page. `gpt-transcribe` has one alias and no dated snapshot to pin (12 of 15).",
          "security": "Model reading. Bearer keys are created per project. Admin API keys are a separate credential that cannot call non-administration endpoints, a project can be limited to a list of models, and the error codes page documents an IP allowlist and keys without permission for an endpoint. The pages on service accounts and workload identity federation were not read (26 of 30). The data controls page says API data has not been used for training since 1 March 2023 unless the customer opts in (20). Both audio endpoints are listed with no abuse-monitoring retention, no application state and Zero Data Retention eligibility (15). The Admin API has an audit log endpoint for user actions and configuration changes (13 of 15). security.txt is PGP-signed, names a Bugcrowd programme and a coordinated disclosure policy and has no Expires field. Certifications were not read, because no page we read links a trust centre and openai.com answered a bot check (12 of 20).",
          "transparency": "The hosted models are closed. The Python SDK is Apache-2.0, the OpenAPI document is MIT, and the SDK says `whisper-1` runs the open-source Whisper V2 model. The service terms were not read, because openai.com answered a bot check (12 of 30). The data controls page states training, retention and application state for each endpoint. The privacy policy and any DPA were not read, so whether they agree with it was not checked (18 of 30). A deprecations page with notice periods by model class, dated shutdowns and replacements (20). Data residency is documented by region and endpoint, with the audio endpoints in all ten listed regions for storage and in the United States and Europe for processing. The sub-processor list the docs link answered a bot check (12 of 20)."
        },
        "sources": [
          {
            "what": "file transcription guide (Markdown twin)",
            "url": "https://developers.openai.com/api/docs/guides/speech-to-text",
            "seen": "2026-10-09"
          },
          {
            "what": "transcription overview",
            "url": "https://developers.openai.com/api/docs/guides/transcription",
            "seen": "2026-10-09"
          },
          {
            "what": "realtime transcription guide",
            "url": "https://developers.openai.com/api/docs/guides/realtime-transcription",
            "seen": "2026-10-09"
          },
          {
            "what": "gpt-transcribe model page, price, endpoints and rate limits",
            "url": "https://developers.openai.com/api/docs/models/gpt-transcribe",
            "seen": "2026-10-09"
          },
          {
            "what": "create transcription reference (Markdown twin)",
            "url": "https://developers.openai.com/api/reference/resources/audio/subresources/transcriptions/methods/create",
            "seen": "2026-10-09"
          },
          {
            "what": "pricing, transcription models table",
            "url": "https://developers.openai.com/api/docs/pricing",
            "seen": "2026-10-09"
          },
          {
            "what": "deprecations, notice periods and the 26 August 2026 transcription entry",
            "url": "https://developers.openai.com/api/docs/deprecations",
            "seen": "2026-10-09"
          },
          {
            "what": "API changelog",
            "url": "https://developers.openai.com/api/docs/changelog",
            "seen": "2026-10-09"
          },
          {
            "what": "rate limits guide, usage tiers and retry guidance",
            "url": "https://developers.openai.com/api/docs/guides/rate-limits",
            "seen": "2026-10-09"
          },
          {
            "what": "error codes",
            "url": "https://developers.openai.com/api/docs/guides/error-codes",
            "seen": "2026-10-09"
          },
          {
            "what": "data controls, retention per endpoint and data residency",
            "url": "https://developers.openai.com/api/docs/guides/your-data",
            "seen": "2026-10-09"
          },
          {
            "what": "Admin APIs guide",
            "url": "https://developers.openai.com/api/docs/guides/admin-apis",
            "seen": "2026-10-09"
          },
          {
            "what": "llms.txt and the API docs indexes",
            "url": "https://developers.openai.com/llms.txt",
            "seen": "2026-10-09"
          },
          {
            "what": "OpenAPI document, read from a shallow clone",
            "url": "https://github.com/openai/openai-openapi",
            "seen": "2026-10-09"
          },
          {
            "what": "Python SDK source, changelog, tags and security policy, read from a shallow clone",
            "url": "https://github.com/openai/openai-python",
            "seen": "2026-10-09"
          },
          {
            "what": "status page, component uptime",
            "url": "https://status.openai.com",
            "seen": "2026-10-09"
          },
          {
            "what": "status incident feed linked from the status page",
            "url": "https://status.openai.com/feed.atom",
            "seen": "2026-10-09"
          },
          {
            "what": "security.txt",
            "url": "https://openai.com/.well-known/security.txt",
            "seen": "2026-10-09"
          }
        ],
        "openQuestions": [
          "unchecked: the service terms, privacy policy, DPA and the contracting legal entity. openai.com answered the sub-processor page with a bot check (HTTP 403) on 9 October 2026 and no other openai.com page was requested after that. `provenance.terms` and `provenance.privacy` are left out",
          "unchecked: the sub-processor list at openai.com/policies/sub-processor-list, which answered 403",
          "unchecked: any SLA. The Scale Tier and Reserved Tier pages the rate limits guide links are on openai.com and were not requested",
          "unchecked: certifications and a trust centre. No page read links one",
          "unchecked: whether the Free tier can call `gpt-transcribe` and whether it needs a card. The model page lists limits from Build only",
          "unchecked: the `gpt-live-transcribe`, `whisper-1` and `gpt-4o-transcribe-diarize` model pages, the service account and workload identity pages, and the Realtime sessions reference",
          "unchecked: durations of the five status incidents that name Audio. Only the status page and its linked feed were read",
          "unchecked: the openai.com registration date and npm and PyPI download counts",
          "unchecked: replies on the SDK issue trackers",
          "The deprecation of 26 August 2026 names `gpt-live-transcribe` or `gpt-transcribe` as replacements for `whisper-1` and `gpt-4o-transcribe-diarize`, and neither has diarisation, word timestamps, subtitle output or translation in the docs read. No deduction was taken because nothing has been removed yet and the notice is six months",
          "security.txt has no Expires field. It is recorded as valid because it is signed and names two contacts",
          "The lead named `gpt-4o-transcribe` and `whisper-1` among the product's models. Both are deprecated, and the listing name drops the model list",
          "This product shares its API, key and status page with `openai-api`. If the owner holds listings whose governing terms were not read, this one qualifies",
          "The rate limits guide's retry examples and the reference examples still name models past or near their shutdown dates",
          "Robots.txt answers: developers.openai.com and openai.com 200 and allow the paths read, status.openai.com and api.github.com 404, read as no rules. developers.openai.com received 17 requests against the guide of about fifteen"
        ]
      },
      "negative": 0,
      "verdict": "`gpt-transcribe` costs $0.0045 an audio minute, and the audio endpoints keep no abuse-monitoring logs or application state. Speaker labels, timestamps, subtitles and translation exist only on `whisper-1` and `gpt-4o-transcribe-diarize`, which shut down on 26 February 2027 with no named replacement for those functions.",
      "bestFor": "Suited to plain transcription of recorded files at a low price a minute and to teams already holding an OpenAI key.",
      "strengths": [
        "`gpt-transcribe` is priced at $0.0045 an audio minute on a public page, with per-tier request limits of 5,000, 10,000 and 30,000 a minute",
        "The data controls page lists `/v1/audio/transcriptions` and `/v1/audio/translations` with no training, no abuse-monitoring retention and no stored application state",
        "A public OpenAPI 3.1 document, llms.txt and a Markdown twin of every docs page cover the audio endpoints",
        "The status page has an Audio component, shown at 100% uptime for July to October 2026",
        "Guide examples cover JavaScript, Python, Go, Java, C#, Ruby, a CLI and curl, and `file` and `model` are the only required fields"
      ],
      "weaknesses": [
        "`whisper-1`, `gpt-4o-transcribe`, `gpt-4o-mini-transcribe` and `gpt-4o-transcribe-diarize` were deprecated on 26 August 2026 and shut down on 26 February 2027",
        "The two named replacements return no speaker labels, word timestamps, `srt` or `vtt` output or English translation in the reviewed documentation",
        "Uploads stop at 25 MB, the caller splits longer recordings, and the `gpt-transcribe` model page marks the Batch API as not supported",
        "The Markdown twin of the endpoint reference lists the response fields and omits the request parameters, and its first example names a deprecated model",
        "openai.com answered our reader with a bot check, so the service terms, privacy policy, sub-processor list and any SLA were not read"
      ],
      "agentNotes": [
        "Send `gpt-transcribe` to `POST /v1/audio/transcriptions` for recorded files. Use `languages` (a list), not `language`, and never send both.",
        "Keep each upload at 25 MB or less. Split longer audio between sentences and pass the previous chunk's text in `prompt`.",
        "For speaker labels send `gpt-4o-transcribe-diarize` with `response_format=diarized_json` and `chunking_strategy=auto` for audio over 30 seconds. Plan for its shutdown on 26 February 2027.",
        "Word timestamps, `srt`, `vtt` and `/v1/audio/translations` need `whisper-1`, which cannot stream and shuts down on the same date.",
        "On 429 or 503 wait at least `Retry-After` when present, then back off with jitter. Do not retry `credit_balance_exhausted` or spend-limit errors."
      ],
      "metrics": {
        "kind": "remote",
        "measured": false
      },
      "reviewCount": 0,
      "avgRating": 0,
      "history": [
        {
          "basis": "public evidence",
          "confidence": "medium",
          "grade": "BB",
          "methodology": "0.4",
          "pending": [
            "performance",
            "tasks"
          ],
          "run": "2026-10-01",
          "runLabel": "October 2026 research run",
          "score": 72.4
        }
      ],
      "editorialScores": {
        "ergonomics": 78,
        "maintenance": 75,
        "payments": 20,
        "reliability": 80,
        "schema": 88,
        "security": 86,
        "transparency": 62
      },
      "provenanceScore": 59
    },
    "connect": {
      "install": "pip install openai",
      "http": "curl --request POST \\\n  --url https://api.openai.com/v1/audio/transcriptions \\\n  --header \"Authorization: Bearer $OPENAI_API_KEY\" \\\n  --header 'Content-Type: multipart/form-data' \\\n  --form file=@/path/to/file/audio.mp3 \\\n  --form model=gpt-transcribe"
    },
    "letme": {
      "capability": "https://letme.dev/speech.stt",
      "tool": "https://letme.dev/openai-speech-to-text"
    },
    "sameCompany": [
      "openai-api",
      "openai-embeddings",
      "openai-guardrails",
      "openai-moderation",
      "openai-image-api",
      "openai-sora",
      "openai-realtime",
      "openai-agents-sdk",
      "openai-decisions-api",
      "openai-codex"
    ],
    "notable": [
      "`gpt-transcribe` is the recommended model for recorded files on `POST /v1/audio/transcriptions`, and `gpt-live-transcribe` for live audio in a Realtime transcription session. Both were released on 28 July 2026 (https://developers.openai.com/api/docs/changelog)",
      "On 26 August 2026 OpenAI deprecated `whisper-1`, `gpt-4o-transcribe`, `gpt-4o-mini-transcribe` and `gpt-4o-transcribe-diarize`, with removal on 26 February 2027 and `gpt-live-transcribe` or `gpt-transcribe` as replacements (https://developers.openai.com/api/docs/deprecations)",
      "The transcription overview sends callers to `gpt-4o-transcribe-diarize` for speaker labels and to `whisper-1` for word timestamps, `srt` and `vtt` subtitles and translation into English. Both are on the deprecation list (https://developers.openai.com/api/docs/guides/transcription)",
      "Files can be up to 25 MB. The guide lists mp3, mp4, mpeg, mpga, m4a, wav and webm, and the OpenAPI document adds flac and ogg",
      "`gpt-transcribe` accepts a free-form `prompt`, `keywords` and a `languages` list, and returns detected languages. `stream=true` returns `transcript.text.delta` events for a completed file without a Realtime session",
      "`gpt-live-transcribe` has no server-side turn detection, so the client commits each audio turn, and it returns no word timestamps, speaker labels or confidence scores (https://developers.openai.com/api/docs/guides/realtime-transcription)",
      "The data controls page lists both audio endpoints as not used for training, with no abuse-monitoring retention, no application state and Zero Data Retention eligibility (https://developers.openai.com/api/docs/guides/your-data)",
      "The same API, key and status page as the OpenAI API listing (`openai-api`), which covers the language models. Every fact here was read afresh on 9 October 2026"
    ],
    "area": "voice",
    "details": [
      {
        "label": "Models",
        "value": "`gpt-transcribe` (files and committed Realtime turns) and `gpt-live-transcribe` (live audio). `whisper-1`, `gpt-4o-transcribe`, `gpt-4o-mini-transcribe` and `gpt-4o-transcribe-diarize` are deprecated and shut down on 26 February 2027"
      },
      {
        "label": "Languages",
        "value": "`languages` takes ISO 639-1 codes, selected ISO 639-3 codes and regional `zh` codes. No count is given for `gpt-transcribe`. The guide says Whisper supports 98 languages"
      },
      {
        "label": "Max audio",
        "value": "25 MB a file. Longer recordings are split by the caller"
      },
      {
        "label": "Input",
        "value": "Multipart `file` in mp3, mp4, mpeg, mpga, m4a, wav or webm per the guide. The OpenAPI document also lists flac and ogg"
      },
      {
        "label": "Output",
        "value": "`json` with text and detected languages, or `text`. `verbose_json`, `srt` and `vtt` on `whisper-1` only, `diarized_json` on `gpt-4o-transcribe-diarize` only"
      },
      {
        "label": "Streaming",
        "value": "`stream=true` on file transcription (not `whisper-1`), and Realtime transcription sessions over WebSocket or WebRTC for live audio"
      },
      {
        "label": "Diarisation",
        "value": "`gpt-4o-transcribe-diarize` only, with up to four known speaker references. Not available in Realtime sessions. The model is deprecated"
      },
      {
        "label": "Timestamps",
        "value": "Word and segment timestamps on `whisper-1` only"
      },
      {
        "label": "Translation",
        "value": "Into English only, through `/v1/audio/translations` with `whisper-1`"
      },
      {
        "label": "Rate limits",
        "value": "`gpt-transcribe` 5,000 requests a minute on Build, 10,000 on Launch and 30,000 on Grow"
      },
      {
        "label": "Batch",
        "value": "The `gpt-transcribe` model page marks the Batch API as not supported"
      },
      {
        "label": "Trains on API data",
        "value": "No, unless the customer opts in, per the data controls page"
      },
      {
        "label": "Data retention",
        "value": "No abuse-monitoring retention and no application state for `/v1/audio/transcriptions` and `/v1/audio/translations`"
      },
      {
        "label": "Data location",
        "value": "Regional storage in ten regions through project settings and prefixed hosts such as `eu.api.openai.com`, on approval through sales. Regional processing in the United States and Europe"
      }
    ],
    "unitPrices": [
      {
        "item": "gpt-transcribe",
        "unit": "audio-minute",
        "usd": 0.0045
      },
      {
        "item": "gpt-live-transcribe (live audio)",
        "unit": "audio-minute",
        "usd": 0.017
      },
      {
        "item": "whisper-1 (deprecated)",
        "unit": "audio-minute",
        "usd": 0.006
      },
      {
        "item": "gpt-4o-transcribe-diarize (deprecated, estimated from token prices)",
        "unit": "audio-minute",
        "usd": 0.006
      }
    ],
    "provenance": {
      "legalEntity": "",
      "domain": "openai.com",
      "domainRegistered": "",
      "domainNote": "openai.com answered our researcher with a bot check on 9 October 2026, so the terms and privacy policy were not read on that day. The links are the two documents OpenAI's other listings here carry. openai.com answers our policy reader with HTTP 403 as well, so neither document has been read and both are recorded as unreadable. security.txt is PGP-signed with Bugcrowd and email contacts and has no Expires field.",
      "endpointOnVendorDomain": true,
      "terms": "https://openai.com/policies/services-agreement/",
      "privacy": "https://openai.com/policies/privacy-policy/",
      "statusPage": "https://status.openai.com",
      "changelog": "https://developers.openai.com/api/docs/changelog",
      "securityTxt": "valid",
      "checked": "2026-10-09",
      "score": 59,
      "checks": [
        {
          "check": "Legal entity named",
          "value": "not found",
          "points": 0,
          "max": 20,
          "state": "no"
        },
        {
          "check": "Domain age",
          "value": "openai.com, no registry record we could read",
          "points": 0,
          "max": 15,
          "state": "no"
        },
        {
          "check": "Endpoint on the vendor's domain",
          "value": "api.openai.com",
          "points": 15,
          "max": 15,
          "state": "ok"
        },
        {
          "check": "Terms of service",
          "value": "published, but our reader couldn't read it",
          "points": 7,
          "max": 10,
          "state": "part"
        },
        {
          "check": "Privacy policy",
          "value": "published, but our reader couldn't read it",
          "points": 7,
          "max": 10,
          "state": "part"
        },
        {
          "check": "Status page",
          "value": "status.openai.com",
          "points": 10,
          "max": 10,
          "state": "ok"
        },
        {
          "check": "Changelog",
          "value": "published",
          "points": 10,
          "max": 10,
          "state": "ok"
        },
        {
          "check": "security.txt",
          "value": "valid",
          "points": 10,
          "max": 10,
          "state": "ok"
        }
      ],
      "policies": [
        {
          "kind": "terms",
          "url": "https://openai.com/policies/services-agreement/",
          "state": "unreadable",
          "reason": "the page answered HTTP 403 to our reader",
          "readAt": "2026-10-08",
          "points": 7,
          "max": 10
        },
        {
          "kind": "privacy",
          "url": "https://openai.com/policies/privacy-policy/",
          "state": "unreadable",
          "reason": "the page answered HTTP 403 to our reader",
          "readAt": "2026-10-08",
          "points": 7,
          "max": 10
        }
      ]
    },
    "pageJsonUrl": "https://www.anchorterminal.com/tools/openai-speech-to-text.json",
    "live": {
      "slug": "openai-speech-to-text",
      "probe": {
        "target": "https://api.openai.com/v1",
        "method": "get",
        "lastAt": "2026-10-10T02:55:04.59384396Z",
        "lastOk": true,
        "lastStatus": 404,
        "lastMs": 120,
        "authRequired": false,
        "uptime24h": 100,
        "uptime30d": 100,
        "p50ms24h": 138,
        "p95ms24h": 172,
        "samples24h": 115,
        "samples30d": 115,
        "days": [
          {
            "date": "2026-10-09",
            "probes": 85,
            "ok": 85
          },
          {
            "date": "2026-10-10",
            "probes": 30,
            "ok": 30
          }
        ]
      },
      "vendorStatus": {
        "page": "https://status.openai.com",
        "indicator": "minor",
        "summary": "Partial System Degradation",
        "checkedAt": "2026-10-10T02:50:38.239143757Z"
      },
      "versions": [
        {
          "registry": "github",
          "name": "openai/openai-python",
          "version": "v3.27.0",
          "released": "2026-10-09",
          "seenAt": "2026-10-09T17:10:31.877366615Z"
        },
        {
          "registry": "pypi",
          "name": "openai",
          "version": "3.27.0",
          "released": "2026-10-09",
          "seenAt": "2026-10-09T17:10:31.73110612Z"
        }
      ],
      "githubStars": 31787,
      "pypiWeekly": 74761714,
      "updatedAt": "2026-10-10T02:55:04.59384396Z"
    }
  }
}
