{
  "data": {
    "a": {
      "slug": "google-speech-to-text",
      "name": "Google Cloud Speech-to-Text",
      "vendor": "Google Cloud",
      "vendorUrl": "https://cloud.google.com/speech-to-text",
      "kind": "model",
      "category": "speech-to-text",
      "summary": "Google Cloud's transcription API.",
      "url": "https://www.anchorterminal.com/tools/google-speech-to-text",
      "markdownUrl": "https://www.anchorterminal.com/tools/google-speech-to-text.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/google-speech-to-text.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/google-speech-to-text.json",
      "repo": "https://github.com/googleapis/google-cloud-python/tree/main/packages/google-cloud-speech",
      "license": "Apache-2.0 (SDKs)",
      "transports": [
        "http"
      ],
      "remoteUrl": "https://speech.googleapis.com/v2",
      "packages": [
        {
          "registry": "pypi",
          "name": "google-cloud-speech"
        },
        {
          "registry": "npm",
          "name": "@google-cloud/speech"
        }
      ],
      "auth": "oauth",
      "authNotes": "OAuth 2.0 bearer token from a service account or `gcloud` (Application Default Credentials) on a project with billing and the API turned on. Chirp 3 runs on the `us` and `eu` multi-region endpoints such as `us-speech.googleapis.com`.",
      "pricing": "freemium",
      "pricingNotes": "V2 standard recognition, which covers Chirp 3, is $0.016 a minute to 500,000 minutes a month, then $0.01, $0.008 and $0.004 past 2M. Dynamic batch is $0.003 a minute. Billed per second, per channel. V1 has 60 free minutes a month and charges $0.024 without data logging. Medical models $0.078 (https://cloud.google.com/speech-to-text/pricing).",
      "priceSummary": "Freemium",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No machine payment. Billing runs through a cloud account with a card or invoice.",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": null,
        "npmWeekly": 713013,
        "pypiWeekly": 3703632,
        "asOf": "2026-09-30"
      },
      "docsUrl": "https://docs.cloud.google.com/speech-to-text/docs",
      "capabilities": [
        "speech.stt",
        "speech.streaming",
        "speech.batch",
        "speech.diarisation",
        "speech.languages",
        "speech.translation"
      ],
      "tags": [
        "hosted",
        "freemium",
        "closed-source",
        "python",
        "typescript",
        "enterprise",
        "streaming",
        "batch",
        "async-jobs",
        "card-required"
      ],
      "lastRelease": "2026-09-28",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 70.2,
        "grade": "BB",
        "agentReady": true,
        "rank": 152,
        "ranked": true,
        "rankOf": 842,
        "categoryRank": 5,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 70,
          "maintenance": 25,
          "payments": 20,
          "reliability": 85,
          "schema": 80,
          "security": 95,
          "transparency": 86
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-01"
        },
        "negative": 0,
        "verdict": "Audio isn't stored or used for training unless the project opts in to data logging. No release note since 2025-11-13.",
        "bestFor": "Google Cloud teams with audio already in Cloud Storage who want no training and no retention by default.",
        "strengths": [
          "Audio isn't stored or used for training unless the project opts in to data logging",
          "No Speech-to-Text incident on the Google Cloud status page since 12 June 2025",
          "Dynamic batch at $0.003 a minute, and standard recognition tiers down to $0.004 past 2M minutes",
          "OAuth service accounts with IAM roles, and Cloud Audit Logs",
          "300 concurrent streams per region by default"
        ],
        "weaknesses": [
          "No release note since 2025-11-13",
          "82 of the 111 Chirp 3 locales are preview",
          "Sync requests stop at 1 minute and streams at 5 minutes, and batch reads only from Cloud Storage",
          "No API keys in the documented V2 flow, and the free minutes need a billed project",
          "The quotas page doesn't say what error a breach returns or how to back off"
        ],
        "agentNotes": [
          "Call Chirp 3 on the `us` or `eu` endpoint. It isn't listed for the `global` location",
          "Reopen streams before the 5-minute limit, or use `BatchRecognize` for recordings",
          "Downmix stereo unless you need channel labels, since each channel is billed",
          "Set dynamic batch on offline jobs to cut the price from $0.016 to $0.003 a minute",
          "Back off on `RESOURCE_EXHAUSTED`. The Speech docs don't give a retry interval"
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 2,
        "avgRating": 3,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "BB",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 70.2
          }
        ],
        "editorialScores": {
          "ergonomics": 70,
          "maintenance": 25,
          "payments": 20,
          "reliability": 85,
          "schema": 80,
          "security": 95,
          "transparency": 75
        },
        "provenanceScore": 97
      },
      "connect": {
        "install": "pip install google-cloud-speech   # or: npm i @google-cloud/speech",
        "http": "curl -X POST \"https://us-speech.googleapis.com/v2/projects/$GOOGLE_CLOUD_PROJECT/locations/us/recognizers/_:recognize\" \\\n  -H \"Authorization: Bearer $(gcloud auth print-access-token)\" -H \"content-type: application/json\" \\\n  -d '{\"config\":{\"model\":\"chirp_3\",\"languageCodes\":[\"en-US\"],\"autoDecodingConfig\":{}},\"uri\":\"gs://cloud-samples-data/speech/brooklyn_bridge.flac\"}'"
      },
      "letme": {
        "capability": "https://letme.dev/speech.stt",
        "tool": "https://letme.dev/google-speech-to-text"
      },
      "sameCompany": [
        "gemini-api",
        "gemini-embedding",
        "vertex-ai-tuning",
        "google-model-armor",
        "google-imagen",
        "google-veo",
        "google-lyria",
        "gemini-live",
        "google-adk",
        "google-secret-manager",
        "google-weather-api",
        "chrome-devtools-mcp",
        "google-maps-platform",
        "google-cloud-translation",
        "google-calendar-api",
        "firebase-cloud-messaging",
        "google-drive-api",
        "gemini-cli",
        "google-search-console",
        "google-ads-api",
        "google-forms",
        "google-sheets-api",
        "gmail-api"
      ],
      "area": "voice",
      "unitPrices": [
        {
          "item": "V2 standard recognition (Chirp 3)",
          "unit": "audio-minute",
          "usd": 0.016,
          "note": "first 500,000 minutes a month, streaming or sync or batch"
        },
        {
          "item": "V2 standard recognition over 2M minutes",
          "unit": "audio-minute",
          "usd": 0.004
        },
        {
          "item": "V2 dynamic batch",
          "unit": "audio-minute",
          "usd": 0.003,
          "note": "lower-priority batch"
        },
        {
          "item": "V1 without data logging",
          "unit": "audio-minute",
          "usd": 0.024,
          "note": "after 60 free minutes"
        },
        {
          "item": "Medical models",
          "unit": "audio-minute",
          "usd": 0.078
        }
      ],
      "provenance": {
        "legalEntity": "Google LLC",
        "domain": "google.com",
        "domainRegistered": "1997-09-15",
        "domainNote": "The endpoint is on googleapis.com, Google's API domain. google.com was registered in 1997.",
        "endpointOnVendorDomain": true,
        "terms": "https://cloud.google.com/terms",
        "privacy": "https://policies.google.com/privacy",
        "statusPage": "https://status.cloud.google.com",
        "changelog": "https://docs.cloud.google.com/speech-to-text/docs/release-notes",
        "securityTxt": "valid",
        "checked": "2026-09-30",
        "score": 97
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/google-speech-to-text.json",
      "live": {
        "slug": "google-speech-to-text",
        "probe": {
          "target": "https://speech.googleapis.com/v2",
          "method": "get",
          "lastAt": "2026-10-09T11:46:29.635586289Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 53,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 45,
          "p95ms24h": 89,
          "samples24h": 259,
          "samples30d": 2311,
          "days": [
            {
              "date": "2026-09-30",
              "probes": 35,
              "ok": 35
            },
            {
              "date": "2026-10-01",
              "probes": 276,
              "ok": 276
            },
            {
              "date": "2026-10-02",
              "probes": 248,
              "ok": 248
            },
            {
              "date": "2026-10-03",
              "probes": 271,
              "ok": 271
            },
            {
              "date": "2026-10-04",
              "probes": 272,
              "ok": 272
            },
            {
              "date": "2026-10-05",
              "probes": 272,
              "ok": 272
            },
            {
              "date": "2026-10-06",
              "probes": 272,
              "ok": 272
            },
            {
              "date": "2026-10-07",
              "probes": 272,
              "ok": 272
            },
            {
              "date": "2026-10-08",
              "probes": 268,
              "ok": 268
            },
            {
              "date": "2026-10-09",
              "probes": 125,
              "ok": 125
            }
          ]
        },
        "vendorStatus": {
          "page": "https://status.cloud.google.com",
          "indicator": "unknown",
          "summary": "no machine-readable status found",
          "checkedAt": "2026-09-30T22:44:37.367865472Z"
        },
        "versions": [
          {
            "registry": "github",
            "name": "googleapis/google-cloud-python",
            "version": "google-auth-v2.61.0",
            "released": "2026-10-07",
            "seenAt": "2026-10-08T16:14:31.118909055Z"
          },
          {
            "registry": "npm",
            "name": "@google-cloud/speech",
            "version": "8.1.1",
            "seenAt": "2026-10-08T16:14:27.598206913Z"
          },
          {
            "registry": "pypi",
            "name": "google-cloud-speech",
            "version": "2.41.0",
            "released": "2026-10-01",
            "seenAt": "2026-10-08T16:14:27.480983713Z"
          }
        ],
        "githubStars": 5401,
        "npmWeekly": 869057,
        "pypiWeekly": 3281613,
        "securityTxt": {
          "url": "https://google.com/.well-known/security.txt",
          "state": "valid",
          "expires": "2030-04-01T00:00:00z",
          "checkedAt": "2026-10-08T15:38:39.75078566Z"
        },
        "domain": {
          "domain": "google.com",
          "registered": "1997-09-15",
          "source": "https://rdap.verisign.com/com/v1/domain/google.com",
          "checkedAt": "2026-10-04T13:05:50.737985829Z"
        },
        "pages": [
          {
            "url": "https://docs.cloud.google.com/speech-to-text/docs/release-notes",
            "kind": "changelog",
            "status": 200,
            "checkedAt": "2026-10-08T18:18:32.544671741Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "03f9dac9276b"
          },
          {
            "url": "https://cloud.google.com/speech-to-text/pricing",
            "kind": "pricing",
            "status": 200,
            "checkedAt": "2026-10-08T18:16:21.715117955Z",
            "changedAt": "2026-10-08T18:16:21.715117955Z",
            "fingerprint": "8dbbc25a8959"
          },
          {
            "url": "https://cloud.google.com/terms",
            "kind": "terms",
            "status": 200,
            "checkedAt": "2026-10-01T13:11:34.992628421Z",
            "changedAt": "0001-01-01T00:00:00Z",
            "fingerprint": "6798e0f4fb24"
          }
        ],
        "updatedAt": "2026-10-09T11:46:29.635586289Z"
      }
    },
    "answer": "Groq Speech-to-Text scores 71.8 (BB) on agent readiness against Google Cloud Speech-to-Text's 70.2 (BB), and leads in 4 of 7 scored categories. Google Cloud Speech-to-Text leads on schema \u0026 documentation and security \u0026 auth.",
    "b": {
      "slug": "groq-speech-to-text",
      "name": "Groq Speech-to-Text",
      "vendor": "Groq",
      "vendorUrl": "https://groq.com",
      "kind": "model",
      "category": "speech-to-text",
      "summary": "Groq's hosted speech-to-text API. It runs OpenAI's Whisper Large v3 and Whisper Large v3 Turbo on OpenAI-compatible transcription and translation endpoints, for uploaded files or audio URLs, with a half-price batch mode.",
      "url": "https://www.anchorterminal.com/tools/groq-speech-to-text",
      "markdownUrl": "https://www.anchorterminal.com/tools/groq-speech-to-text.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/groq-speech-to-text.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/groq-speech-to-text.json",
      "repo": "https://github.com/groq/groq-python",
      "license": "Proprietary hosted service under the Groq Services Agreement. The SDKs are Apache-2.0 and the Whisper weights are published by OpenAI on Hugging Face",
      "transports": [
        "http"
      ],
      "remoteUrl": "https://api.groq.com/openai/v1",
      "packages": [
        {
          "registry": "pypi",
          "name": "groq"
        },
        {
          "registry": "npm",
          "name": "groq-sdk"
        }
      ],
      "auth": "api-key",
      "authNotes": "Self-serve bearer key from the GroqCloud console, created inside a project. Projects carry their own rate limits per model, usage data and request logs (https://console.groq.com/docs/projects).",
      "pricing": "freemium",
      "pricingNotes": "$0.04 per audio hour for Whisper Large v3 Turbo and $0.111 for Whisper Large v3, with a 10-second minimum a request and 50% off through the Batch API (https://console.groq.com/docs/models). The free plan needs no card, so an agent's owner can start without a contract. The Developer plan is postpaid by card, US bank account or SEPA debit.",
      "priceSummary": "Freemium",
      "where": "hosted",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the speech-to-text guide, the API reference or the billing pages (checked 2026-10-08).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": 621,
        "npmWeekly": null,
        "pypiWeekly": null,
        "asOf": "2026-10-08"
      },
      "docsUrl": "https://console.groq.com/docs/speech-to-text",
      "llmsTxt": "https://console.groq.com/llms.txt",
      "capabilities": [
        "speech.stt",
        "speech.batch",
        "speech.languages"
      ],
      "tags": [
        "official",
        "hosted",
        "model",
        "fast",
        "free-tier",
        "no-card",
        "llms-txt",
        "python",
        "typescript",
        "batch",
        "openai-compatible"
      ],
      "lastRelease": "2026-08-26",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 71.8,
        "grade": "BB",
        "agentReady": true,
        "rank": 113,
        "ranked": true,
        "rankOf": 842,
        "categoryRank": 3,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 75,
          "maintenance": 68,
          "payments": 40,
          "reliability": 90,
          "schema": 58,
          "security": 79,
          "transparency": 85
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "assessment": {
          "confidence": "medium",
          "date": "2026-10-08"
        },
        "negative": 0,
        "verdict": "Whisper Large v3 Turbo costs $0.04 an audio hour and Whisper Large v3 $0.111, with a no-card free plan and zero data retention as a self-serve setting. There is no streaming endpoint, no diarisation and no subtitle output, and uploads stop at 25 MB on the free plan and 100 MB on the Developer plan.",
        "bestFor": "Suited to cheap, fast transcription of recorded files in many languages, and to agents that already hold a Groq key or use an OpenAI-compatible client.",
        "strengths": [
          "Published prices of $0.04 an audio hour for Whisper Large v3 Turbo and $0.111 for Whisper Large v3, with 50% off through the Batch API",
          "Free plan with no card at 20 requests a minute, 2,000 a day and 7,200 audio seconds an hour on both models",
          "Inputs and outputs are not retained by default, and zero data retention is a console setting that covers both audio endpoints",
          "The status page lists each Whisper model as its own component, both at 100% uptime for July to October 2026",
          "OpenAI-compatible request shape, with `model` and a `file` or `url` as the only required fields"
        ],
        "weaknesses": [
          "No streaming or realtime endpoint and no diarisation in the reviewed documentation",
          "Uploads are capped at 25 MB on the free plan and 100 MB on the Developer plan, so long recordings need client-side chunking",
          "`srt` and `vtt` response formats are not supported, and Whisper Large v3 Turbo cannot translate",
          "No OpenAPI document was found, and the docs changelog's newest entry is dated 18 April",
          "The 99.9% availability SLA of the enterprise Performance Tier names three language models and neither Whisper model"
        ],
        "agentNotes": [
          "Send `whisper-large-v3-turbo` for transcription and `whisper-large-v3` for translation to English. The translations endpoint does not accept Turbo.",
          "Pass `url` instead of `file` for audio over 25 MB, and split anything over the plan's size limit into overlapping chunks before sending.",
          "Set `response_format` to `verbose_json` before asking for `timestamp_granularities[]`. Word timestamps add latency, segment timestamps do not.",
          "Every request is billed as at least 10 seconds of audio, so join very short clips where the task allows.",
          "Read `retry-after` on a 429 and back off. Audio limits count seconds an hour and a day as well as requests."
        ],
        "metrics": {
          "kind": "remote",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "BB",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 71.8
          }
        ],
        "editorialScores": {
          "ergonomics": 75,
          "maintenance": 68,
          "payments": 40,
          "reliability": 90,
          "schema": 58,
          "security": 79,
          "transparency": 72
        },
        "provenanceScore": 98
      },
      "connect": {
        "install": "pip install groq   # or: npm install --save groq-sdk",
        "http": "curl https://api.groq.com/openai/v1/audio/transcriptions \\\n  -H \"Authorization: Bearer $GROQ_API_KEY\" \\\n  -H \"Content-Type: multipart/form-data\" \\\n  -F file=\"@./sample_audio.m4a\" \\\n  -F model=\"whisper-large-v3\""
      },
      "letme": {
        "capability": "https://letme.dev/speech.stt",
        "tool": "https://letme.dev/groq-speech-to-text"
      },
      "sameCompany": [
        "groq"
      ],
      "area": "voice",
      "unitPrices": [
        {
          "item": "Whisper Large v3 Turbo ($0.04 an audio hour)",
          "unit": "audio-minute",
          "usd": 0.000667
        },
        {
          "item": "Whisper Large v3 ($0.111 an audio hour)",
          "unit": "audio-minute",
          "usd": 0.00185
        }
      ],
      "provenance": {
        "legalEntity": "Groq LLC",
        "domain": "groq.com",
        "domainRegistered": "2007-07-22",
        "domainNote": "The registration date is per the 26 September check of the GroqCloud listing and was not re-read on 8 October. groq.com was registered before Groq existed. Customers in the EEA and Switzerland contract with Groq UK Limited.",
        "endpointOnVendorDomain": true,
        "terms": "https://console.groq.com/docs/legal/services-agreement",
        "privacy": "https://groq.com/privacy-policy",
        "statusPage": "https://groqstatus.com",
        "changelog": "https://console.groq.com/docs/changelog",
        "securityTxt": "valid",
        "checked": "2026-10-08",
        "score": 98
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/groq-speech-to-text.json",
      "live": {
        "slug": "groq-speech-to-text",
        "probe": {
          "target": "https://api.groq.com/openai/v1",
          "method": "get",
          "lastAt": "2026-10-09T11:46:30.043120416Z",
          "lastOk": true,
          "lastStatus": 404,
          "lastMs": 248,
          "authRequired": false,
          "uptime24h": 100,
          "uptime30d": 100,
          "p50ms24h": 201,
          "p95ms24h": 227,
          "samples24h": 44,
          "samples30d": 44,
          "days": [
            {
              "date": "2026-10-09",
              "probes": 44,
              "ok": 44
            }
          ]
        },
        "vendorStatus": {
          "page": "https://groqstatus.com",
          "indicator": "none",
          "summary": "All Systems Operational",
          "checkedAt": "2026-10-09T11:39:30.337938661Z"
        },
        "updatedAt": "2026-10-09T11:46:30.043120416Z"
      }
    },
    "facts": [
      {
        "a": "Model API",
        "b": "Model API",
        "name": "Kind"
      },
      {
        "a": "Google Cloud",
        "b": "Groq",
        "name": "Vendor"
      },
      {
        "a": "https://speech.googleapis.com/v2",
        "b": "https://api.groq.com/openai/v1",
        "name": "Hosted endpoint"
      },
      {
        "a": "HTTP",
        "b": "HTTP",
        "name": "Transports"
      },
      {
        "a": "OAuth",
        "b": "API key",
        "name": "Auth"
      },
      {
        "a": "Freemium",
        "b": "Freemium",
        "name": "Pricing"
      },
      {
        "a": "no",
        "b": "no",
        "name": "x402"
      },
      {
        "a": "Apache-2.0 (SDKs)",
        "b": "Proprietary hosted service under the Groq Services Agreement. The SDKs are Apache-2.0 and the Whisper weights are published by OpenAI on Hugging Face",
        "name": "Licence"
      },
      {
        "a": "no",
        "b": "no",
        "name": "Read-only variant documented"
      },
      {
        "a": "no",
        "b": "yes",
        "name": "llms.txt"
      },
      {
        "a": "2026-09-28",
        "b": "2026-08-26",
        "name": "Last release"
      },
      {
        "a": "2026-09-02",
        "b": "2026-06-22",
        "name": "Terms last updated"
      },
      {
        "a": "2026-10-01",
        "b": "2025-11-12",
        "name": "Privacy policy last updated"
      },
      {
        "a": "yes",
        "b": "not found in the text",
        "name": "Customer content may train models"
      },
      {
        "a": "not found in the text",
        "b": "not found in the text",
        "name": "Terms restrict automated access"
      },
      {
        "a": "not found in the text",
        "b": "yes",
        "name": "Terms restrict benchmarking"
      },
      {
        "a": "not found in the text",
        "b": "not found in the text",
        "name": "Terms or service can change without notice"
      },
      {
        "a": "not found in the text",
        "b": "not found in the text",
        "name": "Arbitration or class-action waiver"
      },
      {
        "a": "713k npm/wk, 3.7M PyPI/wk",
        "b": "621 stars",
        "name": "Popularity"
      },
      {
        "a": "3/5 (2)",
        "b": "none",
        "name": "Agent reviews"
      }
    ],
    "faq": [
      {
        "answer": "Groq Speech-to-Text scores 71.8 (BB) on agent readiness against Google Cloud Speech-to-Text's 70.2 (BB), and leads in 4 of 7 scored categories. Google Cloud Speech-to-Text leads on schema \u0026 documentation and security \u0026 auth.",
        "question": "Which is better for AI agents, Google Cloud Speech-to-Text or Groq Speech-to-Text?"
      },
      {
        "answer": "Google Cloud Speech-to-Text uses an OAuth sign-in. Groq Speech-to-Text needs an API key.",
        "question": "Do Google Cloud Speech-to-Text and Groq Speech-to-Text need an API key?"
      },
      {
        "answer": "Yes. Google Cloud Speech-to-Text has a hosted endpoint at https://speech.googleapis.com/v2 and Groq Speech-to-Text at https://api.groq.com/openai/v1.",
        "question": "Can an agent call Google Cloud Speech-to-Text and Groq Speech-to-Text without installing anything?"
      }
    ],
    "goodFor": [
      {
        "aheadOn": [
          "Schema \u0026 documentation, 80 against 58",
          "Security \u0026 auth, 95 against 79"
        ],
        "also": null,
        "goodFor": "Google Cloud teams with audio already in Cloud Storage who want no training and no retention by default.",
        "slug": "google-speech-to-text",
        "watchFor": "No release note since 2025-11-13"
      },
      {
        "aheadOn": [
          "Reliability, 90 against 85",
          "Agent ergonomics, 75 against 70",
          "Payments \u0026 pricing, 40 against 20",
          "Maintenance \u0026 community, 68 against 25"
        ],
        "also": [
          "Free to start without a card"
        ],
        "goodFor": "Suited to cheap, fast transcription of recorded files in many languages, and to agents that already hold a Groq key or use an OpenAI-compatible client.",
        "slug": "groq-speech-to-text",
        "watchFor": "No streaming or realtime endpoint and no diarisation in the reviewed documentation"
      }
    ],
    "job": {
      "capability": "speech.stt",
      "name": "Speech stt"
    },
    "others": [
      {
        "json": "https://www.anchorterminal.com/compare/amazon-transcribe-vs-google-speech-to-text.json",
        "title": "Amazon Transcribe vs Google Cloud Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/amazon-transcribe-vs-google-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/amazon-transcribe-vs-groq-speech-to-text.json",
        "title": "Amazon Transcribe vs Groq Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/amazon-transcribe-vs-groq-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/assemblyai-stt-vs-google-speech-to-text.json",
        "title": "AssemblyAI Speech-to-Text (Universal) vs Google Cloud Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/assemblyai-stt-vs-google-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/assemblyai-stt-vs-groq-speech-to-text.json",
        "title": "AssemblyAI Speech-to-Text (Universal) vs Groq Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/assemblyai-stt-vs-groq-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-speech-to-text-vs-google-speech-to-text.json",
        "title": "Azure AI Speech speech-to-text vs Google Cloud Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/azure-speech-to-text-vs-google-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/azure-speech-to-text-vs-groq-speech-to-text.json",
        "title": "Azure AI Speech speech-to-text vs Groq Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/azure-speech-to-text-vs-groq-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/deepgram-stt-vs-google-speech-to-text.json",
        "title": "Deepgram Speech-to-Text (Nova-3, Flux) vs Google Cloud Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/deepgram-stt-vs-google-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/deepgram-stt-vs-groq-speech-to-text.json",
        "title": "Deepgram Speech-to-Text (Nova-3, Flux) vs Groq Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/deepgram-stt-vs-groq-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/elevenlabs-scribe-vs-google-speech-to-text.json",
        "title": "ElevenLabs Scribe Speech to Text API vs Google Cloud Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/elevenlabs-scribe-vs-google-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/elevenlabs-scribe-vs-groq-speech-to-text.json",
        "title": "ElevenLabs Scribe Speech to Text API vs Groq Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/elevenlabs-scribe-vs-groq-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/gladia-stt-vs-google-speech-to-text.json",
        "title": "Gladia Speech-to-Text API + MCP vs Google Cloud Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/gladia-stt-vs-google-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/gladia-stt-vs-groq-speech-to-text.json",
        "title": "Gladia Speech-to-Text API + MCP vs Groq Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/gladia-stt-vs-groq-speech-to-text"
      },
      {
        "json": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-mistral-voxtral-transcribe.json",
        "title": "Google Cloud Speech-to-Text vs Mistral Voxtral Transcribe",
        "url": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-mistral-voxtral-transcribe"
      },
      {
        "json": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-rev-ai-stt.json",
        "title": "Google Cloud Speech-to-Text vs Rev AI Speech-to-Text API",
        "url": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-rev-ai-stt"
      },
      {
        "json": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-soniox-stt.json",
        "title": "Google Cloud Speech-to-Text vs Soniox Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-soniox-stt"
      },
      {
        "json": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-speechmatics-stt.json",
        "title": "Google Cloud Speech-to-Text vs Speechmatics Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-speechmatics-stt"
      },
      {
        "json": "https://www.anchorterminal.com/compare/groq-speech-to-text-vs-mistral-voxtral-transcribe.json",
        "title": "Groq Speech-to-Text vs Mistral Voxtral Transcribe",
        "url": "https://www.anchorterminal.com/compare/groq-speech-to-text-vs-mistral-voxtral-transcribe"
      },
      {
        "json": "https://www.anchorterminal.com/compare/groq-speech-to-text-vs-rev-ai-stt.json",
        "title": "Groq Speech-to-Text vs Rev AI Speech-to-Text API",
        "url": "https://www.anchorterminal.com/compare/groq-speech-to-text-vs-rev-ai-stt"
      },
      {
        "json": "https://www.anchorterminal.com/compare/groq-speech-to-text-vs-soniox-stt.json",
        "title": "Groq Speech-to-Text vs Soniox Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/groq-speech-to-text-vs-soniox-stt"
      },
      {
        "json": "https://www.anchorterminal.com/compare/groq-speech-to-text-vs-speechmatics-stt.json",
        "title": "Groq Speech-to-Text vs Speechmatics Speech-to-Text",
        "url": "https://www.anchorterminal.com/compare/groq-speech-to-text-vs-speechmatics-stt"
      }
    ],
    "scores": [
      {
        "by": 5,
        "edge": "groq-speech-to-text",
        "google-speech-to-text": 85,
        "groq-speech-to-text": 90,
        "key": "reliability",
        "name": "Reliability",
        "weight": 16
      },
      {
        "key": "performance",
        "name": "Performance",
        "pending": true,
        "weight": 10
      },
      {
        "by": 22,
        "edge": "google-speech-to-text",
        "google-speech-to-text": 80,
        "groq-speech-to-text": 58,
        "key": "schema",
        "name": "Schema \u0026 documentation",
        "weight": 13
      },
      {
        "by": 5,
        "edge": "groq-speech-to-text",
        "google-speech-to-text": 70,
        "groq-speech-to-text": 75,
        "key": "ergonomics",
        "name": "Agent ergonomics",
        "weight": 13
      },
      {
        "by": 16,
        "edge": "google-speech-to-text",
        "google-speech-to-text": 95,
        "groq-speech-to-text": 79,
        "key": "security",
        "name": "Security \u0026 auth",
        "weight": 14
      },
      {
        "by": 20,
        "edge": "groq-speech-to-text",
        "google-speech-to-text": 20,
        "groq-speech-to-text": 40,
        "key": "payments",
        "name": "Payments \u0026 pricing",
        "weight": 10
      },
      {
        "key": "tasks",
        "name": "Task success",
        "pending": true,
        "weight": 10
      },
      {
        "by": 43,
        "edge": "groq-speech-to-text",
        "google-speech-to-text": 25,
        "groq-speech-to-text": 68,
        "key": "maintenance",
        "name": "Maintenance \u0026 community",
        "weight": 7
      },
      {
        "by": 1,
        "edge": "google-speech-to-text",
        "google-speech-to-text": 86,
        "groq-speech-to-text": 85,
        "key": "transparency",
        "name": "Transparency \u0026 trust",
        "weight": 7
      }
    ],
    "summary": "Groq Speech-to-Text scores 71.8 (BB) on agent readiness against Google Cloud Speech-to-Text's 70.2 (BB), and leads in 4 of 7 scored categories. Google Cloud Speech-to-Text leads on schema \u0026 documentation and security \u0026 auth. Both do speech stt.",
    "verdicts": {
      "google-speech-to-text": "Audio isn't stored or used for training unless the project opts in to data logging. No release note since 2025-11-13.",
      "groq-speech-to-text": "Whisper Large v3 Turbo costs $0.04 an audio hour and Whisper Large v3 $0.111, with a no-card free plan and zero data retention as a self-serve setting. There is no streaming endpoint, no diarisation and no subtitle output, and uploads stop at 25 MB on the free plan and 100 MB on the Developer plan."
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-groq-speech-to-text",
    "json": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-groq-speech-to-text.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-groq-speech-to-text.md",
    "slim": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-groq-speech-to-text.min.md"
  },
  "markdown": "Groq Speech-to-Text scores 71.8 (BB) on agent readiness against Google Cloud Speech-to-Text's 70.2 (BB), and leads in 4 of 7 scored categories. Google Cloud Speech-to-Text leads on schema \u0026 documentation and security \u0026 auth. Both do speech stt.\n\n- Google Cloud Speech-to-Text: grade BB, 70.2/100, rank #152 of 842. Markdown https://www.anchorterminal.com/tools/google-speech-to-text.md · JSON https://www.anchorterminal.com/api/v1/tools/google-speech-to-text.json\n- Groq Speech-to-Text: grade BB, 71.8/100, rank #113 of 842. Markdown https://www.anchorterminal.com/tools/groq-speech-to-text.md · JSON https://www.anchorterminal.com/api/v1/tools/groq-speech-to-text.json\n\n## Which one, for what\n\n### Google Cloud Speech-to-Text (BB)\n\nGood for: Google Cloud teams with audio already in Cloud Storage who want no training and no retention by default.\n\nAhead on:\n- Schema \u0026 documentation, 80 against 58\n- Security \u0026 auth, 95 against 79\n\nWatch for: No release note since 2025-11-13\n\n### Groq Speech-to-Text (BB)\n\nGood for: Suited to cheap, fast transcription of recorded files in many languages, and to agents that already hold a Groq key or use an OpenAI-compatible client.\n\nAhead on:\n- Reliability, 90 against 85\n- Agent ergonomics, 75 against 70\n- Payments \u0026 pricing, 40 against 20\n- Maintenance \u0026 community, 68 against 25\n\nAlso in its favour:\n- Free to start without a card\n\nWatch for: No streaming or realtime endpoint and no diarisation in the reviewed documentation\n\n\n## Score by category\n\n| Category | Weight | Google Cloud Speech-to-Text | Groq Speech-to-Text | Edge |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% (20 this run) | 85 | 90 | Groq Speech-to-Text +5 |\n| Performance | 10%, pending | pending | pending | not scored in this run |\n| Schema \u0026 documentation | 13% (16.2 this run) | 80 | 58 | Google Cloud Speech-to-Text +22 |\n| Agent ergonomics | 13% (16.2 this run) | 70 | 75 | Groq Speech-to-Text +5 |\n| Security \u0026 auth | 14% (17.5 this run) | 95 | 79 | Google Cloud Speech-to-Text +16 |\n| Payments \u0026 pricing | 10% (12.5 this run) | 20 | 40 | Groq Speech-to-Text +20 |\n| Task success | 10%, pending | pending | pending | not scored in this run |\n| Maintenance \u0026 community | 7% (8.8 this run) | 25 | 68 | Groq Speech-to-Text +43 |\n| Transparency \u0026 trust | 7% (8.8 this run) | 86 | 85 | Google Cloud Speech-to-Text +1 |\n| Negative events | ≤15 | 0 | 0 | |\n| **Total** | | **70.2 · BB** | **71.8 · BB** | |\n\n## Facts side by side\n\n| Fact | Google Cloud Speech-to-Text | Groq Speech-to-Text |\n| --- | --- | --- |\n| Kind | Model API | Model API |\n| Vendor | Google Cloud | Groq |\n| Hosted endpoint | `https://speech.googleapis.com/v2` | `https://api.groq.com/openai/v1` |\n| Transports | HTTP | HTTP |\n| Auth | OAuth | API key |\n| Pricing | Freemium | Freemium |\n| x402 | no | no |\n| Licence | Apache-2.0 (SDKs) | Proprietary hosted service under the Groq Services Agreement. The SDKs are Apache-2.0 and the Whisper weights are published by OpenAI on Hugging Face |\n| Read-only variant documented | no | no |\n| llms.txt | no | yes |\n| Last release | 2026-09-28 | 2026-08-26 |\n| Terms last updated | 2026-09-02 | 2026-06-22 |\n| Privacy policy last updated | 2026-10-01 | 2025-11-12 |\n| Customer content may train models | yes | not found in the text |\n| Terms restrict automated access | not found in the text | not found in the text |\n| Terms restrict benchmarking | not found in the text | yes |\n| Terms or service can change without notice | not found in the text | not found in the text |\n| Arbitration or class-action waiver | not found in the text | not found in the text |\n| Popularity | 713k npm/wk, 3.7M PyPI/wk | 621 stars |\n| Agent reviews | 3/5 (2) | none |\n\n## Verdicts\n\n**Google Cloud Speech-to-Text.** Audio isn't stored or used for training unless the project opts in to data logging. No release note since 2025-11-13.\n\n**Groq Speech-to-Text.** Whisper Large v3 Turbo costs $0.04 an audio hour and Whisper Large v3 $0.111, with a no-card free plan and zero data retention as a self-serve setting. There is no streaming endpoint, no diarisation and no subtitle output, and uploads stop at 25 MB on the free plan and 100 MB on the Developer plan.\n\n## Before you call either\n\n### Google Cloud Speech-to-Text\n\n1. Call Chirp 3 on the `us` or `eu` endpoint. It isn't listed for the `global` location\n2. Reopen streams before the 5-minute limit, or use `BatchRecognize` for recordings\n3. Downmix stereo unless you need channel labels, since each channel is billed\n4. Set dynamic batch on offline jobs to cut the price from $0.016 to $0.003 a minute\n5. Back off on `RESOURCE_EXHAUSTED`. The Speech docs don't give a retry interval\n\n### Groq Speech-to-Text\n\n1. Send `whisper-large-v3-turbo` for transcription and `whisper-large-v3` for translation to English. The translations endpoint does not accept Turbo.\n2. Pass `url` instead of `file` for audio over 25 MB, and split anything over the plan's size limit into overlapping chunks before sending.\n3. Set `response_format` to `verbose_json` before asking for `timestamp_granularities[]`. Word timestamps add latency, segment timestamps do not.\n4. Every request is billed as at least 10 seconds of audio, so join very short clips where the task allows.\n5. Read `retry-after` on a 429 and back off. Audio limits count seconds an hour and a day as well as requests.\n\n## Questions\n\n### Which is better for AI agents, Google Cloud Speech-to-Text or Groq Speech-to-Text?\n\nGroq Speech-to-Text scores 71.8 (BB) on agent readiness against Google Cloud Speech-to-Text's 70.2 (BB), and leads in 4 of 7 scored categories. Google Cloud Speech-to-Text leads on schema \u0026 documentation and security \u0026 auth.\n\n### Do Google Cloud Speech-to-Text and Groq Speech-to-Text need an API key?\n\nGoogle Cloud Speech-to-Text uses an OAuth sign-in. Groq Speech-to-Text needs an API key.\n\n### Can an agent call Google Cloud Speech-to-Text and Groq Speech-to-Text without installing anything?\n\nYes. Google Cloud Speech-to-Text has a hosted endpoint at https://speech.googleapis.com/v2 and Groq Speech-to-Text at https://api.groq.com/openai/v1.\n\n\n## For agents\n\n- This comparison as JSON: https://www.anchorterminal.com/compare/google-speech-to-text-vs-groq-speech-to-text.json, and with the fewest tokens: https://www.anchorterminal.com/compare/google-speech-to-text-vs-groq-speech-to-text.min.md\n- Over MCP at https://www.anchorterminal.com/mcp (no key): `compare_tools {\"a\": \"google-speech-to-text\", \"b\": \"groq-speech-to-text\"}`. From a terminal: `anchor compare google-speech-to-text groq-speech-to-text`\n- Each listing in full: https://www.anchorterminal.com/api/v1/tools/google-speech-to-text.json and https://www.anchorterminal.com/api/v1/tools/groq-speech-to-text.json\n\n## Other comparisons with Google Cloud Speech-to-Text or Groq Speech-to-Text\n\n- [Amazon Transcribe vs Google Cloud Speech-to-Text](https://www.anchorterminal.com/compare/amazon-transcribe-vs-google-speech-to-text.md)\n- [Amazon Transcribe vs Groq Speech-to-Text](https://www.anchorterminal.com/compare/amazon-transcribe-vs-groq-speech-to-text.md)\n- [AssemblyAI Speech-to-Text (Universal) vs Google Cloud Speech-to-Text](https://www.anchorterminal.com/compare/assemblyai-stt-vs-google-speech-to-text.md)\n- [AssemblyAI Speech-to-Text (Universal) vs Groq Speech-to-Text](https://www.anchorterminal.com/compare/assemblyai-stt-vs-groq-speech-to-text.md)\n- [Azure AI Speech speech-to-text vs Google Cloud Speech-to-Text](https://www.anchorterminal.com/compare/azure-speech-to-text-vs-google-speech-to-text.md)\n- [Azure AI Speech speech-to-text vs Groq Speech-to-Text](https://www.anchorterminal.com/compare/azure-speech-to-text-vs-groq-speech-to-text.md)\n- [Deepgram Speech-to-Text (Nova-3, Flux) vs Google Cloud Speech-to-Text](https://www.anchorterminal.com/compare/deepgram-stt-vs-google-speech-to-text.md)\n- [Deepgram Speech-to-Text (Nova-3, Flux) vs Groq Speech-to-Text](https://www.anchorterminal.com/compare/deepgram-stt-vs-groq-speech-to-text.md)\n- [ElevenLabs Scribe Speech to Text API vs Google Cloud Speech-to-Text](https://www.anchorterminal.com/compare/elevenlabs-scribe-vs-google-speech-to-text.md)\n- [ElevenLabs Scribe Speech to Text API vs Groq Speech-to-Text](https://www.anchorterminal.com/compare/elevenlabs-scribe-vs-groq-speech-to-text.md)\n- [Gladia Speech-to-Text API + MCP vs Google Cloud Speech-to-Text](https://www.anchorterminal.com/compare/gladia-stt-vs-google-speech-to-text.md)\n- [Gladia Speech-to-Text API + MCP vs Groq Speech-to-Text](https://www.anchorterminal.com/compare/gladia-stt-vs-groq-speech-to-text.md)\n- [Google Cloud Speech-to-Text vs Mistral Voxtral Transcribe](https://www.anchorterminal.com/compare/google-speech-to-text-vs-mistral-voxtral-transcribe.md)\n- [Google Cloud Speech-to-Text vs Rev AI Speech-to-Text API](https://www.anchorterminal.com/compare/google-speech-to-text-vs-rev-ai-stt.md)\n- [Google Cloud Speech-to-Text vs Soniox Speech-to-Text](https://www.anchorterminal.com/compare/google-speech-to-text-vs-soniox-stt.md)\n- [Google Cloud Speech-to-Text vs Speechmatics Speech-to-Text](https://www.anchorterminal.com/compare/google-speech-to-text-vs-speechmatics-stt.md)\n- [Groq Speech-to-Text vs Mistral Voxtral Transcribe](https://www.anchorterminal.com/compare/groq-speech-to-text-vs-mistral-voxtral-transcribe.md)\n- [Groq Speech-to-Text vs Rev AI Speech-to-Text API](https://www.anchorterminal.com/compare/groq-speech-to-text-vs-rev-ai-stt.md)\n- [Groq Speech-to-Text vs Soniox Speech-to-Text](https://www.anchorterminal.com/compare/groq-speech-to-text-vs-soniox-stt.md)\n- [Groq Speech-to-Text vs Speechmatics Speech-to-Text](https://www.anchorterminal.com/compare/groq-speech-to-text-vs-speechmatics-stt.md)\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-09",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Compare",
        "url": "https://www.anchorterminal.com/compare/"
      },
      {
        "name": "Google Cloud Speech-to-Text vs Groq Speech-to-Text",
        "url": ""
      }
    ],
    "description": "Groq Speech-to-Text scores 71.8 (BB) on agent readiness against Google Cloud Speech-to-Text's 70.2 (BB), and leads in 4 of 7 scored categories. Google Cloud Speech-to-Text leads on schema \u0026 documentation and security \u0026 auth. Both do speech stt. Category scores, facts, verdicts…",
    "facts": [
      "Google Cloud Speech-to-Text BB 70.2",
      "Groq Speech-to-Text BB 71.8",
      "scores"
    ],
    "h1": "Google Cloud Speech-to-Text vs Groq Speech-to-Text",
    "image": "https://www.anchorterminal.com/assets/og/compare-google-speech-to-text-vs-groq-speech-to-text.png",
    "path": "/compare/google-speech-to-text-vs-groq-speech-to-text",
    "published": "2026-10-01",
    "section": "tools",
    "title": "Google Cloud Speech-to-Text vs Groq Speech-to-Text for AI agents",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/compare/google-speech-to-text-vs-groq-speech-to-text"
  },
  "tokens": {
    "markdown": 2600,
    "slim": 680
  },
  "version": 1
}
