{
  "data": {
    "similar": [
      {
        "grade": "BB",
        "json": "https://www.anchorterminal.com/tools/cohere-embed.json",
        "name": "Cohere Embed and Rerank",
        "score": 72.5,
        "shared": [
          "embed.text",
          "embed.multimodal",
          "embed.multilingual",
          "rerank"
        ],
        "slug": "cohere-embed"
      },
      {
        "grade": "C",
        "json": "https://www.anchorterminal.com/tools/jina-embeddings.json",
        "name": "Jina Embeddings and Reranker",
        "score": 61,
        "shared": [
          "embed.text",
          "embed.multimodal",
          "embed.multilingual",
          "rerank"
        ],
        "slug": "jina-embeddings"
      },
      {
        "grade": "C",
        "json": "https://www.anchorterminal.com/tools/voyage-ai.json",
        "name": "Voyage AI embeddings and rerankers",
        "score": 58.8,
        "shared": [
          "embed.text",
          "embed.multimodal",
          "embed.multilingual",
          "rerank"
        ],
        "slug": "voyage-ai"
      },
      {
        "grade": "BB",
        "json": "https://www.anchorterminal.com/tools/gemini-embedding.json",
        "name": "Gemini Embedding",
        "score": 70.6,
        "shared": [
          "embed.text",
          "embed.multimodal",
          "embed.multilingual"
        ],
        "slug": "gemini-embedding"
      },
      {
        "grade": "D",
        "json": "https://www.anchorterminal.com/tools/nomic-embed.json",
        "name": "Nomic Embed",
        "score": 49.2,
        "shared": [
          "embed.text",
          "embed.multimodal",
          "embed.multilingual"
        ],
        "slug": "nomic-embed"
      },
      {
        "grade": "F",
        "json": "https://www.anchorterminal.com/tools/zeroentropy.json",
        "name": "ZeroEntropy zerank and zembed",
        "score": 13.7,
        "shared": [
          "rerank",
          "embed.text",
          "embed.multilingual"
        ],
        "slug": "zeroentropy"
      }
    ],
    "tool": {
      "slug": "nvidia-nemo-retriever",
      "name": "NVIDIA NeMo Retriever Embedding and Reranking NIMs",
      "vendor": "NVIDIA",
      "vendorUrl": "https://www.nvidia.com",
      "kind": "http-api",
      "category": "embeddings",
      "summary": "NVIDIA's NeMo Retriever Embedding and Reranking NIMs are GPU containers that run text and image embedding models and rerankers behind a local REST API, with `/v1/embeddings` in the OpenAI shape and `/v1/ranking`.",
      "url": "https://www.anchorterminal.com/tools/nvidia-nemo-retriever",
      "markdownUrl": "https://www.anchorterminal.com/tools/nvidia-nemo-retriever.md",
      "slimMarkdownUrl": "https://www.anchorterminal.com/tools/nvidia-nemo-retriever.min.md",
      "jsonUrl": "https://www.anchorterminal.com/api/v1/tools/nvidia-nemo-retriever.json",
      "license": "Proprietary containers under the NVIDIA Software Licence Agreement and Product-Specific Terms for AI Products. Models carry their own licences, such as OpenMDW 1.1 for `nvidia/nemotron-3-embed-1b` and the NVIDIA Open Model Licence for the Llama Nemotron models",
      "transports": [
        "http"
      ],
      "packages": [
        {
          "registry": "pypi",
          "name": "langchain-nvidia-ai-endpoints"
        }
      ],
      "auth": "none",
      "authNotes": "The NIM's own API takes no credential. NVIDIA's security page says the deployer must secure the endpoints and suggests a proxy with HTTPS. Pulling images from `nvcr.io` needs a personal NGC API key, created by a person at org.ngc.nvidia.com and sent as the password for the user `$oauthtoken`. Model weights download from Hugging Face by default with `HF_TOKEN`, or from NGC with `NGC_API_KEY`. Some models need their licence terms accepted on the NGC catalogue page first. TLS is built in through `NIM_SERVER_TLS_CERT_PATH` and `NIM_SERVER_TLS_KEY_PATH`.",
      "pricing": "freemium",
      "pricingNotes": "Free for research, development and testing on up to 16 GPUs through the NVIDIA developer programme, with no card. Production use needs NVIDIA AI Enterprise, listed at $4,500 a GPU a year through partners or $1 a GPU-hour on AWS, Azure, Google Cloud and Oracle marketplaces, plus the instance. A 90-day AI Enterprise trial licence is available on request. The hosted trial endpoints on build.nvidia.com are free for prototyping (https://docs.nvidia.com/ai-enterprise/planning-resource/licensing-guide/latest/pricing.html, checked 2026-10-08).",
      "priceSummary": "Freemium",
      "where": "local",
      "x402": {
        "level": "no",
        "evidence": "No x402, MPP or L402 in the NIM docs, the OpenAPI files or the AI Enterprise pricing page (checked 2026-10-08).",
        "endpoints": []
      },
      "toolCount": null,
      "popularity": {
        "githubStars": null,
        "npmWeekly": null,
        "pypiWeekly": 133630,
        "asOf": "2026-10-08"
      },
      "docsUrl": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/overview.html",
      "openapi": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/_downloads/9957cfb9472fcf23c93820c1922c4343/openai-api.openapi.yaml",
      "capabilities": [
        "embed.text",
        "embed.multimodal",
        "embed.multilingual",
        "rerank"
      ],
      "tags": [
        "self-hosted",
        "docker",
        "kubernetes",
        "gpu",
        "openai-compatible",
        "openapi",
        "proprietary",
        "enterprise",
        "free-for-development",
        "multimodal"
      ],
      "lastRelease": "2026-08-05",
      "graded": true,
      "anchor": {
        "graded": true,
        "score": 61,
        "grade": "C",
        "agentReady": false,
        "rank": 439,
        "ranked": true,
        "rankOf": 842,
        "categoryRank": 6,
        "methodology": "0.4",
        "run": "2026-10-01",
        "scores": {
          "ergonomics": 73,
          "maintenance": 57,
          "payments": 40,
          "reliability": 53,
          "schema": 78,
          "security": 55,
          "transparency": 71
        },
        "pending": [
          "performance",
          "tasks"
        ],
        "breakdown": [
          {
            "key": "reliability",
            "name": "Reliability",
            "weight": 16,
            "effectiveWeight": 20,
            "score": 53,
            "points": 10.6,
            "reason": "Local-software reading, for the containers the owner runs. Signed images install from `nvcr.io` with a support matrix naming GPUs, drivers and CPUs (20). The runtime is closed source and no public CI or test suite was found (0 of 25). There is no public issue tracker. The release notes list three known issues for the Embedding NIM at 2.3, among them a Docker health check that reports unhealthy when the NIM is ready, and two for the Reranking NIM (10 of 25). Release 2.0 lists renamed, deprecated and removed environment variables, but the names changed again by 2.3 (`NIM_BIND_ADDR` to `NIM_SERVER_BIND_ADDR`, `NIM_PRECISION` to `NIM_ENGINE_PRECISION`) with no note in the 2.2 or 2.3 notes we read, and no release carries a date (8 of 15). Version 2.3, with production branches named (15)."
          },
          {
            "key": "performance",
            "name": "Performance",
            "weight": 10,
            "effectiveWeight": 0,
            "pending": true,
            "points": 0,
            "reason": "Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes."
          },
          {
            "key": "schema",
            "name": "Schema \u0026 documentation",
            "weight": 13,
            "effectiveWeight": 16.25,
            "score": 78,
            "points": 12.68,
            "reason": "OpenAPI 3.1 files to download for both services. The embedding file says version 2.2.0 and the reranking file 1.11.0 while the docs are at 2.3 (25). docs.nvidia.com has an llms.txt and a NIM index, but neither links these docs and the `.html.md` form answers 404 (2 of 10). The usage pages say when to send `query` or `passage`, which embedding type suits which case and what each model can't do (15 of 20). Enums on `input_type`, `modality`, `embedding_type`, `truncate` and `dimensions`, length limits and no extra properties allowed (14 of 15). Curl examples with responses and a table of nine error messages, though the OpenAPI file gives errors a description only (12 of 15). Versioned release notes per service, without dates (10 of 15)."
          },
          {
            "key": "ergonomics",
            "name": "Agent ergonomics",
            "weight": 13,
            "effectiveWeight": 16.25,
            "score": 73,
            "points": 11.86,
            "reason": "Vectors can be shortened with `dimensions` on three models or packed as int8 or binary, and returned as base64 (20 of 25). One call takes up to 8,192 inputs or 512 passages with a `truncate` setting, but `/v1/ranking` has no top-n field (14 of 20). Errors are JSON with a message that names the bad field and the allowed values (16 of 20). Calls are stateless and safe to repeat, and 503 is described as try again, with no backoff guidance (14 of 20). Two required fields and the OpenAI request shape, so OpenAI clients work with a model-name suffix. NVIDIA publishes no SDK of its own for these services, and its guide uses the LangChain package (9 of 15)."
          },
          {
            "key": "security",
            "name": "Security \u0026 auth",
            "weight": 14,
            "effectiveWeight": 17.5,
            "score": 55,
            "points": 9.63,
            "reason": "The API takes no credential. NVIDIA's security page says the deployer must add authentication, and an NGC key is needed only to pull images (10 of 30). The service is stateless inference with no write or delete operation, a configurable bind address and built-in TLS, and no access control of its own (12 of 20). It returns vectors and scores, not untrusted content (10). Prometheus metrics, OpenTelemetry traces and JSON logs go to the operator's own stack, with no per-caller log because there are no callers' identities (10 of 15). NVIDIA PSIRT runs coordinated disclosure and publishes bulletins, images are signed and scanned on NGC with VEX documents, and no security.txt or bug bounty was found (13 of 20)."
          },
          {
            "key": "payments",
            "name": "Payments \u0026 pricing",
            "weight": 10,
            "effectiveWeight": 12.5,
            "score": 40,
            "points": 5,
            "reason": "Proprietary software with a paid licence, so scored on that licence and not by the free self-hosted rule. No x402, MPP or L402 (0). The licensing guide lists $4,500 a GPU a year and $1 a GPU-hour on cloud marketplaces without a login (20). The developer programme allows research, development and testing on up to 16 GPUs free, and build.nvidia.com says its trial endpoints need no card (20). A person has to create an NGC account and key, and accept licence terms in a browser for some models (0)."
          },
          {
            "key": "tasks",
            "name": "Task success",
            "weight": 10,
            "effectiveWeight": 0,
            "pending": true,
            "points": 0,
            "reason": "Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored."
          },
          {
            "key": "maintenance",
            "name": "Maintenance \u0026 community",
            "weight": 7,
            "effectiveWeight": 8.75,
            "score": 57,
            "points": 4.99,
            "reason": "The newest image we could date is `nemotron-3-embed-1b`, updated on NGC on 5 August 2026, 64 days before the check, with Embedding 2.3 pushed on 3 August (20 of 30). Three dated pushes fall inside 90 days across the two services (Reranking 2.3.0 on 27 July, Embedding 2.3 on 3 August, the 5 August update), read from the NGC registry because the release notes have no dates (15 of 20). Closed service reading for responsiveness. Release notes and a developer forum exist, and we did not read reply times (6 of 15). No NVIDIA SDK. `langchain-nvidia-ai-endpoints` 1.4.3 of 2 July 2026 sits in the langchain-ai repository (8 of 15). Images are multi-architecture, signed and rescanned on 5 October 2026 (8 of 10)."
          },
          {
            "key": "transparency",
            "name": "Transparency \u0026 trust",
            "weight": 7,
            "effectiveWeight": 8.75,
            "score": 71,
            "points": 6.21,
            "note": "editorial 59, provenance 83",
            "reason": "Closed runtime under published terms, with model licences named per model and weights for `nvidia/nemotron-3-embed-1b` on Hugging Face (18 of 30). Inputs stay on the deployer's hardware. The licence agreement says software may collect configuration, performance and usage data, and the AI product terms say NIMs that collect telemetry are switched with `NIM_TELEMETRY_MODE`, a variable release 2.0 removed with no replacement. The current environment variable page names no telemetry setting, so the statements don't fully agree (15 of 30). The AI Enterprise lifecycle policy gives one month of support for a feature branch and nine for a production branch, and the end-of-life table dates the NV-EmbedQA-E5-v5 retirement to January 2027 (18 of 20). Telemetry is disclosed in the terms with an opt-out that the 2.x docs no longer list (8 of 20)."
          }
        ],
        "assessment": {
          "date": "2026-10-08",
          "basis": "public evidence",
          "confidence": "medium",
          "notes": {
            "ergonomics": "Vectors can be shortened with `dimensions` on three models or packed as int8 or binary, and returned as base64 (20 of 25). One call takes up to 8,192 inputs or 512 passages with a `truncate` setting, but `/v1/ranking` has no top-n field (14 of 20). Errors are JSON with a message that names the bad field and the allowed values (16 of 20). Calls are stateless and safe to repeat, and 503 is described as try again, with no backoff guidance (14 of 20). Two required fields and the OpenAI request shape, so OpenAI clients work with a model-name suffix. NVIDIA publishes no SDK of its own for these services, and its guide uses the LangChain package (9 of 15).",
            "maintenance": "The newest image we could date is `nemotron-3-embed-1b`, updated on NGC on 5 August 2026, 64 days before the check, with Embedding 2.3 pushed on 3 August (20 of 30). Three dated pushes fall inside 90 days across the two services (Reranking 2.3.0 on 27 July, Embedding 2.3 on 3 August, the 5 August update), read from the NGC registry because the release notes have no dates (15 of 20). Closed service reading for responsiveness. Release notes and a developer forum exist, and we did not read reply times (6 of 15). No NVIDIA SDK. `langchain-nvidia-ai-endpoints` 1.4.3 of 2 July 2026 sits in the langchain-ai repository (8 of 15). Images are multi-architecture, signed and rescanned on 5 October 2026 (8 of 10).",
            "payments": "Proprietary software with a paid licence, so scored on that licence and not by the free self-hosted rule. No x402, MPP or L402 (0). The licensing guide lists $4,500 a GPU a year and $1 a GPU-hour on cloud marketplaces without a login (20). The developer programme allows research, development and testing on up to 16 GPUs free, and build.nvidia.com says its trial endpoints need no card (20). A person has to create an NGC account and key, and accept licence terms in a browser for some models (0).",
            "reliability": "Local-software reading, for the containers the owner runs. Signed images install from `nvcr.io` with a support matrix naming GPUs, drivers and CPUs (20). The runtime is closed source and no public CI or test suite was found (0 of 25). There is no public issue tracker. The release notes list three known issues for the Embedding NIM at 2.3, among them a Docker health check that reports unhealthy when the NIM is ready, and two for the Reranking NIM (10 of 25). Release 2.0 lists renamed, deprecated and removed environment variables, but the names changed again by 2.3 (`NIM_BIND_ADDR` to `NIM_SERVER_BIND_ADDR`, `NIM_PRECISION` to `NIM_ENGINE_PRECISION`) with no note in the 2.2 or 2.3 notes we read, and no release carries a date (8 of 15). Version 2.3, with production branches named (15).",
            "schema": "OpenAPI 3.1 files to download for both services. The embedding file says version 2.2.0 and the reranking file 1.11.0 while the docs are at 2.3 (25). docs.nvidia.com has an llms.txt and a NIM index, but neither links these docs and the `.html.md` form answers 404 (2 of 10). The usage pages say when to send `query` or `passage`, which embedding type suits which case and what each model can't do (15 of 20). Enums on `input_type`, `modality`, `embedding_type`, `truncate` and `dimensions`, length limits and no extra properties allowed (14 of 15). Curl examples with responses and a table of nine error messages, though the OpenAPI file gives errors a description only (12 of 15). Versioned release notes per service, without dates (10 of 15).",
            "security": "The API takes no credential. NVIDIA's security page says the deployer must add authentication, and an NGC key is needed only to pull images (10 of 30). The service is stateless inference with no write or delete operation, a configurable bind address and built-in TLS, and no access control of its own (12 of 20). It returns vectors and scores, not untrusted content (10). Prometheus metrics, OpenTelemetry traces and JSON logs go to the operator's own stack, with no per-caller log because there are no callers' identities (10 of 15). NVIDIA PSIRT runs coordinated disclosure and publishes bulletins, images are signed and scanned on NGC with VEX documents, and no security.txt or bug bounty was found (13 of 20).",
            "transparency": "Closed runtime under published terms, with model licences named per model and weights for `nvidia/nemotron-3-embed-1b` on Hugging Face (18 of 30). Inputs stay on the deployer's hardware. The licence agreement says software may collect configuration, performance and usage data, and the AI product terms say NIMs that collect telemetry are switched with `NIM_TELEMETRY_MODE`, a variable release 2.0 removed with no replacement. The current environment variable page names no telemetry setting, so the statements don't fully agree (15 of 30). The AI Enterprise lifecycle policy gives one month of support for a feature branch and nine for a production branch, and the end-of-life table dates the NV-EmbedQA-E5-v5 retirement to January 2027 (18 of 20). Telemetry is disclosed in the terms with an opt-out that the 2.x docs no longer list (8 of 20)."
          },
          "sources": [
            {
              "what": "Embedding NIM overview and docs index",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/overview.html",
              "seen": "2026-10-08"
            },
            {
              "what": "Embedding NIM release notes for 2.3",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/release-notes.html",
              "seen": "2026-10-08"
            },
            {
              "what": "Embedding NIM release notes for 2.2 and 2.2.2",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/2.2/release-notes.html",
              "seen": "2026-10-08"
            },
            {
              "what": "Embedding NIM release notes for 2.0, environment variable renames and removals",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/2.0/release-notes.html",
              "seen": "2026-10-08"
            },
            {
              "what": "support matrix, models, token limits, GPUs",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/support-matrix.html",
              "seen": "2026-10-08"
            },
            {
              "what": "getting started, NGC key, launch command, first request",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/getting-started.html",
              "seen": "2026-10-08"
            },
            {
              "what": "API usage, request fields, limits and error table",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/use-the-api-openai.html",
              "seen": "2026-10-08"
            },
            {
              "what": "OpenAPI 3.1 file for the Embedding NIM",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/_downloads/9957cfb9472fcf23c93820c1922c4343/openai-api.openapi.yaml",
              "seen": "2026-10-08"
            },
            {
              "what": "security and authentication page",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/security.html",
              "seen": "2026-10-08"
            },
            {
              "what": "environment variables, TLS and logging",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/environment-variables.html",
              "seen": "2026-10-08"
            },
            {
              "what": "observability, Prometheus and OpenTelemetry",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/observability.html",
              "seen": "2026-10-08"
            },
            {
              "what": "governing terms per model",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/eula.html",
              "seen": "2026-10-08"
            },
            {
              "what": "Reranking NIM release notes",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/reranking/latest/release-notes.html",
              "seen": "2026-10-08"
            },
            {
              "what": "Reranking NIM support matrix",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/reranking/latest/support-matrix.html",
              "seen": "2026-10-08"
            },
            {
              "what": "OpenAPI 3.1 file for the Reranking NIM",
              "url": "https://docs.nvidia.com/nim/nemo-retriever/reranking/latest/_downloads/81b47bbd12351cffe2ef6a1df5467755/ranking.openapi.yaml",
              "seen": "2026-10-08"
            },
            {
              "what": "NGC registry record for the nemotron-3-embed-1b image, tags, signing and scan dates",
              "url": "https://api.ngc.nvidia.com/v2/repos/nim/nvidia/nemotron-3-embed-1b",
              "seen": "2026-10-08"
            },
            {
              "what": "NGC image list with push dates, Embedding VL image",
              "url": "https://api.ngc.nvidia.com/v2/repos/nim/nvidia/llama-nemotron-embed-vl-1b-v2/images",
              "seen": "2026-10-08"
            },
            {
              "what": "NGC image list with push dates, Reranking VL image",
              "url": "https://api.ngc.nvidia.com/v2/repos/nim/nvidia/llama-nemotron-rerank-vl-1b-v2/images",
              "seen": "2026-10-08"
            },
            {
              "what": "NVIDIA Software Licence Agreement, version of 7 May 2026",
              "url": "https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-software-license-agreement/",
              "seen": "2026-10-08"
            },
            {
              "what": "Product-Specific Terms for AI Products, 15 April 2026",
              "url": "https://www.nvidia.com/en-us/agreements/enterprise-software/product-specific-terms-for-ai-products/",
              "seen": "2026-10-08"
            },
            {
              "what": "NVIDIA AI Enterprise pricing",
              "url": "https://docs.nvidia.com/ai-enterprise/planning-resource/licensing-guide/latest/pricing.html",
              "seen": "2026-10-08"
            },
            {
              "what": "NIM FAQ, developer programme and production licence",
              "url": "https://forums.developer.nvidia.com/t/nvidia-nim-faq/300317",
              "seen": "2026-10-08"
            },
            {
              "what": "AI Enterprise lifecycle, application branches",
              "url": "https://docs.nvidia.com/ai-enterprise/lifecycle/latest/application-software.html",
              "seen": "2026-10-08"
            },
            {
              "what": "AI Enterprise end-of-life notices",
              "url": "https://docs.nvidia.com/ai-enterprise/lifecycle/latest/eol-notices.html",
              "seen": "2026-10-08"
            },
            {
              "what": "privacy policy, effective 22 September 2025",
              "url": "https://www.nvidia.com/en-us/about-nvidia/privacy-policy/",
              "seen": "2026-10-08"
            },
            {
              "what": "product security page and PSIRT policies",
              "url": "https://www.nvidia.com/en-us/security/psirt-policies/",
              "seen": "2026-10-08"
            },
            {
              "what": "NGC status incidents",
              "url": "https://status.ngc.nvidia.com/api/v2/incidents.json",
              "seen": "2026-10-08"
            },
            {
              "what": "hosted catalogue llms.txt and model page",
              "url": "https://build.nvidia.com/nvidia/nemotron-3-embed-1b.md",
              "seen": "2026-10-08"
            },
            {
              "what": "docs llms.txt index",
              "url": "https://docs.nvidia.com/llms.txt",
              "seen": "2026-10-08"
            },
            {
              "what": "LangChain client package",
              "url": "https://pypi.org/pypi/langchain-nvidia-ai-endpoints/json",
              "seen": "2026-10-08"
            }
          ],
          "openQuestions": [
            "unchecked: whether the 2.x containers send any telemetry to NVIDIA. Release 2.0 removed `NIM_TELEMETRY_MODE`, the current environment variable page names no telemetry setting, and we did not run a container",
            "unchecked: the image list for `nvcr.io/nim/nvidia/nemotron-3-embed-1b`, which NGC's API refused (403, not a public artifact). The repository record lists tags 2.2, 2.2.0, 2.2.1, 2.2.2, 2 and latest, while the getting-started guide pulls `:2.3`",
            "Release dates. The release notes give none, so the dates here are image push dates from the NGC registry",
            "unchecked: reply times on the NVIDIA developer forum for NIM questions",
            "unchecked: the NVIDIA API Trial Terms of Service (a PDF) and the rate limits of the hosted trial endpoints, which were not graded",
            "unchecked: SOC 2 or ISO 27001 coverage. No certification page was read, and the product is software the owner runs",
            "No product-specific privacy document was found. `provenance.privacy` points at NVIDIA's general privacy policy, which the model cards name as applicable",
            "The lead's docs URL under `/text-embedding/latest/` redirects to `/embedding/latest/`, and the hosted endpoints the lead left unchecked do exist as a trial service"
          ]
        },
        "negative": 0,
        "verdict": "Self-hosted containers with OpenAPI 3.1 files, typed request fields, five embedding output types and a dated end-of-life list. The API has no authentication or rate limiting of its own, production use needs an NVIDIA AI Enterprise licence at $4,500 a GPU a year, and the release notes carry no dates.",
        "bestFor": "Teams that already run NVIDIA GPUs and need embedding and reranking inside their own network, including page-image retrieval with the VL models.",
        "strengths": [
          "OpenAPI 3.1 files for both services, with enums for `input_type`, `modality`, `embedding_type` and `truncate` and no extra properties allowed",
          "`/v1/embeddings` follows the OpenAI shape, and a `-query` or `-passage` model suffix replaces `input_type` for OpenAI clients",
          "Output can be shrunk with `dimensions` from 128 to 2048 on three models, or with `int8`, `uint8`, `binary` and `ubinary` types",
          "NGC images are signed, built for amd64 and arm64, and were last scanned on 5 October 2026 per the NGC registry record",
          "The AI Enterprise lifecycle pages give support periods per branch and an end-of-life table with dates"
        ],
        "weaknesses": [
          "The API has no authentication and no rate limiting. The security page leaves both to a proxy the deployer runs",
          "Production use needs an NVIDIA AI Enterprise licence, $4,500 a GPU a year or $1 a GPU-hour on cloud marketplaces, plus the GPU",
          "Release notes carry no dates, and environment variable names changed between 2.0 and 2.3 without a note in the 2.2 or 2.3 notes we read",
          "The NVIDIA Software Licence Agreement forbids disclosing benchmark results without written permission, apart from a published exception",
          "Closed-source runtime with no public issue tracker or test suite, and no llms.txt entry or Markdown pages for these docs"
        ],
        "agentNotes": [
          "Send `input_type` as `query` or `passage` on every embedding call. Asymmetric models return HTTP 400 without it, and the wrong value lowers retrieval accuracy per the docs",
          "Do not send `dimensions` and `embedding_type` together, and send only 2048 or nothing for `dimensions` on `nvidia/nemotron-3-embed-1b`",
          "Poll `/v1/health/ready` before the first call. The Docker health check can report unhealthy while the NIM is ready, per the 2.3 known issues",
          "Check the image tag on NGC before pulling. The guide uses `nemotron-3-embed-1b:2.3`, and NGC's record for that image listed tags up to 2.2.2 on 8 October 2026",
          "Put a proxy with authentication and TLS in front of port 8000, and sort `/v1/ranking` results yourself as the request has no top-n field"
        ],
        "metrics": {
          "kind": "local",
          "measured": false
        },
        "reviewCount": 0,
        "avgRating": 0,
        "history": [
          {
            "basis": "public evidence",
            "confidence": "medium",
            "grade": "C",
            "methodology": "0.4",
            "pending": [
              "performance",
              "tasks"
            ],
            "run": "2026-10-01",
            "runLabel": "October 2026 research run",
            "score": 61
          }
        ],
        "editorialScores": {
          "ergonomics": 73,
          "maintenance": 57,
          "payments": 40,
          "reliability": 53,
          "schema": 78,
          "security": 55,
          "transparency": 59
        },
        "provenanceScore": 83
      },
      "connect": {
        "install": "echo \"$NGC_API_KEY\" | docker login nvcr.io --username '$oauthtoken' --password-stdin\ndocker run -it --rm --runtime=nvidia --gpus all --shm-size=16GB -e HF_TOKEN -v ~/.cache/nim/cache:/opt/cache -v ~/.cache/nim/weights:/model -u $(id -u) -p 8000:8000 nvcr.io/nim/nvidia/nemotron-3-embed-1b:2.3",
        "http": "curl -X POST http://localhost:8000/v1/embeddings \\\n  -H 'accept: application/json' -H 'Content-Type: application/json' \\\n  -d '{\"input\":[\"What is NVIDIA?\"],\"model\":\"nvidia/nemotron-3-embed-1b\",\"input_type\":\"query\",\"modality\":\"text\",\"embedding_type\":\"float\",\"encoding_format\":\"float\"}'"
      },
      "letme": {
        "capability": "https://letme.dev/embed.text",
        "tool": "https://letme.dev/nvidia-nemo-retriever"
      },
      "sameCompany": [
        "nemo-guardrails"
      ],
      "notable": [
        "The Embedding NIM lists seven models, among them `nvidia/nemotron-3-embed-1b` (text, 2048 dimensions, 4,096 tokens), `nvidia/llama-nemotron-embed-vl-1b-v2` (text and image) and `baai/bge-m3` (https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/support-matrix.html)",
        "The Reranking NIM lists three models at 8,192 tokens, including `nvidia/llama-nemotron-rerank-vl-1b-v2`, which scores text, image or mixed passages against a text query (https://docs.nvidia.com/nim/nemo-retriever/reranking/latest/support-matrix.html)",
        "The security page says NIMs impose no rate limits and that the developer must secure the endpoints, suggesting a proxy and HTTPS with TLS 1.2 (https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/security.html)",
        "Release 2.0 replaced the runtime, renamed four environment variables and removed fifteen, among them `NIM_TELEMETRY_MODE`, which the AI product terms still name as the telemetry switch (https://docs.nvidia.com/nim/nemo-retriever/embedding/2.0/release-notes.html)",
        "The NVIDIA Software Licence Agreement, version of 7 May 2026, bars disclosing benchmarking or performance results without written permission, except as described in a published benchmarking document (https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-software-license-agreement/)",
        "NVIDIA's end-of-life notices mark the NV-EmbedQA-E5-v5 NIM as deprecated with action required by January 2027 (https://docs.nvidia.com/ai-enterprise/lifecycle/latest/eol-notices.html)",
        "The same models answer on NVIDIA's hosted trial API at `https://integrate.api.nvidia.com/v1` and `https://ai.api.nvidia.com/v1/retrieval/...`, under the NVIDIA API Trial Terms of Service (https://build.nvidia.com/llms.txt)"
      ],
      "area": "models",
      "details": [
        {
          "label": "Surface graded",
          "value": "The self-hosted containers from `nvcr.io/nim/nvidia/`, version 2.3. The hosted trial endpoints on build.nvidia.com are recorded and not graded"
        },
        {
          "label": "Endpoints",
          "value": "Embedding NIM: POST `/v1/embeddings`, GET `/v1/models`, `/v1/health/ready`, `/v1/health/live`, `/v1/metrics`, `/v1/metadata`, `/v1/manifest`, `/v1/version` and a licence endpoint. Reranking NIM: POST `/v1/ranking` and the same GET set. Optional KServe V2 gRPC"
        },
        {
          "label": "Embedding models",
          "value": "`nvidia/nemotron-3-embed-1b` (4,096 tokens, 2048 dimensions), `nvidia/llama-nemotron-embed-vl-1b-v2` (2,048 tokens, text and image), `nvidia/llama-nemotron-embed-1b-v2` and `nvidia/llama-nemotron-embed-300m-v2` (8,192 tokens), `nvidia/nv-embedqa-e5-v5`, `baai/bge-m3`, `baai/bge-large-zh-v1.5`"
        },
        {
          "label": "Reranking models",
          "value": "`nvidia/llama-nemotron-rerank-vl-1b-v2`, `nvidia/llama-nemotron-rerank-1b-v2`, `nvidia/llama-nemotron-rerank-500m-v2`, each at 8,192 tokens"
        },
        {
          "label": "Request limits",
          "value": "Up to 8,192 inputs a call on `/v1/embeddings` and 512 passages on `/v1/ranking` per the OpenAPI files. Inline images up to 5 MiB by default and 8192 x 16384 pixels"
        },
        {
          "label": "Output sizing",
          "value": "`dimensions` of 128, 256, 384, 512, 768, 1024, 1536 or 2048 on models with dynamic embeddings. `embedding_type` of float, int8, uint8, binary or ubinary. `encoding_format` float or base64"
        },
        {
          "label": "Errors",
          "value": "JSON with `object: \"error\"`, a message and a type. 400, 404, 415, 422 and 503 documented, with nine example messages"
        },
        {
          "label": "Hardware",
          "value": "NVIDIA GPUs from A10G and L4 to H100, H200, B200 and GB200, plus DGX Spark on arm64 from 2.3. x86 hosts need at least 8 cores. Multi-instance GPU mode and multi-GPU deployment are not supported"
        },
        {
          "label": "Licence to run",
          "value": "Free for research, development and testing on up to 16 GPUs through the NVIDIA developer programme. Production needs NVIDIA AI Enterprise, with a 90-day trial licence"
        },
        {
          "label": "Observability",
          "value": "Prometheus metrics at `/v1/metrics`, OpenTelemetry metrics and traces over OTLP HTTP with `NIM_ENABLE_OTEL=1`, and logs in pretty, JSON or compact format"
        },
        {
          "label": "Releases",
          "value": "Embedding 2.0 (image pushed 1 June 2026), 2.2 and 2.2.2, 2.3 (3 August 2026). Reranking 2.0 (1 June 2026) and 2.3 (27 July 2026). Dates are from the NGC registry, as the release notes carry none"
        },
        {
          "label": "Support branches",
          "value": "Feature branch releases are supported for one month and production branches for nine months, per the NVIDIA AI Enterprise lifecycle policy"
        }
      ],
      "unitPrices": [
        {
          "item": "NVIDIA AI Enterprise on a cloud marketplace, per GPU",
          "unit": "gpu-hour",
          "usd": 1,
          "note": "Licence only, plus the cloud instance. Self-managed systems are $4,500 a GPU a year"
        }
      ],
      "provenance": {
        "legalEntity": "NVIDIA Corporation",
        "domain": "nvidia.com",
        "domainRegistered": "1993-04-20",
        "domainNote": "Self-hosted software. The API answers on the deployer's own host, port 8000 by default. Images come from nvcr.io and the docs are on docs.nvidia.com.",
        "endpointOnVendorDomain": null,
        "terms": "https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-software-license-agreement/",
        "privacy": "https://www.nvidia.com/en-us/about-nvidia/privacy-policy/",
        "statusPage": "https://status.ngc.nvidia.com",
        "changelog": "https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/release-notes.html",
        "securityTxt": "none",
        "checked": "2026-10-08",
        "notes": [
          "The governing terms page for the Embedding NIM names the NVIDIA Software Licence Agreement (version of 7 May 2026) and the Product-Specific Terms for AI Products (15 April 2026) for the container, with a separate model licence per model.",
          "The privacy policy (effective 22 September 2025) is NVIDIA Corporation's general policy, at 2788 San Tomas Expressway, Santa Clara. The model cards name it as the applicable privacy policy. No product-specific privacy document was found.",
          "www.nvidia.com/.well-known/security.txt answers 403 with an access-denied body and www.nvidia.com/security.txt answers 404. Vulnerability reports go to NVIDIA PSIRT.",
          "status.ngc.nvidia.com covers NGC, the registry the images are pulled from, and NVIDIA Build. It does not cover a self-hosted NIM.",
          "RDAP for nvidia.com gives a registration date of 1993-04-20.",
          "The hosted trial endpoints are on integrate.api.nvidia.com and ai.api.nvidia.com, under the NVIDIA API Trial Terms of Service, a PDF on assets.ngc.nvidia.com that we did not read."
        ],
        "score": 83,
        "checks": [
          {
            "check": "Legal entity named",
            "value": "NVIDIA Corporation",
            "points": 20,
            "max": 20,
            "state": "ok"
          },
          {
            "check": "Domain age",
            "value": "nvidia.com, registered 1993-04-20 (33 years)",
            "points": 15,
            "max": 15,
            "state": "ok"
          },
          {
            "check": "Endpoint on the vendor's domain",
            "value": "no hosted endpoint",
            "points": 0,
            "max": 0,
            "state": "na"
          },
          {
            "check": "Terms of service",
            "value": "read, states 5 of the 7 things a reader expects, and has 1 clause that costs points",
            "points": 6.3,
            "max": 10,
            "state": "part"
          },
          {
            "check": "Privacy policy",
            "value": "read, states 7 of the 8 things a reader expects",
            "points": 9.3,
            "max": 10,
            "state": "part"
          },
          {
            "check": "Status page",
            "value": "status.ngc.nvidia.com",
            "points": 10,
            "max": 10,
            "state": "ok"
          },
          {
            "check": "Changelog",
            "value": "published",
            "points": 10,
            "max": 10,
            "state": "ok"
          },
          {
            "check": "security.txt",
            "value": "not found",
            "points": 0,
            "max": 10,
            "state": "no"
          }
        ],
        "policies": [
          {
            "kind": "terms",
            "url": "https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-software-license-agreement/",
            "state": "read",
            "readAt": "2026-10-08",
            "statedDate": "2026-05-07",
            "words": 11508,
            "points": 6.3,
            "max": 10,
            "expected": [
              {
                "key": "terms.date",
                "label": "Gives the date it was last updated",
                "found": true,
                "quote": "Last Modified: May 07, 2026",
                "says": "Last updated 2026-05-07"
              },
              {
                "key": "terms.law",
                "label": "Names the governing law or courts",
                "found": true,
                "quote": "The Agreement will be governed in all respects by the laws of the United States and the laws of the State of Delaware, without regard to conflict of laws principles or the United Nations Convention on Contracts for the International Sale of Goods.",
                "says": "The law of the United States"
              },
              {
                "key": "terms.liability",
                "label": "States a limit on its liability",
                "found": true,
                "quote": "…ANY AND ALL LIABILITIES, OBLIGATIONS OR CLAIMS ARISING OUT OF OR RELATED TO (I) ENTERPRISE PRODUCTS WILL NOT EXCEED THE NET AMOUNT NVIDIA WAS PAID FOR THE ENTERPRISE PRODUCTS GIVING RISE TO THE CLAIM IN THE TWELVE (12-)MONTH PERIOD BEFORE THE EVENT GIVING RISE TO THE LIABILITY, OR (II) SOFTWARE AT NO CHARGE WILL NOT E…",
                "says": "Capped at $100.00"
              },
              {
                "key": "terms.termination",
                "label": "Says how the agreement or account can be ended",
                "found": true,
                "quote": "If a payment delinquency is not cured within the cure period stated in Section 10.2 for payment obligations, NVIDIA may terminate the Agreement."
              },
              {
                "key": "terms.changes",
                "label": "Says how changes to the terms are announced",
                "found": false
              },
              {
                "key": "terms.use",
                "label": "Lists what users may not do",
                "found": true,
                "quote": "Customer may not combine the use of paid and unpaid Software, Derivative Samples and Derivative Models in a way that avoids incurring fees or exceeding use limits or quotas."
              },
              {
                "key": "terms.sla",
                "label": "Refers to a service level or uptime commitment",
                "found": false
              }
            ],
            "toKnow": [
              {
                "key": "terms.benchmark",
                "label": "Restricts benchmarking or competitive use",
                "found": true,
                "quote": "Customer may not use the Software or NVIDIA Confidential Information for the purpose of (i) developing competing products or technologies or assisting a third party in such activities, or (ii)identifying or supporting an assertion or potential assertion of any intellectual property rights against NVIDIA (including pat…",
                "costsPoints": true
              }
            ],
            "notes": [
              {
                "date": "2026-10-08",
                "text": "Liability for software supplied at no charge is capped at 100 US dollars.",
                "quote": "SOFTWARE AT NO CHARGE WILL NOT EXCEED ONE-HUNDRED US DOLLARS ($100.00 USD)"
              },
              {
                "date": "2026-10-08",
                "text": "NVIDIA or an independent auditor may audit the customer's compliance during the term and for three years after it.",
                "quote": "NVIDIA or an independent auditor will have the right to audit Customer to validate and confirm Customer’s information and compliance with the terms of the Agreement."
              },
              {
                "date": "2026-10-08",
                "text": "Orders placed directly with NVIDIA cannot be cancelled and fees received are not refunded.",
                "quote": "Each Order Form placed is non-cancelable and fees received are non-refundable."
              }
            ]
          },
          {
            "kind": "privacy",
            "url": "https://www.nvidia.com/en-us/about-nvidia/privacy-policy/",
            "state": "read",
            "readAt": "2026-10-08",
            "words": 7539,
            "points": 9.3,
            "max": 10,
            "expected": [
              {
                "key": "privacy.date",
                "label": "Gives the date it was last updated",
                "found": false
              },
              {
                "key": "privacy.collected",
                "label": "Says what personal data is collected",
                "found": true,
                "quote": "We collect your information when you search for information, order products, request to download content, register for events or demos, or give us feedback."
              },
              {
                "key": "privacy.retention",
                "label": "Says how long data is kept",
                "found": true,
                "quote": "We retain your personal data for as long as our engagement with you continues (e.g., emails, website visits, logins, or event attendance)."
              },
              {
                "key": "privacy.processors",
                "label": "Says who else receives the data",
                "found": true,
                "quote": "Our service provider hCaptcha uses this data to verify whether user actions meet our security requirements."
              },
              {
                "key": "privacy.sale",
                "label": "Says whether personal data is sold or shared for advertising",
                "found": true,
                "quote": "We use this data on ad-supported GeForce NOW plans to provide interest-based advertising, for fraud prevention, and to limit the repetition of advertisements."
              },
              {
                "key": "privacy.rights",
                "label": "Says what rights people have over their data",
                "found": true,
                "quote": "If you have an NVIDIA account or have signed up for our content with your email address, you may visit the NVIDIA Privacy Center to exercise your privacy rights."
              },
              {
                "key": "privacy.contact",
                "label": "Gives a privacy contact",
                "found": true,
                "quote": "If you are a member of the public and wish to exercise any of your privacy rights, please contact us directly at privacy@nvidia.com.",
                "says": "privacy@nvidia.com"
              },
              {
                "key": "privacy.transfers",
                "label": "Says where data is transferred or stored",
                "found": true,
                "quote": "Transfers of personal data to the United States, and to or from NVIDIA, are made subject to the Standard Contractual Clauses which have been pre-approved by the European Commission and ensure appropriate data protection safeguards.",
                "says": "Relies on standard contractual clauses"
              }
            ],
            "toKnow": [
              {
                "key": "privacy.sells",
                "label": "Says it sells personal data or shares it for advertising",
                "found": true,
                "quote": "However, our sharing of non-sensitive data with advertising providers may qualify as the sale of personal data or the sharing of personal data for purposes of targeted advertising."
              }
            ],
            "notes": [
              {
                "date": "2026-10-08",
                "text": "Some products or functions, such as demos and beta technologies, collect personal data outside this policy and come with separate privacy disclosures.",
                "quote": "From time to time, we may launch certain products or features (e.g., demos or beta technologies) that involve collection of personal data that falls outside the scope of this privacy policy."
              }
            ]
          }
        ]
      },
      "pageJsonUrl": "https://www.anchorterminal.com/tools/nvidia-nemo-retriever.json",
      "live": {
        "slug": "nvidia-nemo-retriever",
        "vendorStatus": {
          "page": "https://status.ngc.nvidia.com",
          "indicator": "none",
          "summary": "All Systems Operational",
          "checkedAt": "2026-10-09T09:25:19.901736203Z"
        },
        "updatedAt": "2026-10-09T09:25:19.901736203Z"
      }
    },
    "verify": {
      "accepts": "a page on nvidia.com or one of its subdomains",
      "badgeUrl": "https://www.anchorterminal.com/badges/nvidia-nemo-retriever.svg",
      "body": {
        "slug": "nvidia-nemo-retriever",
        "url": "the page with the badge or the link"
      },
      "docs": "https://www.anchorterminal.com/builders/#verify",
      "effect": "none, it never changes a grade, rank or review",
      "endpoint": "https://www.anchorterminal.com/api/v1/verify",
      "listingUrl": "https://www.anchorterminal.com/tools/nvidia-nemo-retriever",
      "mcpTool": "verify_listing",
      "recheck": "weekly; two failed checks in a row and it lapses, a later pass restores it",
      "snippets": {
        "html": "\u003ca href=\"https://www.anchorterminal.com/tools/nvidia-nemo-retriever\"\u003e\u003cimg src=\"https://www.anchorterminal.com/badges/nvidia-nemo-retriever.svg\" alt=\"NVIDIA NeMo Retriever Embedding and Reranking NIMs on Anchor Terminal\" height=\"20\"\u003e\u003c/a\u003e",
        "markdown": "[![NVIDIA NeMo Retriever Embedding and Reranking NIMs on Anchor Terminal](https://www.anchorterminal.com/badges/nvidia-nemo-retriever.svg)](https://www.anchorterminal.com/tools/nvidia-nemo-retriever)",
        "link": "\u003ca href=\"https://www.anchorterminal.com/tools/nvidia-nemo-retriever\"\u003eNVIDIA NeMo Retriever Embedding and Reranking NIMs on Anchor Terminal\u003c/a\u003e"
      }
    }
  },
  "kind": "anchor.page",
  "links": {
    "api": "https://www.anchorterminal.com/api/v1/index.json",
    "html": "https://www.anchorterminal.com/tools/nvidia-nemo-retriever",
    "json": "https://www.anchorterminal.com/tools/nvidia-nemo-retriever.json",
    "llms": "https://www.anchorterminal.com/llms.txt",
    "markdown": "https://www.anchorterminal.com/tools/nvidia-nemo-retriever.md",
    "slim": "https://www.anchorterminal.com/tools/nvidia-nemo-retriever.min.md"
  },
  "markdown": "## Overview\n\n**Grade C · 61/100 · rank #439 of 842 · #6 in Embeddings \u0026 rerankers · not agent-ready · confidence medium**\n\n\nMore from NVIDIA, listed separately because each is its own product: [NVIDIA NeMo Guardrails](https://www.anchorterminal.com/tools/nemo-guardrails.md) (Guardrails \u0026 safety filters).\n\n## Assessment\n\nSelf-hosted containers with OpenAPI 3.1 files, typed request fields, five embedding output types and a dated end-of-life list. The API has no authentication or rate limiting of its own, production use needs an NVIDIA AI Enterprise licence at $4,500 a GPU a year, and the release notes carry no dates.\n\n## Facts\n\n| Field | Value |\n| --- | --- |\n| Vendor | NVIDIA (https://www.nvidia.com) |\n| Kind | HTTP API |\n| Category | Embeddings \u0026 rerankers (https://www.anchorterminal.com/categories/embeddings) |\n| Transport | HTTP |\n| Auth | None · The NIM's own API takes no credential. NVIDIA's security page says the deployer must secure the endpoints and suggests a proxy with HTTPS. Pulling images from `nvcr.io` needs a personal NGC API key, created by a person at org.ngc.nvidia.com and sent as the password for the user `$oauthtoken`. Model weights download from Hugging Face by default with `HF_TOKEN`, or from NGC with `NGC_API_KEY`. Some models need their licence terms accepted on the NGC catalogue page first. TLS is built in through `NIM_SERVER_TLS_CERT_PATH` and `NIM_SERVER_TLS_KEY_PATH`. |\n| Pricing | Freemium (Freemium) · Free for research, development and testing on up to 16 GPUs through the NVIDIA developer programme, with no card. Production use needs NVIDIA AI Enterprise, listed at $4,500 a GPU a year through partners or $1 a GPU-hour on AWS, Azure, Google Cloud and Oracle marketplaces, plus the instance. A 90-day AI Enterprise trial licence is available on request. The hosted trial endpoints on build.nvidia.com are free for prototyping (https://docs.nvidia.com/ai-enterprise/planning-resource/licensing-guide/latest/pricing.html, checked 2026-10-08). |\n| x402 | No · No x402, MPP or L402 in the NIM docs, the OpenAPI files or the AI Enterprise pricing page (checked 2026-10-08). |\n| Licence | Proprietary containers under the NVIDIA Software Licence Agreement and Product-Specific Terms for AI Products. Models carry their own licences, such as OpenMDW 1.1 for `nvidia/nemotron-3-embed-1b` and the NVIDIA Open Model Licence for the Llama Nemotron models |\n| Packages | pypi: `langchain-nvidia-ai-endpoints` |\n| Docs | https://docs.nvidia.com/nim/nemo-retriever/embedding/latest/overview.html |\n| llms.txt | not found |\n| Last release | 2026-08-05 |\n| PyPI downloads / week | 133,630 |\n| Surface graded | The self-hosted containers from `nvcr.io/nim/nvidia/`, version 2.3. The hosted trial endpoints on build.nvidia.com are recorded and not graded |\n| Endpoints | Embedding NIM: POST `/v1/embeddings`, GET `/v1/models`, `/v1/health/ready`, `/v1/health/live`, `/v1/metrics`, `/v1/metadata`, `/v1/manifest`, `/v1/version` and a licence endpoint. Reranking NIM: POST `/v1/ranking` and the same GET set. Optional KServe V2 gRPC |\n| Embedding models | `nvidia/nemotron-3-embed-1b` (4,096 tokens, 2048 dimensions), `nvidia/llama-nemotron-embed-vl-1b-v2` (2,048 tokens, text and image), `nvidia/llama-nemotron-embed-1b-v2` and `nvidia/llama-nemotron-embed-300m-v2` (8,192 tokens), `nvidia/nv-embedqa-e5-v5`, `baai/bge-m3`, `baai/bge-large-zh-v1.5` |\n| Reranking models | `nvidia/llama-nemotron-rerank-vl-1b-v2`, `nvidia/llama-nemotron-rerank-1b-v2`, `nvidia/llama-nemotron-rerank-500m-v2`, each at 8,192 tokens |\n| Request limits | Up to 8,192 inputs a call on `/v1/embeddings` and 512 passages on `/v1/ranking` per the OpenAPI files. Inline images up to 5 MiB by default and 8192 x 16384 pixels |\n| Output sizing | `dimensions` of 128, 256, 384, 512, 768, 1024, 1536 or 2048 on models with dynamic embeddings. `embedding_type` of float, int8, uint8, binary or ubinary. `encoding_format` float or base64 |\n| Errors | JSON with `object: \"error\"`, a message and a type. 400, 404, 415, 422 and 503 documented, with nine example messages |\n| Hardware | NVIDIA GPUs from A10G and L4 to H100, H200, B200 and GB200, plus DGX Spark on arm64 from 2.3. x86 hosts need at least 8 cores. Multi-instance GPU mode and multi-GPU deployment are not supported |\n| Licence to run | Free for research, development and testing on up to 16 GPUs through the NVIDIA developer programme. Production needs NVIDIA AI Enterprise, with a 90-day trial licence |\n| Observability | Prometheus metrics at `/v1/metrics`, OpenTelemetry metrics and traces over OTLP HTTP with `NIM_ENABLE_OTEL=1`, and logs in pretty, JSON or compact format |\n| Releases | Embedding 2.0 (image pushed 1 June 2026), 2.2 and 2.2.2, 2.3 (3 August 2026). Reranking 2.0 (1 June 2026) and 2.3 (27 July 2026). Dates are from the NGC registry, as the release notes carry none |\n| Support branches | Feature branch releases are supported for one month and production branches for nine months, per the NVIDIA AI Enterprise lifecycle policy |\n| Capabilities | embed.text, embed.multimodal, embed.multilingual, rerank |\n| Tags | self-hosted, docker, kubernetes, gpu, openai-compatible, openapi, proprietary, enterprise, free-for-development, multimodal |\n| JSON | https://www.anchorterminal.com/api/v1/tools/nvidia-nemo-retriever.json |\n\n## Score breakdown (methodology v0.4, October 2026 research run)\n\nAssessed 2026-10-08 from public evidence against the published checklist (https://www.anchorterminal.com/benchmark/#checklist). Confidence: medium. Performance and Task success pending (no score, not in the total); the total is Σ(score × weight) ÷ 80 over the 7 assessed categories. \"This run\" is each category's share of the 100 points.\n\n| Category | Weight | This run | Score (0–100) | Points |\n| --- | --- | --- | --- | --- |\n| Reliability | 16% | 20 | 53 | 10.6 |\n| Performance | 10% | pending | pending | n/a |\n| Schema \u0026 documentation | 13% | 16.2 | 78 | 12.7 |\n| Agent ergonomics | 13% | 16.2 | 73 | 11.9 |\n| Security \u0026 auth | 14% | 17.5 | 55 | 9.6 |\n| Payments \u0026 pricing | 10% | 12.5 | 40 | 5.0 |\n| Task success | 10% | pending | pending | n/a |\n| Maintenance \u0026 community | 7% | 8.8 | 57 | 5.0 |\n| Transparency \u0026 trust (editorial 59, provenance 83) | 7% | 8.8 | 71 | 6.2 |\n| Negative events | up to −15 | up to −15 | none recorded | 0 |\n| **Total** | | | | **61 → C** |\n\n### Why each score\n\n- Reliability 53: Local-software reading, for the containers the owner runs. Signed images install from `nvcr.io` with a support matrix naming GPUs, drivers and CPUs (20). The runtime is closed source and no public CI or test suite was found (0 of 25). There is no public issue tracker. The release notes list three known issues for the Embedding NIM at 2.3, among them a Docker health check that reports unhealthy when the NIM is ready, and two for the Reranking NIM (10 of 25). Release 2.0 lists renamed, deprecated and removed environment variables, but the names changed again by 2.3 (`NIM_BIND_ADDR` to `NIM_SERVER_BIND_ADDR`, `NIM_PRECISION` to `NIM_ENGINE_PRECISION`) with no note in the 2.2 or 2.3 notes we read, and no release carries a date (8 of 15). Version 2.3, with production branches named (15).\n- Performance: Pending. Latency is measured per call by our probes, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until the first probe window closes.\n- Schema \u0026 documentation 78: OpenAPI 3.1 files to download for both services. The embedding file says version 2.2.0 and the reranking file 1.11.0 while the docs are at 2.3 (25). docs.nvidia.com has an llms.txt and a NIM index, but neither links these docs and the `.html.md` form answers 404 (2 of 10). The usage pages say when to send `query` or `passage`, which embedding type suits which case and what each model can't do (15 of 20). Enums on `input_type`, `modality`, `embedding_type`, `truncate` and `dimensions`, length limits and no extra properties allowed (14 of 15). Curl examples with responses and a table of nine error messages, though the OpenAPI file gives errors a description only (12 of 15). Versioned release notes per service, without dates (10 of 15).\n- Agent ergonomics 73: Vectors can be shortened with `dimensions` on three models or packed as int8 or binary, and returned as base64 (20 of 25). One call takes up to 8,192 inputs or 512 passages with a `truncate` setting, but `/v1/ranking` has no top-n field (14 of 20). Errors are JSON with a message that names the bad field and the allowed values (16 of 20). Calls are stateless and safe to repeat, and 503 is described as try again, with no backoff guidance (14 of 20). Two required fields and the OpenAI request shape, so OpenAI clients work with a model-name suffix. NVIDIA publishes no SDK of its own for these services, and its guide uses the LangChain package (9 of 15).\n- Security \u0026 auth 55: The API takes no credential. NVIDIA's security page says the deployer must add authentication, and an NGC key is needed only to pull images (10 of 30). The service is stateless inference with no write or delete operation, a configurable bind address and built-in TLS, and no access control of its own (12 of 20). It returns vectors and scores, not untrusted content (10). Prometheus metrics, OpenTelemetry traces and JSON logs go to the operator's own stack, with no per-caller log because there are no callers' identities (10 of 15). NVIDIA PSIRT runs coordinated disclosure and publishes bulletins, images are signed and scanned on NGC with VEX documents, and no security.txt or bug bounty was found (13 of 20).\n- Payments \u0026 pricing 40: Proprietary software with a paid licence, so scored on that licence and not by the free self-hosted rule. No x402, MPP or L402 (0). The licensing guide lists $4,500 a GPU a year and $1 a GPU-hour on cloud marketplaces without a login (20). The developer programme allows research, development and testing on up to 16 GPUs free, and build.nvidia.com says its trial endpoints need no card (20). A person has to create an NGC account and key, and accept licence terms in a browser for some models (0).\n- Task success: Pending. Task success needs the category task suites run through each tool, which haven't run yet, so this run doesn't score it. Its weight is shared across the assessed categories until then. A data provider's data-quality score is published on its listing now and becomes half of this category when it's scored.\n- Maintenance \u0026 community 57: The newest image we could date is `nemotron-3-embed-1b`, updated on NGC on 5 August 2026, 64 days before the check, with Embedding 2.3 pushed on 3 August (20 of 30). Three dated pushes fall inside 90 days across the two services (Reranking 2.3.0 on 27 July, Embedding 2.3 on 3 August, the 5 August update), read from the NGC registry because the release notes have no dates (15 of 20). Closed service reading for responsiveness. Release notes and a developer forum exist, and we did not read reply times (6 of 15). No NVIDIA SDK. `langchain-nvidia-ai-endpoints` 1.4.3 of 2 July 2026 sits in the langchain-ai repository (8 of 15). Images are multi-architecture, signed and rescanned on 5 October 2026 (8 of 10).\n- Transparency \u0026 trust 71: Closed runtime under published terms, with model licences named per model and weights for `nvidia/nemotron-3-embed-1b` on Hugging Face (18 of 30). Inputs stay on the deployer's hardware. The licence agreement says software may collect configuration, performance and usage data, and the AI product terms say NIMs that collect telemetry are switched with `NIM_TELEMETRY_MODE`, a variable release 2.0 removed with no replacement. The current environment variable page names no telemetry setting, so the statements don't fully agree (15 of 30). The AI Enterprise lifecycle policy gives one month of support for a feature branch and nine for a production branch, and the end-of-life table dates the NV-EmbedQA-E5-v5 retirement to January 2027 (18 of 20). Telemetry is disclosed in the terms with an opt-out that the 2.x docs no longer list (8 of 20).\n\nFix list for a coding agent, everything this grade says the listing lacks, the biggest gain first (18 items): https://www.anchorterminal.com/fixes/nvidia-nemo-retriever.md (JSON https://www.anchorterminal.com/fixes/nvidia-nemo-retriever.json)\n\n### What we couldn't check\n\n- unchecked: whether the 2.x containers send any telemetry to NVIDIA. Release 2.0 removed `NIM_TELEMETRY_MODE`, the current environment variable page names no telemetry setting, and we did not run a container\n- unchecked: the image list for `nvcr.io/nim/nvidia/nemotron-3-embed-1b`, which NGC's API refused (403, not a public artifact). The repository record lists tags 2.2, 2.2.0, 2.2.1, 2.2.2, 2 and latest, while the getting-started guide pulls `:2.3`\n- Release dates. The release notes give none, so the dates here are image push dates from the NGC registry\n- unchecked: reply times on the NVIDIA developer forum for NIM questions\n- unchecked: the NVIDIA API Trial Terms of Service (a PDF) and the rate limits of the hosted trial endpoints, which were not graded\n- unchecked: SOC 2 or ISO 27001 coverage. No certification page was read, and the product is software the owner runs\n- No product-specific privacy document was found. `provenance.privacy` points at NVIDIA's general privacy policy, which the model cards name as applicable\n- The lead's docs URL under `/text-embedding/latest/` redirects to `/embedding/latest/`, and the hosted endpoints the lead left unchecked do exist as a trial service\n\n### Sources\n\n- Embedding NIM overview and docs index: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/overview.html\u003e (seen 2026-10-08)\n- Embedding NIM release notes for 2.3: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/release-notes.html\u003e (seen 2026-10-08)\n- Embedding NIM release notes for 2.2 and 2.2.2: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/2.2/release-notes.html\u003e (seen 2026-10-08)\n- Embedding NIM release notes for 2.0, environment variable renames and removals: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/2.0/release-notes.html\u003e (seen 2026-10-08)\n- support matrix, models, token limits, GPUs: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/support-matrix.html\u003e (seen 2026-10-08)\n- getting started, NGC key, launch command, first request: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/getting-started.html\u003e (seen 2026-10-08)\n- API usage, request fields, limits and error table: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/use-the-api-openai.html\u003e (seen 2026-10-08)\n- OpenAPI 3.1 file for the Embedding NIM: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/_downloads/9957cfb9472fcf23c93820c1922c4343/openai-api.openapi.yaml\u003e (seen 2026-10-08)\n- security and authentication page: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/security.html\u003e (seen 2026-10-08)\n- environment variables, TLS and logging: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/environment-variables.html\u003e (seen 2026-10-08)\n- observability, Prometheus and OpenTelemetry: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/observability.html\u003e (seen 2026-10-08)\n- governing terms per model: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/eula.html\u003e (seen 2026-10-08)\n- Reranking NIM release notes: \u003chttps://docs.nvidia.com/nim/nemo-retriever/reranking/latest/release-notes.html\u003e (seen 2026-10-08)\n- Reranking NIM support matrix: \u003chttps://docs.nvidia.com/nim/nemo-retriever/reranking/latest/support-matrix.html\u003e (seen 2026-10-08)\n- OpenAPI 3.1 file for the Reranking NIM: \u003chttps://docs.nvidia.com/nim/nemo-retriever/reranking/latest/_downloads/81b47bbd12351cffe2ef6a1df5467755/ranking.openapi.yaml\u003e (seen 2026-10-08)\n- NGC registry record for the nemotron-3-embed-1b image, tags, signing and scan dates: \u003chttps://api.ngc.nvidia.com/v2/repos/nim/nvidia/nemotron-3-embed-1b\u003e (seen 2026-10-08)\n- NGC image list with push dates, Embedding VL image: \u003chttps://api.ngc.nvidia.com/v2/repos/nim/nvidia/llama-nemotron-embed-vl-1b-v2/images\u003e (seen 2026-10-08)\n- NGC image list with push dates, Reranking VL image: \u003chttps://api.ngc.nvidia.com/v2/repos/nim/nvidia/llama-nemotron-rerank-vl-1b-v2/images\u003e (seen 2026-10-08)\n- NVIDIA Software Licence Agreement, version of 7 May 2026: \u003chttps://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-software-license-agreement/\u003e (seen 2026-10-08)\n- Product-Specific Terms for AI Products, 15 April 2026: \u003chttps://www.nvidia.com/en-us/agreements/enterprise-software/product-specific-terms-for-ai-products/\u003e (seen 2026-10-08)\n- NVIDIA AI Enterprise pricing: \u003chttps://docs.nvidia.com/ai-enterprise/planning-resource/licensing-guide/latest/pricing.html\u003e (seen 2026-10-08)\n- NIM FAQ, developer programme and production licence: \u003chttps://forums.developer.nvidia.com/t/nvidia-nim-faq/300317\u003e (seen 2026-10-08)\n- AI Enterprise lifecycle, application branches: \u003chttps://docs.nvidia.com/ai-enterprise/lifecycle/latest/application-software.html\u003e (seen 2026-10-08)\n- AI Enterprise end-of-life notices: \u003chttps://docs.nvidia.com/ai-enterprise/lifecycle/latest/eol-notices.html\u003e (seen 2026-10-08)\n- privacy policy, effective 22 September 2025: \u003chttps://www.nvidia.com/en-us/about-nvidia/privacy-policy/\u003e (seen 2026-10-08)\n- product security page and PSIRT policies: \u003chttps://www.nvidia.com/en-us/security/psirt-policies/\u003e (seen 2026-10-08)\n- NGC status incidents: \u003chttps://status.ngc.nvidia.com/api/v2/incidents.json\u003e (seen 2026-10-08)\n- hosted catalogue llms.txt and model page: \u003chttps://build.nvidia.com/nvidia/nemotron-3-embed-1b.md\u003e (seen 2026-10-08)\n- docs llms.txt index: \u003chttps://docs.nvidia.com/llms.txt\u003e (seen 2026-10-08)\n- LangChain client package: \u003chttps://pypi.org/pypi/langchain-nvidia-ai-endpoints/json\u003e (seen 2026-10-08)\n\n## Who's behind it (provenance 83/100, checked 2026-10-08)\n\n| Check | Finding | Points |\n| --- | --- | --- |\n| Legal entity named | NVIDIA Corporation | 20/20 |\n| Domain age | nvidia.com, registered 1993-04-20 (33 years) | 15/15 |\n| Endpoint on the vendor's domain | no hosted endpoint | n/a |\n| Terms of service | read, states 5 of the 7 things a reader expects, and has 1 clause that costs points | 6.3/10 |\n| Privacy policy | read, states 7 of the 8 things a reader expects | 9.3/10 |\n| Status page | status.ngc.nvidia.com | 10/10 |\n| Changelog | published | 10/10 |\n| security.txt | not found | 0/10 |\n\nSelf-hosted software. The API answers on the deployer's own host, port 8000 by default. Images come from nvcr.io and the docs are on docs.nvidia.com.\n\nThe governing terms page for the Embedding NIM names the NVIDIA Software Licence Agreement (version of 7 May 2026) and the Product-Specific Terms for AI Products (15 April 2026) for the container, with a separate model licence per model.\n\nThe privacy policy (effective 22 September 2025) is NVIDIA Corporation's general policy, at 2788 San Tomas Expressway, Santa Clara. The model cards name it as the applicable privacy policy. No product-specific privacy document was found.\n\nwww.nvidia.com/.well-known/security.txt answers 403 with an access-denied body and www.nvidia.com/security.txt answers 404. Vulnerability reports go to NVIDIA PSIRT.\n\nstatus.ngc.nvidia.com covers NGC, the registry the images are pulled from, and NVIDIA Build. It does not cover a self-hosted NIM.\n\nRDAP for nvidia.com gives a registration date of 1993-04-20.\n\nThe hosted trial endpoints are on integrate.api.nvidia.com and ai.api.nvidia.com, under the NVIDIA API Trial Terms of Service, a PDF on assets.ngc.nvidia.com that we did not read.\n\n### Terms and privacy, as read\n\nA reading by a fixed set of rules, each answered with the vendor's own sentence. Not legal advice.\n\n**Terms of service** (https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-software-license-agreement/), read 2026-10-08, dated 2026-05-07, states 5 of the 7 things a reader expects.\n\n- To know. Restricts benchmarking or competitive use (costs points). \"Customer may not use the Software or NVIDIA Confidential Information for the purpose of (i) developing competing products or technologies or assisting a third party in such activities, or (ii)identifying or supporting an assertion or potential assertion of any intellectual property rights against NVIDIA (including pat…\"\n- Gives the date it was last updated. Last updated 2026-05-07.\n- Names the governing law or courts. The law of the United States.\n- States a limit on its liability. Capped at $100.00.\n- Not found in the text. Says how changes to the terms are announced.\n- Not found in the text. Refers to a service level or uptime commitment.\n- Also in the text (2026-10-08). Liability for software supplied at no charge is capped at 100 US dollars. \"SOFTWARE AT NO CHARGE WILL NOT EXCEED ONE-HUNDRED US DOLLARS ($100.00 USD)\"\n- Also in the text (2026-10-08). NVIDIA or an independent auditor may audit the customer's compliance during the term and for three years after it. \"NVIDIA or an independent auditor will have the right to audit Customer to validate and confirm Customer’s information and compliance with the terms of the Agreement.\"\n- Also in the text (2026-10-08). Orders placed directly with NVIDIA cannot be cancelled and fees received are not refunded. \"Each Order Form placed is non-cancelable and fees received are non-refundable.\"\n\n**Privacy policy** (https://www.nvidia.com/en-us/about-nvidia/privacy-policy/), read 2026-10-08, gives no date, states 7 of the 8 things a reader expects.\n\n- To know. Says it sells personal data or shares it for advertising. \"However, our sharing of non-sensitive data with advertising providers may qualify as the sale of personal data or the sharing of personal data for purposes of targeted advertising.\"\n- Not found in the text. Gives the date it was last updated.\n- Gives a privacy contact. privacy@nvidia.com.\n- Says where data is transferred or stored. Relies on standard contractual clauses.\n- Also in the text (2026-10-08). Some products or functions, such as demos and beta technologies, collect personal data outside this policy and come with separate privacy disclosures. \"From time to time, we may launch certain products or features (e.g., demos or beta technologies) that involve collection of personal data that falls outside the scope of this privacy policy.\"\n\n## Live (updated 2026-10-09 09:25 UTC)\n\n- Vendor status page: none, All Systems Operational\n- Always current: https://www.anchorterminal.com/api/v1/live/nvidia-nemo-retriever.json\n\n## Probe metrics\n\nNot measured yet. Our benchmark probes haven't run, so there's no availability, latency or error rate from a run and Performance is pending. Live uptime, where we poll the endpoint, is under Live and doesn't change the score.\n\n## Prices\n\n| Item | Price | Unit | Note |\n| --- | --- | --- | --- |\n| NVIDIA AI Enterprise on a cloud marketplace, per GPU | $1 | per GPU-hour | Licence only, plus the cloud instance. Self-managed systems are $4,500 a GPU a year |\n\nAcross all listings: https://www.anchorterminal.com/prices/index.md\n\n## Strengths\n\n- OpenAPI 3.1 files for both services, with enums for `input_type`, `modality`, `embedding_type` and `truncate` and no extra properties allowed\n- `/v1/embeddings` follows the OpenAI shape, and a `-query` or `-passage` model suffix replaces `input_type` for OpenAI clients\n- Output can be shrunk with `dimensions` from 128 to 2048 on three models, or with `int8`, `uint8`, `binary` and `ubinary` types\n- NGC images are signed, built for amd64 and arm64, and were last scanned on 5 October 2026 per the NGC registry record\n- The AI Enterprise lifecycle pages give support periods per branch and an end-of-life table with dates\n\n## Weaknesses\n\n- The API has no authentication and no rate limiting. The security page leaves both to a proxy the deployer runs\n- Production use needs an NVIDIA AI Enterprise licence, $4,500 a GPU a year or $1 a GPU-hour on cloud marketplaces, plus the GPU\n- Release notes carry no dates, and environment variable names changed between 2.0 and 2.3 without a note in the 2.2 or 2.3 notes we read\n- The NVIDIA Software Licence Agreement forbids disclosing benchmark results without written permission, apart from a published exception\n- Closed-source runtime with no public issue tracker or test suite, and no llms.txt entry or Markdown pages for these docs\n\n## Before you call it (notes for agents)\n\n1. Send `input_type` as `query` or `passage` on every embedding call. Asymmetric models return HTTP 400 without it, and the wrong value lowers retrieval accuracy per the docs\n2. Do not send `dimensions` and `embedding_type` together, and send only 2048 or nothing for `dimensions` on `nvidia/nemotron-3-embed-1b`\n3. Poll `/v1/health/ready` before the first call. The Docker health check can report unhealthy while the NIM is ready, per the 2.3 known issues\n4. Check the image tag on NGC before pulling. The guide uses `nemotron-3-embed-1b:2.3`, and NGC's record for that image listed tags up to 2.2.2 on 8 October 2026\n5. Put a proxy with authentication and TLS in front of port 8000, and sort `/v1/ranking` results yourself as the request has no top-n field\n\n## Connect\n\nInstall:\n\n```bash\necho \"$NGC_API_KEY\" | docker login nvcr.io --username '$oauthtoken' --password-stdin\ndocker run -it --rm --runtime=nvidia --gpus all --shm-size=16GB -e HF_TOKEN -v ~/.cache/nim/cache:/opt/cache -v ~/.cache/nim/weights:/model -u $(id -u) -p 8000:8000 nvcr.io/nim/nvidia/nemotron-3-embed-1b:2.3\n```\n\nFirst request:\n\n```bash\ncurl -X POST http://localhost:8000/v1/embeddings \\\n  -H 'accept: application/json' -H 'Content-Type: application/json' \\\n  -d '{\"input\":[\"What is NVIDIA?\"],\"model\":\"nvidia/nemotron-3-embed-1b\",\"input_type\":\"query\",\"modality\":\"text\",\"embedding_type\":\"float\",\"encoding_format\":\"float\"}'\n```\n\nThrough letme (picks today, calling later): https://letme.dev/nvidia-nemo-retriever. letme answers with the pick and how to call it direct; calling through letme (one key, the vendor's own price) comes later. How it works: https://www.anchorterminal.com/letme/index.md\n\n## Similar tools\n\nRanked by shared capabilities, then score. Same-category tools with no shared capability key are listed last.\n\n| Tool | Grade | Score | Rank | Shared capabilities | x402 | Markdown |\n| --- | --- | --- | --- | --- | --- | --- |\n| Cohere Embed and Rerank | BB | 72.5 | 100 | embed.text, embed.multimodal, embed.multilingual, rerank | no | https://www.anchorterminal.com/tools/cohere-embed.md |\n| Jina Embeddings and Reranker | C | 61 | 438 | embed.text, embed.multimodal, embed.multilingual, rerank | no | https://www.anchorterminal.com/tools/jina-embeddings.md |\n| Voyage AI embeddings and rerankers | C | 58.8 | 514 | embed.text, embed.multimodal, embed.multilingual, rerank | no | https://www.anchorterminal.com/tools/voyage-ai.md |\n| Gemini Embedding | BB | 70.6 | 143 | embed.text, embed.multimodal, embed.multilingual | no | https://www.anchorterminal.com/tools/gemini-embedding.md |\n| Nomic Embed | D | 49.2 | 711 | embed.text, embed.multimodal, embed.multilingual | no | https://www.anchorterminal.com/tools/nomic-embed.md |\n| ZeroEntropy zerank and zembed | F | 13.7 | 839 | rerank, embed.text, embed.multilingual | no | https://www.anchorterminal.com/tools/zeroentropy.md |\n\n## Panel reviews (0)\n\nReviewed by the Anchor panel (https://www.anchorterminal.com/reviewers/index.md): .\n\nDesk reviews, written from public documentation, pricing, terms, source and status history on 1 October 2026. No calls made. For a desk review, the outcome says whether the reviewer's questions could be answered from public material: success, partial or failure. How reviews work: https://www.anchorterminal.com/reviews/how-it-works.md\n\n## Notable\n\n- The Embedding NIM lists seven models, among them `nvidia/nemotron-3-embed-1b` (text, 2048 dimensions, 4,096 tokens), `nvidia/llama-nemotron-embed-vl-1b-v2` (text and image) and `baai/bge-m3` (source: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/support-matrix.html\u003e)\n- The Reranking NIM lists three models at 8,192 tokens, including `nvidia/llama-nemotron-rerank-vl-1b-v2`, which scores text, image or mixed passages against a text query (source: \u003chttps://docs.nvidia.com/nim/nemo-retriever/reranking/latest/support-matrix.html\u003e)\n- The security page says NIMs impose no rate limits and that the developer must secure the endpoints, suggesting a proxy and HTTPS with TLS 1.2 (source: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/latest/security.html\u003e)\n- Release 2.0 replaced the runtime, renamed four environment variables and removed fifteen, among them `NIM_TELEMETRY_MODE`, which the AI product terms still name as the telemetry switch (source: \u003chttps://docs.nvidia.com/nim/nemo-retriever/embedding/2.0/release-notes.html\u003e)\n- The NVIDIA Software Licence Agreement, version of 7 May 2026, bars disclosing benchmarking or performance results without written permission, except as described in a published benchmarking document (source: \u003chttps://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-software-license-agreement/\u003e)\n- NVIDIA's end-of-life notices mark the NV-EmbedQA-E5-v5 NIM as deprecated with action required by January 2027 (source: \u003chttps://docs.nvidia.com/ai-enterprise/lifecycle/latest/eol-notices.html\u003e)\n- The same models answer on NVIDIA's hosted trial API at `https://integrate.api.nvidia.com/v1` and `https://ai.api.nvidia.com/v1/retrieval/...`, under the NVIDIA API Trial Terms of Service (source: \u003chttps://build.nvidia.com/llms.txt\u003e)\n\n## Compare\n\n- [Amazon Nova Multimodal Embeddings vs NVIDIA NeMo Retriever Embedding and Reranking NIMs](https://www.anchorterminal.com/compare/amazon-nova-embeddings-vs-nvidia-nemo-retriever.md): BB 75 vs C 61\n- [Cohere Embed and Rerank vs NVIDIA NeMo Retriever Embedding and Reranking NIMs](https://www.anchorterminal.com/compare/cohere-embed-vs-nvidia-nemo-retriever.md): BB 72.5 vs C 61\n- [Gemini Embedding vs NVIDIA NeMo Retriever Embedding and Reranking NIMs](https://www.anchorterminal.com/compare/gemini-embedding-vs-nvidia-nemo-retriever.md): BB 70.6 vs C 61\n- [Jina Embeddings and Reranker vs NVIDIA NeMo Retriever Embedding and Reranking NIMs](https://www.anchorterminal.com/compare/jina-embeddings-vs-nvidia-nemo-retriever.md): C 61 vs C 61\n- [Mistral Embed and Codestral Embed vs NVIDIA NeMo Retriever Embedding and Reranking NIMs](https://www.anchorterminal.com/compare/mistral-embeddings-vs-nvidia-nemo-retriever.md): C 57.9 vs C 61\n- [Nomic Embed vs NVIDIA NeMo Retriever Embedding and Reranking NIMs](https://www.anchorterminal.com/compare/nomic-embed-vs-nvidia-nemo-retriever.md): D 49.2 vs C 61\n- [NVIDIA NeMo Retriever Embedding and Reranking NIMs vs OpenAI embeddings](https://www.anchorterminal.com/compare/nvidia-nemo-retriever-vs-openai-embeddings.md): C 61 vs BB 73.2\n- [NVIDIA NeMo Retriever Embedding and Reranking NIMs vs Voyage AI embeddings and rerankers](https://www.anchorterminal.com/compare/nvidia-nemo-retriever-vs-voyage-ai.md): C 61 vs C 58.8\n- [NVIDIA NeMo Retriever Embedding and Reranking NIMs vs ZeroEntropy zerank and zembed](https://www.anchorterminal.com/compare/nvidia-nemo-retriever-vs-zeroentropy.md): C 61 vs F 13.7\n\n## Verify this listing\n\nFor the vendor. The badge or a plain link to this page verifies the listing, from a page on nvidia.com or one of its subdomains. It shows the listing is the vendor's and that the vendor knows it's here, and it never changes a grade, rank or review. The vendor sends the page's address to `POST https://www.anchorterminal.com/api/v1/verify` as `{\"slug\": \"nvidia-nemo-retriever\", \"url\": \"…\"}`, or calls the `verify_listing` tool at https://www.anchorterminal.com/mcp. We fetch the page once, then again every week; two failed checks in a row and the verification lapses, and a later pass restores it. What we check: https://www.anchorterminal.com/builders/index.md#verify\n\nHTML badge:\n\n```html\n\u003ca href=\"https://www.anchorterminal.com/tools/nvidia-nemo-retriever\"\u003e\u003cimg src=\"https://www.anchorterminal.com/badges/nvidia-nemo-retriever.svg\" alt=\"NVIDIA NeMo Retriever Embedding and Reranking NIMs on Anchor Terminal\" height=\"20\"\u003e\u003c/a\u003e\n```\n\nMarkdown badge, for a README:\n\n```markdown\n[![NVIDIA NeMo Retriever Embedding and Reranking NIMs on Anchor Terminal](https://www.anchorterminal.com/badges/nvidia-nemo-retriever.svg)](https://www.anchorterminal.com/tools/nvidia-nemo-retriever)\n```\n\nPlain link:\n\n```html\n\u003ca href=\"https://www.anchorterminal.com/tools/nvidia-nemo-retriever\"\u003eNVIDIA NeMo Retriever Embedding and Reranking NIMs on Anchor Terminal\u003c/a\u003e\n```\n\n## Share this listing\n\nFor the vendor. Sharing assets for social media, two PNGs of 1200 × 630 that say NVIDIA NeMo Retriever Embedding and Reranking NIMs is listed on Anchor Terminal, with the vendor's logo and this page's address and no grade or score.\n\n- Dark: https://www.anchorterminal.com/assets/share/nvidia-nemo-retriever-dark.png\n- Light: https://www.anchorterminal.com/assets/share/nvidia-nemo-retriever-light.png\n",
  "meta": {
    "attribution": "Anchor Terminal (https://www.anchorterminal.com)",
    "docs": "https://www.anchorterminal.com/docs/",
    "generatedAt": "2026-10-09",
    "license": "CC-BY-4.0",
    "method": "https://www.anchorterminal.com/benchmark/",
    "methodology": "0.4",
    "openapi": "https://www.anchorterminal.com/openapi.json",
    "preview": false,
    "run": "2026-10-01",
    "runLabel": "October 2026 research run"
  },
  "page": {
    "breadcrumbs": [
      {
        "name": "Home",
        "url": "https://www.anchorterminal.com/"
      },
      {
        "name": "Terminal",
        "url": "https://www.anchorterminal.com/tools/"
      },
      {
        "name": "Embeddings \u0026 rerankers",
        "url": "https://www.anchorterminal.com/categories/embeddings"
      },
      {
        "name": "NVIDIA NeMo Retriever Embedding and Reranking NIMs",
        "url": ""
      }
    ],
    "description": "NVIDIA's NeMo Retriever Embedding and Reranking NIMs are GPU containers that run text and image embedding models and rerankers behind a local REST API, with /v1/embeddings in the OpenAI shape and /v1/ranking.",
    "facts": [
      "rank #439 of 842",
      "None auth",
      "0 desk reviews"
    ],
    "h1": "NVIDIA NeMo Retriever Embedding and Reranking NIMs",
    "image": "https://www.anchorterminal.com/assets/og/tools-nvidia-nemo-retriever.png",
    "path": "/tools/nvidia-nemo-retriever",
    "published": "2026-10-01",
    "section": "tools",
    "title": "NVIDIA NeMo Retriever Embedding and Reranking NIMs, grade C (61/100)",
    "toc": null,
    "updated": "2026-10-09",
    "url": "https://www.anchorterminal.com/tools/nvidia-nemo-retriever"
  },
  "tokens": {
    "markdown": 8450,
    "slim": 1980
  },
  "version": 1
}
