schema_version: 2
title: Free hosted model inference providers and routers
as_of: "2026-08-21"
checked_at: "2026-08-21T23:37:26-05:00"
timezone: America/Chicago

snapshot_counts:
  current_offer_records: 76
  headline_ongoing_or_recurring_records: 38
  conditional_delivery_current_records: 7
  finite_trial_or_promotion_records: 31
  conditional_or_needs_verification_records: 20
  historical_or_excluded_records: 37
  note: Current offers include trials and conditional delivery models; filter free_kind and headline_ongoing_free rather than treating 76 as a count of permanent free tiers.

scope:
  completeness: Best-effort exhaustive public-web audit, not a guarantee that every regional, private-beta, dashboard-only, or newly launched service has been found.
  primary_subject: Hosted foundation-model inference, routers, and serverless model-deployment platforms.
  modalities: [text_generation, embeddings, reranking, image, audio, speech, safety, custom_model_deployments]
  included:
    - Remotely callable hosted model inference with an API.
    - Ongoing free tiers, recurring credits, experimental access, and model-specific zero-price routes.
    - Time-limited trials and promotions when clearly labeled as such.
  excluded_from_current:
    - Open weights that require self-hosting.
    - Web chat without a remotely callable inference API.
    - A free gateway whose upstream model inference is still billed, unless it also has named zero-price models.
    - Offers found only in stale or third-party pages without current first-party confirmation.
  verification_levels:
    live_catalog: An official models endpoint was queried without credentials; this does not prove an authenticated generation succeeded.
    live_call: A credential-free inference call succeeded; this still does not prove authenticated-account behavior or billing.
    official_current: Current first-party pricing, documentation, or product pages explicitly support the claim.
    dashboard_only: The provider says the limit or catalog is visible only after sign-in.
    conflicting: Current first-party sources disagree; both claims are retained.
  authenticated_generation_policy: No provider credential was created and no authenticated generation was billed for this audit. A live_catalog label validates published server metadata, not end-to-end callability.
  important_note: Free catalogs and quotas are volatile. Re-fetch each discovery URL before relying on this snapshot.

website_filtering:
  default_current_list: Read records only from providers.
  headline_ongoing_free:
    omitted_or_true: Normal ongoing, recurring, restricted-research, or active experimental offer.
    conditional: Show with a community-capacity, user-pays, data-use, or sustainability warning.
    false: One-time trial, application credit, or limited promotion; never label as a permanent free tier.
  candidate_list: Read conditional_or_needs_verification separately and never merge it into verified counts.
  unavailable_list: historical_or_excluded contains retired, paid-only, non-API, nonfunctional, and self-hosted-only results.
  recommended_badges: [free_kind, status, confidence, payment_method_required, geography, modality, verification]

discovery_coverage:
  interpretation: Ecosystem registries are candidate generators, not proof that each upstream provider has a direct free API.
  openrouter_provider_registry:
    checked_at: "2026-08-21"
    discovery_url: https://openrouter.ai/api/v1/providers
    live_provider_count: 104
    caveat: The live API includes a synthetic-looking fake-provider entry; it is retained for exact snapshot fidelity and is not treated as a real provider candidate.
    provider_slugs:
      - ai21
      - aion-labs
      - akashml
      - alibaba
      - amazon-bedrock
      - amazon-nova
      - ambient
      - anthropic
      - arcee-ai
      - atlas-cloud
      - avian
      - azure
      - baidu
      - baseten
      - black-forest-labs
      - cerebras
      - chutes
      - cirrascale
      - clarifai
      - claude-on-aws
      - cloudflare
      - cohere
      - coreweave
      - crucible
      - crusoe
      - darkbloom
      - databricks
      - decart
      - deepgram
      - deepinfra
      - deepseek
      - dekallm
      - digitalocean
      - fake-provider
      - featherless
      - fireworks
      - fish-audio
      - friendli
      - gmicloud
      - google-ai-studio
      - google-vertex
      - groq
      - heygen
      - inception
      - inceptron
      - inferact-vllm
      - inference-net
      - infermatic
      - inflection
      - io-net
      - ionstream
      - krea
      - liquid
      - makora
      - mancer
      - mara
      - meta
      - minimax
      - mistral
      - modal
      - modelrun
      - modular
      - moonshotai
      - morph
      - ncompass
      - nebius
      - nex-agi
      - nextbit
      - novita
      - nvidia
      - open-inference
      - openai
      - parasail
      - perceptron
      - perplexity
      - phala
      - poolside
      - quiver
      - recraft
      - reka
      - relace
      - runpod
      - runway
      - sail-research
      - sakana
      - sambanova
      - seed
      - siliconflow
      - sourceful
      - stealth
      - stepfun
      - streamlake
      - switchpoint
      - tencent
      - tenstorrent
      - thinkingmachines
      - together
      - upstage
      - venice
      - voyageai
      - wafer
      - xai
      - xiaomi
      - z-ai
  huggingface_inference_provider_registry:
    checked_at: "2026-08-21"
    discovery_url: https://huggingface.co/docs/inference-providers/main/en/index
    documented_integration_count: 19
    provider_ids: [cerebras, cohere, deepinfra, fal_ai, featherless_ai, fireworks, groq, hf_inference, hyperbolic, novita, nscale, ovhcloud, public_ai, replicate, sambanova, scaleway, together, wavespeedai, zai]
  other_systematic_sources:
    - NVIDIA Build/NIM catalog, developer terms, model cards, and partner announcements.
    - Vercel AI Gateway, OrcaRouter, Kilo, FastRouter, OpenCode Zen, and community-router catalogs.
    - Official pricing, billing, rate-limit, changelog, status, GitHub, press-release, X/Twitter, and news searches for every named candidate.
  coverage_note: Every positive record below has provider-specific evidence. High-probability negative results are retained under historical_or_excluded to prevent rediscovery from stale announcements.

free_kinds:
  ongoing_free: Recurring or indefinite no-cost API usage under a quota.
  rotating_zero_price: Named models cost zero while they remain in a changing catalog.
  recurring_credit: A monetary allowance that refreshes on a stated schedule.
  experimental: Free only while a preview or experiment remains active.
  development_only: Free for prototyping or evaluation, not production use.
  promotional: A temporary model promotion or expiring allowance.
  trial: A one-time new-account credit or time window.
  community_capacity: No-cost inference supplied by volunteer or community-operated capacity with no guaranteed throughput.
  user_pays: The developer pays nothing because each end user authenticates and consumes that user's own allowance.

providers:
  - id: openrouter
    name: OpenRouter
    website: https://openrouter.ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: rotating_zero_price
    api:
      base_url: https://openrouter.ai/api/v1
      compatibility: [openai_chat_completions, openai_completions, openai_models]
      authentication: Bearer API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      requests_per_minute: 20
      requests_per_day_without_purchase: 50
      requests_per_day_after_10_usd_credit_purchase: 1000
      scope: Total across free-model requests; failed requests count.
    free_router:
      model_id: openrouter/free
      price_usd: 0
      behavior: Randomly selects a compatible model from the current free pool.
    models:
      snapshot_at: "2026-08-21T21:13:22-05:00"
      verification: live_catalog
      discovery_url: https://openrouter.ai/api/v1/models
      count_including_router: 22
      free_model_ids:
        - stealth/ox-alpha
        - dots-studio/dots-3-note-preview:free
        - liquid/lfm-2.5-2.6b:free
        - nvidia/nemotron-3.5-lightning:free
        - thinkingmachines/inkling-small:free
        - poolside/laguna-s-2.1:free
        - thinkingmachines/inkling:free
        - poolside/laguna-xs-2.1:free
        - cohere/north-mini-code:free
        - z-ai/glm-5.2:free
        - nvidia/nemotron-3.5-content-safety:free
        - nvidia/nemotron-3-ultra-550b-a55b:free
        - nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free
        - google/gemma-4-26b-a4b-it:free
        - google/gemma-4-31b-it:free
        - google/lyria-3-pro-preview
        - google/lyria-3-clip-preview
        - nvidia/nemotron-3-super-120b-a12b:free
        - openrouter/free
        - nvidia/nemotron-3-nano-30b-a3b:free
        - nvidia/nemotron-nano-12b-v2-vl:free
        - nvidia/nemotron-nano-9b-v2:free
      expiring_models:
        dots-studio/dots-3-note-preview:free: "2026-09-30"
        nvidia/nemotron-3-nano-30b-a3b:free: "2026-08-24"
        nvidia/nemotron-nano-12b-v2-vl:free: "2026-08-24"
        nvidia/nemotron-nano-9b-v2:free: "2026-08-24"
      caveats:
        - Lyria preview entries are zero-priced despite lacking a :free suffix.
        - Marketing counts and the live API count differ; this snapshot uses zero prompt and completion prices from the API.
    history:
      - {date: "2025-07-10", event: "OpenRouter described how it would sustain its free tier after two upstream providers moved paid-only."}
      - {date: "2026-02-23", event: "The openrouter/free router was announced."}
    sources:
      - {type: official_docs, title: OpenRouter FAQ, url: https://openrouter.ai/docs/faq}
      - {type: live_catalog, title: Models endpoint, url: https://openrouter.ai/api/v1/models}
      - {type: official_docs, title: Free Models Router, url: https://openrouter.ai/docs/guides/routing/routers/free-router}
      - {type: announcement, title: Updates to Our Free Tier, url: https://openrouter.ai/blog/announcements/updates-to-our-free-tier-sustaining-accessible-ai-for-everyone/, published_at: "2025-07-10"}

  - id: nous_portal
    name: Nous Research / Nous Portal
    website: https://nousresearch.com/
    portal_url: https://portal.nousresearch.com/
    status: current
    qualifies: true
    confidence: high
    free_kind: rotating_zero_price
    requested_name_note: User-confirmed resolution of “NUS Research” is Nous Research at https://nousresearch.com/.
    api:
      base_url: https://inference-api.nousresearch.com/v1
      compatibility: [openai_chat_completions, openai_models]
      authentication: Nous Portal OAuth or bearer credential
    eligibility:
      account_required: true
      plan: Free — $0/month
      payment_method_required: not_explicitly_documented
    limits:
      observed_portal_requests_per_minute: 50
      observed_portal_tokens_per_minute: 500000
      caveat: One current portal surface exposes these numbers while another says only “Standard rate limits”; account limits remain authoritative.
    models:
      snapshot_at: "2026-08-21T21:13:22-05:00"
      verification: live_catalog
      discovery_url: https://inference-api.nousresearch.com/v1/models
      count: 7
      free_model_ids:
        - stealth/ox-alpha
        - poolside/laguna-s-2.1:free
        - poolside/laguna-xs-2.1:free
        - tencent/hy3:free
        - stepfun/step-3.7-flash:free
        - upstage/solar-pro4:free
        - meituan/longcat-2.0:free
    history:
      - date: "2025-03-12"
        event: Initial inference API launch with a waitlist and $5 signup credit.
        initial_models: [Hermes 3 Llama 70B, DeepHermes 3 8B Preview]
        current_status: superseded_by_zero_price_catalog
    sources:
      - {type: official_website, title: Nous Research, url: https://nousresearch.com/}
      - {type: official_product, title: Nous Portal, url: https://portal.nousresearch.com/}
      - {type: live_catalog, title: Models endpoint, url: https://inference-api.nousresearch.com/v1/models}
      - {type: announcement, title: Announcing the Nous Portal, url: https://forum.nousresearch.com/t/announcing-the-nous-portal/62, published_at: "2025-03-12"}
      - {type: official_repository_docs, title: Hermes Agent Nous Portal integration, url: https://github.com/NousResearch/hermes-agent/blob/main/website/docs/integrations/nous-portal.md}

  - id: orcarouter
    name: OrcaRouter
    operator: Continuum AI Corp.
    website: https://www.orcarouter.ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: rotating_zero_price
    api:
      base_url: https://api.orcarouter.ai/v1
      compatibility: [openai_chat_completions, openai_responses, anthropic_style, gemini_style]
      authentication: Bearer API key
    eligibility:
      account_required: true
      payment_method_required: false
      free_models_may_require_claim: true
    limits:
      published_general_numeric_limits: false
      exhausted_behavior: HTTP 429 or free-quota-exhausted; the wallet is not charged by the free router.
      recent_announcement_snapshot:
        requests_per_minute: 10
        requests_per_day: 50
        after_20_usd_cumulative_workspace_payments: {requests_per_minute: 20, requests_per_day: 1000}
      caveat: The launch announcement, localized offer page, and live catalog differed during the 2026-08-21 audit; prefer the signed-in offer/account state.
    free_router:
      model_id: orcarouter/free
      documented_routes:
        - {model: deepseek/deepseek-v4-flash, advertised_free_calls: 50}
        - {model: deepseek/deepseek-v4-pro, advertised_free_calls: 10}
      reset_period: unpublished
    models:
      snapshot_at: "2026-08-21T21:13:22-05:00"
      verification: live_catalog
      discovery_url: https://api.orcarouter.ai/v1/models
      free_models:
        - {id: deepseek/deepseek-v4-flash-free, context_tokens: 1048576, max_output_tokens: 384000}
        - {id: deepseek/deepseek-v4-pro-free, context_tokens: 1048576, max_output_tokens: 384000}
        - id: qwen/qwen3.8-27b-free
          context_tokens_live_api: 65536
          context_tokens_marketing_page: 262144
          conflict: Use the conservative live-API value until an authenticated call resolves the discrepancy.
        - id: tencent/hy3-free
          context_tokens: 262144
          callability: not_fully_proven
          caveat: The live entry has no supported endpoint type.
    important_distinction: The $0 Hacker gateway plan does not make paid upstream inference free; only the aliases and router above are zero-cost.
    sources:
      - {type: official_product, title: OrcaRouter offers, url: https://www.orcarouter.ai/offers}
      - {type: official_product, title: Free router, url: https://www.orcarouter.ai/models/orcarouter/free}
      - {type: live_catalog, title: Models endpoint, url: https://api.orcarouter.ai/v1/models}
      - {type: official_pricing, title: Pricing, url: https://www.orcarouter.ai/pricing}
      - {type: company_press_release, title: Hosted OrcaRouter launch, url: https://www.prnewswire.com/news-releases/orcarouter-launches-the-open-llm-api-router--zero-markup-mit-licensed-100-models-302766356.html, published_at: "2026-05"}

  - id: nvidia_build
    name: NVIDIA Build / NVIDIA-hosted NIM
    website: https://build.nvidia.com/
    status: current
    qualifies: true
    confidence: high
    free_kind: development_only
    production_use_allowed: false
    api:
      base_url: https://integrate.api.nvidia.com/v1
      compatibility: LLM NIMs are OpenAI-compatible; non-LLM NIMs can use task-specific APIs.
      authentication: NVIDIA Developer API key
    eligibility:
      account_required: true
      developer_program_membership_required: true
      membership_cost_usd: 0
      payment_method_required: not_documented
    limits:
      public_numeric_limits: false
      official_rule: Varies by model and concurrency; inspect the signed-in account.
      current_marketing_phrase: Unlimited prototyping
      allowed_use: [prototyping, research, development, testing, learning]
    models:
      snapshot_at: "2026-08-21T21:13:22-05:00"
      verification: live_catalog
      discovery_url: https://integrate.api.nvidia.com/v1/models
      count: 102
      caveat: The catalog includes chat, vision, embedding, safety, detection, parsing, and translation models; not every ID accepts chat completions.
      model_ids:
        - 01-ai/yi-large
        - adept/fuyu-8b
        - ai21labs/jamba-1.5-large-instruct
        - aisingapore/sea-lion-7b-instruct
        - baai/bge-m3
        - bigcode/starcoder2-15b
        - databricks/dbrx-instruct
        - deepseek-ai/deepseek-coder-6.7b-instruct
        - deepseek-ai/deepseek-v4-flash-0731
        - google/codegemma-1.1-7b
        - google/codegemma-7b
        - google/deplot
        - google/diffusiongemma-26b-a4b-it
        - google/gemma-2b
        - google/gemma-3-12b-it
        - google/gemma-3-4b-it
        - google/gemma-4-31b-it
        - google/recurrentgemma-2b
        - ibm/granite-3.0-3b-a800m-instruct
        - ibm/granite-3.0-8b-instruct
        - ibm/granite-34b-code-instruct
        - ibm/granite-8b-code-instruct
        - meta/codellama-70b
        - meta/llama-3.1-70b-instruct
        - meta/llama-3.1-8b-instruct
        - meta/llama-3.2-11b-vision-instruct
        - meta/llama-3.2-1b-instruct
        - meta/llama-3.2-3b-instruct
        - meta/llama-3.2-90b-vision-instruct
        - meta/llama-3.3-70b-instruct
        - meta/llama-guard-4-12b
        - meta/llama2-70b
        - meta/muse-glimmer-30b
        - microsoft/kosmos-2
        - microsoft/phi-3-vision-128k-instruct
        - microsoft/phi-3.5-moe-instruct
        - minimaxai/minimax-m3
        - mistralai/codestral-22b-instruct-v0.1
        - mistralai/mistral-7b-instruct-v0.3
        - mistralai/mistral-large
        - mistralai/mistral-large-2-instruct
        - mistralai/mistral-nemotron
        - mistralai/mixtral-8x22b-v0.1
        - moonshotai/kimi-k2.6
        - moonshotai/kimi-k3
        - nv-mistralai/mistral-nemo-12b-instruct
        - nvidia/ai-synthetic-video-detector
        - nvidia/cosmos-reason2-8b
        - nvidia/embed-qa-4
        - nvidia/ising-calibration-1.5-31b
        - nvidia/llama-3.1-nemoguard-8b-content-safety
        - nvidia/llama-3.1-nemoguard-8b-topic-control
        - nvidia/llama-3.1-nemotron-51b-instruct
        - nvidia/llama-3.1-nemotron-70b-instruct
        - nvidia/llama-3.1-nemotron-nano-8b-v1
        - nvidia/llama-3.1-nemotron-nano-vl-8b-v1
        - nvidia/llama-3.1-nemotron-safety-guard-8b-v3
        - nvidia/llama-3.1-nemotron-ultra-253b-v1
        - nvidia/llama-3.2-nemoretriever-1b-vlm-embed-v1
        - nvidia/llama-3.2-nv-embedqa-1b-v1
        - nvidia/llama-3.3-nemotron-super-49b-v1
        - nvidia/llama-3.3-nemotron-super-49b-v1.5
        - nvidia/llama-nemotron-embed-1b-v2
        - nvidia/llama-nemotron-embed-vl-1b-v2
        - nvidia/llama3-chatqa-1.5-70b
        - nvidia/mistral-nemo-minitron-8b-8k-instruct
        - nvidia/nemoretriever-parse
        - nvidia/nemotron-3-embed-1b
        - nvidia/nemotron-3-nano-30b-a3b
        - nvidia/nemotron-3-nano-omni-30b-a3b-reasoning
        - nvidia/nemotron-3-super-120b-a12b
        - nvidia/nemotron-3-ultra-550b-a55b
        - nvidia/nemotron-3.5-content-safety
        - nvidia/nemotron-3.5-lightning-30b-a3b
        - nvidia/nemotron-4-340b-instruct
        - nvidia/nemotron-4-340b-reward
        - nvidia/nemotron-mini-4b-instruct
        - nvidia/nemotron-nano-12b-v2-vl
        - nvidia/nemotron-nano-3-30b-a3b
        - nvidia/nemotron-parse
        - nvidia/neva-22b
        - nvidia/nv-embed-v1
        - nvidia/nv-embedcode-7b-v1
        - nvidia/nv-embedqa-e5-v5
        - nvidia/nv-embedqa-mistral-7b-v2
        - nvidia/nvclip
        - nvidia/nvidia-nemotron-nano-9b-v2
        - nvidia/riva-translate-4b-instruct
        - nvidia/riva-translate-4b-instruct-v1.1
        - nvidia/riva-translate-4b-instruct-v2
        - nvidia/vila
        - openai/gpt-oss-120b
        - openai/gpt-oss-20b
        - poolside/laguna-xs-2.1
        - snowflake/arctic-embed-l
        - stepfun-ai/step-3.7-flash
        - thinkingmachines/inkling
        - writer/palmyra-creative-122b
        - writer/palmyra-fin-70b-32k
        - writer/palmyra-med-70b
        - writer/palmyra-med-70b-32k
        - zyphra/zamba2-7b-instruct
    history:
      - {date: "2024-07-29", event: "Free NIM access for Developer Program members announced."}
      - {date: "2024-09-04", event: "Older scheme documented 1,000 initial credits and up to 5,000; current 2026 docs instead describe prototyping access.", current_status: superseded_or_unconfirmed}
    sources:
      - {type: official_product, title: NVIDIA NIM for Developers, url: https://developer.nvidia.com/nim}
      - {type: official_docs, title: Run NIM Anywhere, url: https://docs.api.nvidia.com/nim/re/docs/run-anywhere}
      - {type: live_catalog, title: Models endpoint, url: https://integrate.api.nvidia.com/v1/models}
      - {type: announcement, title: Free NIM access announcement, url: "https://developer.nvidia.com/blog/?p=86238", published_at: "2024-07-29"}

  - id: hetzner_experiments
    name: Hetzner Experiments Inference API
    website: https://experiments.hetzner.com/
    status: current
    qualifies: true
    confidence: high
    free_kind: experimental
    production_use_allowed: false
    api:
      base_url: https://inference.hetzner.com/api/v1
      compatibility: [openai_models, openai_completions, openai_chat_completions]
      authentication: Hetzner API token
    eligibility:
      account_required: true
      hetzner_customer_required: true
      payment_for_inference_required: false
    limits:
      window_seconds: 60
      requests: 10
      input_tokens: 4000000
      output_tokens: 100000
      scope: per_api_key
    models:
      snapshot_at: "2026-08-21T21:13:22-05:00"
      verification: official_current
      discovery_url: https://inference.hetzner.com/api/v1/models
      free_models:
        - {id: Qwen/Qwen3.6-35B-A3B-FP8, context_tokens: 262144, modalities: [text, image]}
        - {id: Qwen3.8-27B, context_tokens: 262144, modalities: [text, image]}
    duration: Free while experimental; Hetzner promises advance email notice if that changes.
    service_level: Best effort with no availability guarantee.
    data_handling: Usage metadata is retained, but prompt and response content is not retained absent legal compulsion.
    sources:
      - {type: official_docs, title: Inference API, url: https://docs.hetzner.com/general/company-and-policy/experiments/inference/, published_at: "2026-07-24"}
      - {type: official_docs, title: Experiments Platform, url: https://docs.hetzner.com/general/company-and-policy/experiments/experiments-platform/, published_at: "2026-07-24"}
      - {type: official_tutorial, title: OpenCode with Hetzner Inference API, url: https://community.hetzner.com/tutorials/opencode-with-hetzner-inference-api-systemd-sandbox/, published_at: "2026-07-15"}

  - id: groqcloud
    name: GroqCloud
    website: https://console.groq.com/
    status: current
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    api:
      base_url: https://api.groq.com/openai/v1
      compatibility: [openai_chat_completions, openai_responses, audio_transcriptions, audio_translations]
      authentication: Bearer API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      scope: organization
      caveat: The signed-in organization Limits page is authoritative if it differs from this public-table snapshot.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://console.groq.com/docs/models
      free_models_and_limits:
        - {id: canopylabs/orpheus-arabic-saudi, rpm: 10, rpd: 100, tpm: 1200, tpd: 3600}
        - {id: canopylabs/orpheus-v1-english, rpm: 10, rpd: 100, tpm: 1200, tpd: 3600}
        - {id: groq/compound, rpm: 30, rpd: 250, tpm: 70000}
        - {id: groq/compound-mini, rpm: 30, rpd: 250, tpm: 70000}
        - {id: meta-llama/llama-prompt-guard-2-22m, rpm: 30, rpd: 14400, tpm: 15000, tpd: 500000}
        - {id: meta-llama/llama-prompt-guard-2-86m, rpm: 30, rpd: 14400, tpm: 15000, tpd: 500000}
        - {id: openai/gpt-oss-120b, rpm: 30, rpd: 1000, tpm: 8000, tpd: 200000}
        - {id: openai/gpt-oss-20b, rpm: 30, rpd: 1000, tpm: 8000, tpd: 200000}
        - {id: openai/gpt-oss-safeguard-20b, rpm: 30, rpd: 1000, tpm: 8000, tpd: 200000}
        - {id: qwen/qwen3.6-27b, rpm: 30, rpd: 1000, tpm: 8000, tpd: 200000}
        - {id: whisper-large-v3, rpm: 20, rpd: 2000, audio_seconds_per_hour: 7200, audio_seconds_per_day: 28800}
        - {id: whisper-large-v3-turbo, rpm: 20, rpd: 2000, audio_seconds_per_hour: 7200, audio_seconds_per_day: 28800}
    history:
      - {date: "2024-03-01", event: GroqCloud launch}
      - {date: "2024-04-02", event: Early demand and free-access announcement}
    sources:
      - {type: official_docs, title: Rate limits, url: https://console.groq.com/docs/rate-limits}
      - {type: official_docs, title: Billing FAQ, url: https://console.groq.com/docs/billing-faqs}
      - {type: official_docs, title: Models, url: https://console.groq.com/docs/models}
      - {type: announcement, title: GroqCloud demand announcement, url: https://groq.com/newsroom/demand-for-real-time-ai-inference-from-groq-accelerates-week-over-week, published_at: "2024-04-02"}

  - id: google_gemini_api
    name: Google AI Studio / Gemini Developer API
    website: https://ai.google.dev/
    status: current
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    api:
      native_base_url: https://generativelanguage.googleapis.com
      openai_compatible_base_url: https://generativelanguage.googleapis.com/v1beta/openai/
      authentication: Google API key
    eligibility:
      account_required: true
      payment_method_required: false
      geographic_availability_applies: true
    privacy_caveat: Free-tier content may be used to improve Google products; paid-tier content is not used that way under the published terms.
    limits:
      dimensions: [requests_per_minute, tokens_per_minute, requests_per_day]
      public_fixed_numbers: null
      source_of_truth: AI Studio project rate-limit dashboard
      reset: Daily request quotas reset at midnight Pacific time.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://ai.google.dev/gemini-api/docs
      free_model_ids:
        - gemini-3.7-flash
        - gemini-3.6-flash
        - gemini-3.5-flash
        - gemini-3.5-flash-lite
        - gemini-3.1-flash-lite
        - gemini-2.5-flash
        - gemini-2.5-flash-lite
        - gemini-3.5-live-translate-preview
        - gemini-3.1-flash-live-preview
        - gemini-3.1-flash-tts-preview
        - gemini-2.5-flash-native-audio-preview-12-2025
      explicit_exclusions:
        - {id: gemini-3.1-pro-preview, reason: No free-token tier}
        - {id: gemini-3.1-flash-image, reason: Image generation is paid}
        - {id: gemini-2.0-flash, reason: Shut down on 2026-06-01 despite stale pricing rows}
        - {id: gemini-2.0-flash-lite, reason: Shut down on 2026-06-01 despite stale pricing rows}
    sources:
      - {type: official_pricing, title: Gemini API pricing, url: https://ai.google.dev/gemini-api/docs/pricing}
      - {type: official_docs, title: Rate limits, url: https://ai.google.dev/gemini-api/docs/rate-limits}
      - {type: official_docs, title: Deprecations, url: https://ai.google.dev/gemini-api/docs/deprecations}
      - {type: official_changelog, title: Gemini API changelog, url: https://ai.google.dev/gemini-api/docs/changelog}

  - id: mistral
    name: Mistral Studio / La Plateforme
    website: https://mistral.ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: recurring_credit
    api:
      base_url: https://api.mistral.ai/v1
      compatibility: openai_style_rest
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
      intended_use: evaluation_and_prototyping
    limits:
      monthly_api_credit_usd: 10
      dimensions: [requests_per_second, tokens_per_minute, tokens_per_month]
      numeric_limits: dashboard_only
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://docs.mistral.ai/models
      always_zero_priced_model_ids: [mistral-moderation-2603, labs-leanstral-1-5]
      representative_ids_eligible_for_monthly_credit:
        - mistral-small-2603
        - mistral-medium-3-5
        - mistral-large-2512
        - ministral-14b-2512
        - ministral-8b-2512
        - ministral-3b-2512
      caveat: The $10 allowance is monetary, so usable token volume depends on model prices.
    history:
      - {date: "2023-12-11", event: La Plateforme announcement}
      - {date: "2024-09-17", event: Free API tier announcement}
    sources:
      - {type: official_pricing, title: Mistral pricing, url: https://mistral.ai/pricing}
      - {type: official_docs, title: Inference pricing, url: https://docs.mistral.ai/inference/pricing}
      - {type: official_docs, title: Model catalog, url: https://docs.mistral.ai/models}
      - {type: announcement, title: September 2024 release, url: https://mistral.ai/fr/news/september-24-release/, published_at: "2024-09-17"}

  - id: huggingface_inference_providers
    name: Hugging Face Inference Providers
    website: https://huggingface.co/inference/models
    status: current
    qualifies: true
    confidence: high
    free_kind: recurring_credit
    api:
      base_url: https://router.huggingface.co/v1
      compatibility: [openai_chat_completions, openai_responses, openai_models, huggingface_sdk]
      authentication: Hugging Face access token
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      monthly_routed_credit_usd: 0.10
      rate_limits: Provider and model specific.
      overage: Purchased credits required after the monthly allowance.
      caveat: BYOK calls do not consume Hugging Face monthly credits.
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      coverage: Dynamic routed catalog of more than 200 models; the credit applies to eligible routed models.
      discovery_url: https://router.huggingface.co/v1/models
      human_catalog_url: https://huggingface.co/inference/models
    history:
      - {date: "2025-01-28", event: Inference Providers launched}
    sources:
      - {type: official_docs, title: Pricing, url: https://huggingface.co/docs/inference-providers/main/en/pricing}
      - {type: official_docs, title: Inference Providers overview, url: https://huggingface.co/docs/inference-providers/en/index}
      - {type: announcement, title: Inference Providers launch, url: https://huggingface.co/blog/inference-providers, published_at: "2025-01-28"}

  - id: cloudflare_workers_ai
    name: Cloudflare Workers AI
    website: https://developers.cloudflare.com/workers-ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    api:
      native_pattern: https://api.cloudflare.com/client/v4/accounts/{account_id}/ai/run/{model}
      compatibility: [openai_chat_completions, openai_responses, openai_embeddings]
      authentication: Cloudflare API token
    eligibility:
      account_required: true
      payment_method_required: false
      model_license_acceptance_may_be_required: true
    limits:
      neurons_per_day: 10000
      reset: "00:00 UTC"
      free_plan_exhausted_behavior: Requests fail until reset.
      paid_plan_overage: $0.011 per 1,000 neurons.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: All current Workers AI catalog entries except those marked as requiring Workers Paid.
      current_catalog_count: 84
      discovery_url: https://developers.cloudflare.com/workers-ai/models/
      representative_free_model_ids:
        - "@cf/google/gemma-4-26b-a4b-it"
        - "@cf/zai-org/glm-4.7-flash"
        - "@cf/nvidia/nemotron-3-120b-a12b"
        - "@cf/openai/gpt-oss-120b"
        - "@cf/meta/llama-3.1-8b-instruct"
        - "@cf/deepseek-ai/deepseek-r1-distill-qwen-32b"
        - "@cf/mistralai/mistral-small-3.1-24b-instruct"
      paid_only_model_ids:
        - "@cf/moonshotai/kimi-k2.6"
        - "@cf/moonshotai/kimi-k2.7-code"
        - "@cf/zai-org/glm-5.2"
        - "@cf/deepseek-ai/deepseek-v4-flash-0731"
        - "@cf/deepseek-ai/deepseek-v4-pro-0813"
    history:
      - {date: "2024-04", event: General availability with a daily free allocation}
      - {date: "2026-07-28", event: Cloudflare began marking selected premium models as Workers Paid-only}
    sources:
      - {type: official_pricing, title: Workers AI pricing, url: https://developers.cloudflare.com/workers-ai/platform/pricing/}
      - {type: official_catalog, title: Workers AI models, url: https://developers.cloudflare.com/workers-ai/models/}
      - {type: official_changelog, title: Models that require Workers Paid, url: https://developers.cloudflare.com/changelog/post/2026-07-28-models-require-workers-paid/, published_at: "2026-07-28"}
      - {type: announcement, title: Workers AI general availability, url: https://blog.cloudflare.com/workers-ai-ga-huggingface-loras-python-support/, published_at: "2024-04"}

  - id: cohere
    name: Cohere
    website: https://cohere.com/
    status: current_restricted
    qualifies: true
    confidence: high
    free_kind: development_only
    production_use_allowed: false
    commercial_use_allowed: false
    api:
      base_url: https://api.cohere.ai/v2
      compatibility: cohere_v2_rest
      authentication: Trial API key
    eligibility:
      account_required: true
      payment_method_required: false
      allowed_use: evaluation_and_prototyping
    limits:
      calls_per_month: 1000
      chat_rpm: 20
      rerank_rpm: 10
      embed_inputs_per_minute: 2000
      embed_image_inputs_per_minute: 5
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Trial keys can evaluate the current Cohere catalog under endpoint limits.
      current_chat_families:
        - Command A+
        - Command A Reasoning
        - Command A Translate
        - Command A Vision
        - Command A
        - Command R+
        - Command R
        - Command R7B
        - North Mini Code
      exact_verified_id: command-a-plus-05-2026
    sources:
      - {type: official_pricing, title: Cohere pricing, url: https://cohere.com/pricing}
      - {type: official_docs, title: Rate limits, url: https://docs.cohere.com/v2/docs/rate-limits}
      - {type: official_docs, title: Chat API, url: https://docs.cohere.com/v2/docs/chat-api}
      - {type: official_changelog, title: Trial key pricing update, url: https://docs.cohere.com/v1/changelog/pricing-update-and-new-dashboard-ui, published_at: "2022-10-18"}

  - id: vercel_ai_gateway
    name: Vercel AI Gateway
    website: https://vercel.com/ai-gateway
    status: current
    qualifies: true
    confidence: high
    free_kind: recurring_credit
    api:
      base_url: https://ai-gateway.vercel.sh/v1
      compatibility: [openai_chat_completions, openai_responses, anthropic_and_provider_sdks]
      authentication: Vercel AI Gateway key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      included_credit_usd_per_month: 5
      important_transition: After a team purchases credits it moves to paid status and no longer receives the monthly free credit.
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      coverage: The $5 monthly credit can pay for any available catalog model until exhausted.
      discovery_url: https://ai-gateway.vercel.sh/v1/models
      native_zero_price_model_ids:
        - poolside/laguna-s-2.1-free
      caveat: The native $0 model is volatile; query the discovery URL for current prices.
    history:
      - {date: "2025-08-21", event: AI Gateway became generally available}
    sources:
      - {type: official_pricing, title: AI Gateway pricing, url: https://vercel.com/docs/ai-gateway/pricing}
      - {type: live_catalog, title: Models API, url: https://ai-gateway.vercel.sh/v1/models}
      - {type: official_catalog, title: AI Gateway models, url: https://vercel.com/ai-gateway/models}
      - {type: announcement, title: AI Gateway general availability, url: https://vercel.com/changelog/ai-gateway-is-now-generally-available, published_at: "2025-08-21"}

  - id: ibm_watsonx_ai_runtime
    name: IBM watsonx.ai Runtime
    website: https://www.ibm.com/products/watsonx-ai
    status: current
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    api:
      base_url_pattern: https://{region}.ml.cloud.ibm.com/ml/v1
      compatibility: ibm_watsonx_rest_and_sdks
      authentication: IBM Cloud IAM
    eligibility:
      account_required: true
      payment_method_required_for_new_accounts: true
      caveat: New IBM Cloud accounts require a card for identity verification even when using Lite services.
    limits:
      foundation_tokens_per_month: 300000
      requests_per_second: 2
      capacity_unit_hours: 20
      extraction_pages: 100
      plan: Lite
      inactivity: A Lite service can be deleted after 30 days of inactivity.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Region-available foundation models eligible under the Lite token quota.
      discovery_url_pattern: GET https://{region}.ml.cloud.ibm.com/ml/v1/foundation_model_specs
      representative_current_model_ids:
        - granite-4-h-small
        - granite-4-h-tiny
        - granite-4-h-micro
        - granite-3-1-8b-base
      caveat: Model availability is regional and dynamic; use the API rather than older static catalog pages.
    history:
      - {date: "2023-05-09", event: IBM announced the watsonx platform}
    sources:
      - {type: official_docs, title: watsonx.ai Runtime plans, url: "https://www.ibm.com/docs/en/watsonx/saas?topic=cloud-watsonxai-runtime-plans"}
      - {type: official_catalog, title: watsonx.ai Runtime service, url: https://cloud.ibm.com/catalog/services/pm-20}
      - {type: official_docs, title: List foundation models programmatically, url: "https://dataplatform.cloud.ibm.com/docs/content/wsj/analyze-data/fm-prompt-notebook-list-models.html?context=wx"}
      - {type: announcement, title: IBM unveils watsonx, url: https://newsroom.ibm.com/2023-05-09-IBM-Unveils-the-Watsonx-Platform-to-Power-Next-Generation-Foundation-Models-for-Business, published_at: "2023-05-09"}

  - id: opencode_zen
    name: OpenCode Zen
    website: https://opencode.ai/zen
    status: current_promotional
    qualifies: true
    headline_ongoing_free: false
    confidence: medium_high
    free_kind: promotional
    api:
      base_url: https://opencode.ai/zen/v1
      compatibility: openai_chat_completions
      authentication: Zen API key; account and billing setup details may vary.
    limits:
      published_numeric_limits: false
      duration: Limited-time free models with no common public expiration stated.
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://opencode.ai/zen/v1/models
      free_model_ids:
        - big-pickle
        - deepseek-v4-flash-free
        - x-preview-f-free
        - muse-spark-1.2-contributor-free
        - mimo-v2.5-free
        - hy3-free
        - nemotron-3-ultra-free
        - nemotron-3.5-lightning-free
        - laguna-s-2.1-free
      caveat: Model objects did not expose price fields; free status was cross-checked against the first-party Zen documentation and labels.
    sources:
      - {type: official_docs, title: Zen documentation, url: https://opencode.ai/docs/zen}
      - {type: live_catalog, title: Zen models API, url: https://opencode.ai/zen/v1/models}

  - id: zai
    name: Z.AI Model API
    website: https://z.ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: rotating_zero_price
    api:
      base_url: https://api.z.ai/api/paas/v4
      compatibility: openai_style_chat_completions
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: not_documented
    limits:
      published_numeric_limits: false
      source_of_truth: Signed-in rate-limit dashboard.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://docs.z.ai/guides/overview/pricing
      free_models:
        - {id: glm-4.7-flash, context_tokens: 200000}
        - {id: glm-4.6v-flash, context_tokens: 128000, modality: vision_and_text}
        - id: glm-4.5-flash
          context_tokens: 200000
          caveat: Global pricing still lists it, while Chinese documentation says it was to route to 4.7 after a 2026-01-30 shutdown; treat as an alias/conflict.
    sources:
      - {type: official_pricing, title: Z.AI pricing, url: https://docs.z.ai/guides/overview/pricing}
      - {type: official_docs, title: Models overview, url: https://docs.z.ai/guides/overview/overview}
      - {type: official_api_reference, title: Chat completion API, url: https://docs.z.ai/api-reference/llm/chat-completion}

  - id: llmapi_ai
    name: LLM.API
    website: https://llmapi.ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: rotating_zero_price
    api:
      base_url: https://api.llmapi.ai/v1
      compatibility: openai_compatible
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      without_purchased_credits: 5 requests per 10 minutes
      after_adding_credits: 20 requests per minute for free models
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://api.llmapi.ai/v1/models
      free_models:
        - {id: zaya1-8b, context_tokens: 131072, modalities: text_to_text}
    sources:
      - {type: official_docs, title: API resources and limits, url: https://docs.llmapi.ai/resources}
      - {type: live_catalog, title: Models API, url: https://api.llmapi.ai/v1/models}
      - {type: official_catalog, title: ZAYA1-8B model page, url: https://llmapi.ai/models/}

  - id: api_airforce
    name: Api.Airforce
    website: https://api.airforce/
    status: current
    qualifies: true
    confidence: medium_high
    free_kind: rotating_zero_price
    api:
      base_url: https://api.airforce/v1
      compatibility: openai_compatible
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      requests_per_minute: 1
      requests_per_day: 1000
      per_model_daily_token_cap: true
      per_model_daily_token_cap_amount: unpublished
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://api.airforce/v1/models
      coverage: Exact free variants are identified by :free aliases in the authenticated dynamic catalog.
      verified_examples:
        - gpt-oss-120b
        - gpt-oss-20b
        - qwen3-30b-a3b-fp8
        - glm-4.7-flash
      caveat: The unauthenticated catalog timed out during the audit, so this is not asserted as a complete snapshot.
    sources:
      - {type: official_pricing, title: Pricing, url: https://api.airforce/pricing/}
      - {type: official_docs, title: Quickstart, url: https://api.airforce/docs/quickstart/}
      - {type: official_docs, title: Models API, url: https://api.airforce/docs/api/models/}
      - {type: official_catalog, title: GPT OSS 120B, url: https://api.airforce/models/gpt-oss-120b/}

  - id: llm7
    name: LLM7
    website: https://llm7.io/
    status: current
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    api:
      base_url: https://api.llm7.io/v1
      compatibility: openai_compatible
      authentication: Optional for anonymous access; free token raises quota.
    eligibility:
      account_required: false
      payment_method_required: false
    limits:
      anonymous:
        tokens_per_day: 500000
        requests_per_hour: 60
        requests_per_minute: 10
        requests_per_second: 1
      free_account:
        tokens_per_day: 1000000
        requests_per_hour: 250
        requests_per_minute: 60
        requests_per_second: 2
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://docs.llm7.io/guides/models
      free_router_ids: [default, fast]
      paid_router_ids: [pro]
      caveat: Concrete model IDs are being phased out; default and fast dynamically select free routes.
    sources:
      - {type: official_product, title: LLM7 home and quota table, url: https://llm7.io/}
      - {type: official_docs, title: Models, url: https://docs.llm7.io/guides/models}

  - id: modelscope_inference
    name: ModelScope API-Inference
    website: https://modelscope.cn/
    status: current_restricted
    qualifies: true
    confidence: high
    free_kind: development_only
    commercial_use_allowed: false
    api:
      base_url: https://api-inference.modelscope.cn/v1/
      compatibility: openai_compatible
      authentication: ModelScope token
    eligibility:
      account_required: true
      alibaba_cloud_link_required: true
      real_name_verification_required: true
      allowed_use: noncommercial_and_nonprofit
    limits:
      calls_per_day_per_account: 2000
      calls_per_model_per_day: 200
      selected_expensive_models_calls_per_day: 100
      concurrency: Dynamic and model-specific.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Models carrying the API-Inference badge in the live ModelScope catalog.
      discovery_url: https://modelscope.cn/models?filter=inference_type&page=1
      current_example: Qwen/Qwen3.5-35B-A3B
      lower_quota_examples: [DeepSeek-R1-0528, DeepSeek-V3.2-Exp]
    history:
      - {date: "2024-12-06", event: "Early free API-Inference announcement with a Qwen 2.5-era catalog, now superseded"}
      - {date: "2026-08-14", event: Current free API-Inference resource page updated}
    sources:
      - {type: official_docs, title: API-Inference limits, url: https://modelscope.cn/docs/model-service/API-Inference/limits}
      - {type: official_docs, title: API-Inference introduction, url: https://modelscope.cn/docs/model-service/API-Inference/intro}
      - {type: official_product, title: Free API-Inference resources, url: https://www.modelscope.cn/learn/1409, updated_at: "2026-08-14"}

  - id: awanllm
    name: AwanLLM
    website: https://www.awanllm.com/
    status: current
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    api:
      base_url: https://api.awanllm.com/v1
      compatibility: openai_compatible
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      tokens: unlimited
      requests_per_minute: 20
      small_model_requests_per_day: 200
      medium_model_requests_per_day: 10
      large_model_requests_per_day: 10
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://www.awanllm.com/models
      free_model_ids:
        - Meta-Llama-3.1-8B-Instruct
        - Meta-Llama-3-8B-Instruct
        - Awanllm-Llama-3-8B-Dolfin
        - Awanllm-Llama-3-8B-Cumulus
        - Meta-Llama-3.1-70B-Instruct
        - Meta-Llama-3-70B-Instruct
    sources:
      - {type: official_pricing, title: Pricing, url: https://www.awanllm.com/pricing}
      - {type: official_catalog, title: Models, url: https://www.awanllm.com/models}
      - {type: official_docs, title: Quick start, url: https://www.awanllm.com/quick-start}

  - id: arliai
    name: Arli AI
    website: https://www.arliai.com/
    status: current
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    api:
      base_url: https://api.arliai.com/v1
      compatibility: openai_compatible
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      requests_per_model_per_two_days: 5
      max_context_tokens: 12000
      concurrent_requests: 1
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      coverage: All models in the dynamic text-generation catalog are trialable under the per-model quota.
      discovery_url: https://api.arliai.com/model/all
    sources:
      - {type: official_pricing, title: Pricing, url: https://www.arliai.com/pricing}
      - {type: official_docs, title: Text generation limits, url: https://www.arliai.com/docs/textgen}
      - {type: live_catalog, title: Public model catalog, url: https://api.arliai.com/model/all}

  - id: freeinference_org
    name: FreeInference.org
    website: https://freeinference.org/
    operator: Harvard SEAS MadSys Lab
    status: current
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    api:
      base_url: https://freeinference.org/v1
      compatibility: [openai_compatible, anthropic_compatible]
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
      intended_use: research_and_education
    limits:
      public_numeric_limits: false
      official_text: Generous quota
    privacy_caveat: Prompts and responses are logged; anonymized derivatives may be open-sourced.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://doc.freeinference.org/models
      free_chat_model_ids:
        - glm-5.1
        - minimax-m2.5
        - minimax-m3
        - qwen3.6-35b
        - diffusiongemma
        - deepseek-v4-flash
      free_embedding_model_ids: [bge-m3]
      paid_or_pro_exclusions: [glm-5.2, glm-5.3, kimi-k2.7-code]
    sources:
      - {type: official_product, title: FreeInference home, url: https://freeinference.org/}
      - {type: official_docs, title: Models, url: https://doc.freeinference.org/models}

  - id: fastrouter
    name: FastRouter
    website: https://fastrouter.ai/
    status: current
    qualifies: true
    confidence: medium_high
    free_kind: rotating_zero_price
    api:
      base_url: https://api.fastrouter.ai/v1
      compatibility: openai_compatible
      authentication: API key
    limits:
      published_numeric_limits: false
      caveat: The free gateway plan alone is BYOK/pass-through; only the model-specific free routes below are zero-priced inference.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://fastrouter.ai/models
      free_model_ids:
        - openai/gpt-oss-120b:free
        - google/gemma-4-26b-a4b-it
        - nvidia/nemotron-3-nano-30b:free
        - nvidia/nemotron-3-super:free
        - sarvam/sarvam-105b:free
    sources:
      - {type: official_catalog, title: Models, url: https://fastrouter.ai/models}
      - {type: official_pricing, title: Pricing, url: https://fastrouter.ai/pricing}

  - id: kilo_ai_gateway
    name: Kilo AI Gateway
    website: https://kilo.ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: rotating_zero_price
    api:
      base_url: https://api.kilo.ai/api/gateway
      compatibility: openai_compatible
      authentication: Optional for free models; authenticated use supported.
    eligibility:
      account_required: false
      payment_method_required: false
    limits:
      requests_per_hour_per_ip: 200
      price_usd: 0
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://api.kilo.ai/api/gateway/models
      free_model_ids:
        - cohere/north-mini-code:free
        - dots-studio/dots-3-note-preview:free
        - kilo-auto/free
        - liquid/lfm-2.5-2.6b:free
        - nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free
        - nvidia/nemotron-3-super-120b-a12b:free
        - nvidia/nemotron-3-ultra-550b-a55b:free
        - nvidia/nemotron-3.5-content-safety:free
        - nvidia/nemotron-3.5-lightning:free
        - poolside/laguna-s-2.1:free
        - poolside/laguna-xs-2.1:free
        - stepfun/step-3.7-flash:free
        - tencent/hy3:free
        - thinkingmachines/inkling-small:free
        - thinkingmachines/inkling:free
    sources:
      - {type: official_docs, title: Models and providers, url: https://kilo.ai/docs/gateway/models-and-providers}
      - {type: official_docs, title: Usage and billing, url: https://kilo.ai/docs/gateway/usage-and-billing}
      - {type: live_catalog, title: Gateway models API, url: https://api.kilo.ai/api/gateway/models}

  - id: scaleway_generative_apis
    name: Scaleway Generative APIs
    website: https://www.scaleway.com/en/generative-apis/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      compatibility: openai_compatible
      authentication: Scaleway credentials
    eligibility:
      account_required: true
      valid_payment_method_required_for_base_limits: true
    limits:
      free_tokens: 1000000
      transcription_minutes: 60
      recurrence: not_explicitly_documented
      caveat: Recorded as an account-level trial allowance, not a recurring free tier.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Free allowance is pooled across the current serverless catalog.
      discovery_url: https://www.scaleway.com/en/docs/generative-apis/reference-content/supported-models/
      representative_current_model_ids:
        - glm-5.2
        - deepseek-v4-flash-0731
        - gpt-oss-120b
        - mistral-small-3.2-24b-instruct-2506
        - pixtral-12b-2409
        - qwen3.6-35b-a3b
    history:
      - {event: Earlier beta pages advertised fully free access; current token-metered pricing supersedes that claim.}
    sources:
      - {type: official_pricing, title: Model as a Service pricing, url: https://www.scaleway.com/en/pricing/model-as-a-service/}
      - {type: official_docs, title: Supported models, url: https://www.scaleway.com/en/docs/generative-apis/reference-content/supported-models/}
      - {type: official_docs, title: Rate limits, url: https://www.scaleway.com/en/docs/generative-apis/reference-content/rate-limits/}

  - id: qwen_cloud
    name: Alibaba Cloud Model Studio / Qwen API
    website: https://www.alibabacloud.com/en/product/modelstudio
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      compatibility: [dashscope_native, openai_compatible]
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
      new_user_only: true
    limits:
      duration_days: 90
      basis: Each eligible model has its own token quota; existing models start at account activation and newly released models start at release.
      recurrence: false
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Models and per-model token amounts in the current free-quota table.
      discovery_url: https://docs.qwencloud.com/resources/free-quota
    history:
      - {date: "2026-04-15", event: Separate Qwen OAuth free access for Qwen Code was retired; this does not retire Model Studio new-user quotas.}
    sources:
      - {type: official_docs, title: Free quota, url: https://docs.qwencloud.com/resources/free-quota}
      - {type: official_pricing, title: Pricing overview, url: https://docs.qwencloud.com/developer-guides/getting-started/pricing}
      - {type: official_docs, title: Qwen Code authentication changes, url: https://qwenlm.github.io/qwen-code-docs/en/users/configuration/auth/}

  - id: sea_lion_api
    name: SEA-LION API
    operator: AI Singapore
    website: https://sea-lion.ai/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    nus_relationship: AI Singapore is a national program hosted by the National University of Singapore. This is independent of the requested Nous Research provider.
    api:
      base_url: https://api.sea-lion.ai/v1
      compatibility: [openai_chat_completions, openai_embeddings]
      authentication: Bearer trial API key
    eligibility:
      account_required: true
      signup_method: Google account through SEA-LION Playground
      keys_per_user: 1
      payment_method_required: not_documented
    limits:
      requests_per_minute_per_user: 10
      effective_date: "2026-06-04"
      total_tokens_or_duration: not_documented
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://api.sea-lion.ai/v1/models
      documented_current_model_ids:
        - aisingapore/Qwen-SEA-LION-v4.5-27B-IT
        - aisingapore/Llama-SEA-LION-v3.5-70B-R
        - aisingapore/SEA-Guard
        - aisingapore/SEA-LION-ModernBERT-Embedding-600M
      other_recently_documented_model_ids:
        - aisingapore/Gemma-SEA-LION-v4-27B-IT
    caveat: The credential is explicitly called a trial key and no permanent total allowance is published.
    sources:
      - {type: official_docs, title: SEA-LION API inference guide, url: https://docs.sea-lion.ai/guides/inferencing/api}
      - {type: official_catalog, title: SEA-LION models, url: https://sea-lion.ai/models/}

  - id: ndif
    name: NSF National Deep Inference Fabric
    operator: Northeastern University with NCSA/UIUC
    website: https://ndif.us/
    status: current_restricted
    qualifies: true
    confidence: high
    free_kind: development_only
    audience: research
    production_use_allowed: false
    api:
      compatibility: NNsight Python remote-execution API with access to and intervention on model internals; not OpenAI-compatible.
      authentication: Free NDIF API key
    eligibility:
      account_required: true
      payment_method_required: false
      allowed_use: research
    limits:
      published_numeric_limits: false
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://ndif.us/status/
      featured_model_ids:
        - meta-llama/Llama-3.1-70B
        - meta-llama/Llama-3.1-8B
        - meta-llama/Llama-3.1-405B
        - openai/gpt-oss-120b
      caveat: Live deployment status, not this static featured list, determines availability.
    sustainability: NSF Award 2408455 with compute from Delta at NCSA.
    sources:
      - {type: official_product, title: NDIF, url: https://ndif.us/}
      - {type: official_docs, title: Get started, url: https://ndif.us/get-started/}
      - {type: live_status, title: Model deployment status, url: https://ndif.us/status/}

  - id: ai_horde
    name: AI Horde
    website: https://aihorde.net/
    status: current
    qualifies: true
    headline_ongoing_free: conditional
    confidence: high
    free_kind: community_capacity
    api:
      base_url: https://aihorde.net/api/v2
      compatibility: Native asynchronous REST; submit text generation then poll status. Not OpenAI-compatible.
      authentication: Anonymous key 0000000000 or free registered key
    eligibility:
      account_required: false
      payment_method_required: false
    limits:
      mechanism: Nonmonetary kudos and queue priority rather than a fixed request quota.
      anonymous_priority: lowest
      load_shedding: Anonymous use can be restricted during load.
      monetary_purchase: Kudos cannot be bought or sold.
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://aihorde.net/api/v2/status/models?type=text
      caveat: Availability, speed, queue, and context depend on volunteer workers.
      live_text_model_ids:
        - aphrodite/TheDrummer/Cydonia-24B-v4.3
        - aphrodite/TheDrummer/Skyfall-31B-v4.2
        - coder3101/gemma-4-E4B-it-qat-q4_0-unquantized-heretic
        - google/gemma-4-31b
        - koboldcpp/Angelic_Eclipse-12B
        - koboldcpp/Cydonia-24B-v4.3
        - koboldcpp/digo-prayudha/unsloth-llama-3.2-1b-gguf
        - koboldcpp/Gemma-3-1B
        - koboldcpp/gemma-4-31B-it-heretic
        - koboldcpp/gemma-4-E2B-it-qat-UD-Q4_K_XL.gguf
        - koboldcpp/Gemma-4-E4B-it-Ultra-Uncensored-Heretic
        - koboldcpp/Gemma-4-E4B-Uncensored-HauhauCS-Aggressive
        - koboldcpp/khrystan-worker
        - koboldcpp/L3-8B-Stheno-v3.2-Q5_K_M
        - koboldcpp/L3-Super-Nova-RP-8B
        - koboldcpp/Llama-3.2-1B-Instruct
        - koboldcpp/Llama-3.2-3B
        - koboldcpp/llama-3.2-3b-instruct-q4_k_m
        - koboldcpp/Meta-Llama-3-2-3B-Instruct.Q4_K_M
        - koboldcpp/mini-magnum-12b-v1.1
        - koboldcpp/MN-12B-Mag-Mell-R1.Q5_K_M
        - koboldcpp/mradermacher/Cerebras-GPT-111M-instruction-GGUF
        - koboldcpp/mradermacher/pythia-70m-deduped.f16.gguf
        - koboldcpp/Qwen_Qwen3-0.6B-IQ4_XS
        - koboldcpp/Qwen/Qwen3.5-0.8B
        - koboldcpp/Rocinante-X-12B
    sustainability: Distributed volunteer inference capacity.
    sources:
      - {type: official_repository, title: AI Horde, url: https://github.com/Haidra-Org/AI-Horde}
      - {type: official_integration_docs, title: Integration guide, url: https://github.com/Haidra-Org/AI-Horde/blob/main/README_integration.md}
      - {type: live_catalog, title: Live text models, url: "https://aihorde.net/api/v2/status/models?type=text"}

  - id: pollinations
    name: Pollinations.ai
    website: https://pollinations.ai/
    status: current_community
    qualifies: true
    headline_ongoing_free: conditional
    confidence: medium
    free_kind: rotating_zero_price
    api:
      base_url: https://gen.pollinations.ai
      compatibility: [openai_chat_completions, images, embeddings, audio]
      authentication: Bearer key from enter.pollinations.ai
    limits:
      mechanism: Endpoint-specific rate limits plus Pollen credits and BYOP.
      caveat: The recurring free Pollen refill amount is not clearly documented.
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://gen.pollinations.ai/text/models
      openai_discovery_url: https://gen.pollinations.ai/v1/models
      live_catalog_count: 188
      observed_zero_price_community_models:
        - {id: Spit-fires/muse-glimmer, rpm: null}
        - {id: chigwell/llm7-fast, rpm: 250}
        - {id: "YoannDev90/muse-glimmer-30b:free", rpm: 10}
        - {id: chirag-gamer/gpt-oss-120b, rpm: 12}
        - {id: "YoannDev90/laguna-s-2.1:free", rpm: 30}
        - {id: "vendouple/laguna-s-2.1:free", rpm: 5}
        - {id: MarcosFRG/glm-4.6v-flash, rpm: 2}
        - {id: "YoannDev90/diffusiongemma-26b-a4b-it:free", rpm: 10}
        - {id: "vendouple/muse-glimmer-30b:free", rpm: 5}
      caveat: These are independently operated alpha/community endpoints, not a permanence guarantee; native Pollinations models now have positive Pollen prices.
    sources:
      - {type: official_repository, title: Pollinations repository, url: https://github.com/pollinations/pollinations}
      - {type: official_api_docs, title: API docs, url: https://github.com/pollinations/pollinations/blob/main/APIDOCS.md}
      - {type: live_catalog, title: Rich text model catalog, url: https://gen.pollinations.ai/text/models}

  - id: puter_js
    name: Puter.js AI
    website: https://puter.com/
    status: current_user_pays
    qualifies: true
    headline_ongoing_free: conditional
    confidence: high
    free_kind: user_pays
    api:
      compatibility: Browser and Node JavaScript puter.ai.chat() plus Puter Workers; not a conventional shared server API key.
      authentication: Each end user signs into Puter.
    eligibility:
      developer_provider_key_required: false
      end_user_account_required: true
    limits:
      free_monthly_end_user_allowance: amount_not_documented
      exhausted_behavior: End user is prompted to upgrade.
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://api.puter.com/puterai/chat/models/details
      live_catalog_count: 879
      exact_zero_price_model_count: 0
      coverage: The user allowance can fund supported models; “free” means developer-free/user-pays, not zero model prices.
    sources:
      - {type: official_docs, title: User-pays model, url: https://docs.puter.com/user-pays-model/}
      - {type: official_docs, title: Puter AI, url: https://docs.puter.com/AI/}
      - {type: official_tutorial, title: Free LLM API tutorial, url: https://developer.puter.com/tutorials/free-llm-api/}
      - {type: live_catalog, title: Model details API, url: https://api.puter.com/puterai/chat/models/details}

  - id: public_ai
    name: Public AI Inference Utility
    operator: Public AI nonprofit
    website: https://platform.publicai.co/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      base_url: https://api.publicai.co/v1
      compatibility: [openai_compatible, huggingface_inference_provider]
      authentication: Bearer API key plus User-Agent header
    eligibility:
      account_required: true
      payment_method_required: not_documented
    limits:
      free_tier_requests_per_minute: 100
      starter_credit_amount: not_documented
      overage: Positive token prices are deducted from the wallet after starter credit.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://platform.publicai.co/models
      current_model_ids:
        - swiss-ai/apertus-v1.5-8b
        - swiss-ai/apertus-v1.5-8b-thinking
        - swiss-ai/apertus-v1.5-70b
        - swiss-ai/apertus-v1.5-70b-thinking
        - swiss-ai/apertus-8b-instruct
        - swiss-ai/apertus-70b-instruct
        - aisingapore/Gemma-SEA-LION-v4-27B-IT
        - aisingapore/Qwen-SEA-LION-v4-32B-IT
        - allenai/Olmo-3-7B-Instruct
        - speakleash/Bielik-11B-v3.0-Instruct
        - utter-project/EuroLLM-22B-Instruct-2512
    history:
      - {date: "2025-09-17", event: A Hugging Face launch article said Public AI was free at the time; current starter-credit and wallet pricing supersedes that statement.}
    sources:
      - {type: official_docs, title: Public AI API docs, url: https://platform.publicai.co/docs}
      - {type: official_pricing, title: Plans, url: https://platform.publicai.co/plans}
      - {type: official_catalog, title: Models, url: https://platform.publicai.co/models}
      - {type: historical_announcement, title: Public AI joins Inference Providers, url: https://huggingface.co/blog/inference-providers-publicai, published_at: "2025-09-17"}

  - id: lightning_ai_model_apis
    name: Lightning AI Model APIs / litAI
    website: https://lightning.ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: recurring_credit
    api:
      base_url: https://lightning.ai/api/v1
      compatibility: [openai_chat_completions, litai]
      authentication: Lightning API key
    eligibility:
      account_required: true
      payment_method_required: false
      phone_verification_required: true
      country_availability_applies: true
      one_free_account_per_person: true
    limits:
      advertised_model_api_tokens_per_month: 30000000
      requests_per_minute: 15
      tokens_per_minute: 120000
      separate_platform_credits_usd_per_month: 15
      reset: Free platform balance is topped up to $15 on the first of each month and does not accumulate.
      caveat: Public pages do not fully reconcile the 30M-token promise with token-priced litAI calls and the separate $15 balance; the signed-in billing page is authoritative.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://api.lightning.ai/models
      coverage: Dynamic routed catalog, plus custom models deployed with Lightning credits.
      representative_current_ids: [lightning-ai/nvidia-nemotron-3-ultra-550b-a55b, lightning-ai/deepseek-v4-pro, lightning-ai/gemma-4-31B-it, lightning-ai/gpt-oss-120b, anthropic/claude-opus-4-8, google/gemini-3.5-flash, openai/gpt-5.5-2026-04-23]
    sources:
      - {type: official_docs, title: Model APIs, url: https://lightning.ai/docs/overview/model-apis}
      - {type: official_catalog, title: Model API catalog, url: https://api.lightning.ai/models}
      - {type: official_pricing, title: Lightning pricing, url: https://lightning.ai/pricing}
      - {type: official_docs, title: Account creation and free-credit eligibility, url: https://lightning.ai/docs/platform/overview/faq/create-account}
      - {type: official_docs, title: Billing, url: https://lightning.ai/docs/overview/faq/billing}

  - id: modal
    name: Modal
    website: https://modal.com/
    status: current
    qualifies: true
    confidence: high
    free_kind: recurring_credit
    api:
      deployment_base_url_pattern: https://{workspace}--{endpoint}.{region}.modal.direct/v1
      shared_base_url: https://inference.us-west.modal.direct/v1
      compatibility: [openai_chat_completions, openai_responses, custom_http]
      authentication: Modal proxy token
    eligibility:
      account_required: true
      payment_method_required: true
    limits:
      plan: Starter
      plan_price_usd_per_month: 0
      included_credit_usd_per_month: 30
      containers: 100
      gpu_concurrency: 10
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Public Hugging Face models and custom/private fine-tunes deployed by the user; there is no fixed free-model list.
      exact_official_examples: [Qwen/Qwen3.5-4B, Qwen/Qwen3.6-27B, aisingapore/Qwen-SEA-LION-v4.5-27B-IT]
    sources:
      - {type: official_pricing, title: Pricing, url: https://modal.com/pricing}
      - {type: official_docs, title: Billing, url: https://modal.com/docs/guide/billing}
      - {type: official_docs, title: Endpoints, url: https://modal.com/docs/guide/endpoints}
      - {type: official_docs, title: Shared endpoint integrations, url: https://modal.com/docs/guide/endpoint-integrations}

  - id: beam_cloud
    name: Beam
    website: https://www.beam.cloud/
    status: current
    qualifies: true
    confidence: high
    free_kind: recurring_credit
    api:
      base_url_pattern: https://{app}-{deployment}-v{version}.app.beam.cloud
      compatibility: [custom_rest, openai_compatible_via_vllm_or_sglang]
      authentication: Bearer Beam token
    eligibility:
      account_required: true
      payment_method_required: true
      plan_selection_required: Developer pay-as-you-go
    limits:
      plan_price_usd_per_month: 0
      included_credit_usd_per_month: 30
      reset: monthly
      gpu_concurrency: 5
      cpu_concurrency: 30
      api_requests: unlimited
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Arbitrary public, private, or custom user-deployed models; no fixed hosted catalog.
      exact_official_examples: [OpenGVLab/InternVL3-8B-AWQ, 01-ai/Yi-Coder-9B-Chat, Qwen/Qwen2.5-7B-Instruct]
    sources:
      - {type: official_pricing, title: Pricing, url: https://www.beam.cloud/pricing}
      - {type: official_docs, title: FAQ, url: https://docs.beam.cloud/v2/resources/faq}
      - {type: official_docs, title: Endpoint overview, url: https://docs.beam.cloud/v2/endpoint/overview}
      - {type: official_example, title: vLLM OpenAI-compatible server, url: https://docs.beam.cloud/v2/examples/vllm}
      - {type: official_announcement, title: Monthly credit confirmation, url: https://www.beam.cloud/blog/serverless-gpu-reinforcement-learning, published_at: "2026-07-02"}

  - id: sail_research
    name: Sail Research
    website: https://www.sailresearch.com/
    status: current
    qualifies: true
    confidence: high
    free_kind: recurring_credit
    api:
      base_url: https://api.sailresearch.com/v1
      compatibility: [openai_responses, openai_chat_completions, anthropic_messages]
      authentication: Bearer API key
    eligibility:
      account_required: true
      payment_method_required_for_monthly_refresh: true
    limits:
      included_credit_usd_per_month: 5
      reset: monthly
      rate_limits: No strict public limits; service priority depends on completion window.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://docs.sailresearch.com/
      representative_current_models: [GLM-5.2, DeepSeek V4 Pro, Kimi-K2.6, gpt-oss-120b, Nemotron 3 Super 120B, Gemma 4 31B IT]
    sources:
      - {type: official_pricing, title: Sail Research pricing and API overview, url: https://www.sailresearch.com/}
      - {type: official_docs, title: Sail Research documentation, url: https://docs.sailresearch.com/}

  - id: cartesia
    name: Cartesia
    website: https://www.cartesia.ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    modalities: [text_to_speech, speech_to_text]
    api:
      base_url: https://api.cartesia.ai
      compatibility: cartesia_rest_and_websocket
      authentication: Cartesia API key
    eligibility:
      account_required: true
      payment_method_required: not_explicitly_documented
      commercial_use_on_free_plan: false
    limits:
      plan_price_usd_per_month: 0
      credits_per_month: 20000
      approximate_tts_minutes_per_month: 27
      approximate_stt_hours_per_month: 1.85
      tts_concurrent_requests: 2
      stt_concurrent_requests: 8
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      current_families: [Sonic-3.5, Ink-2]
    sources:
      - {type: official_pricing, title: Pricing, url: https://www.cartesia.ai/pricing}
      - {type: official_docs, title: API documentation, url: https://docs.cartesia.ai/}

  - id: elevenlabs_api
    name: ElevenLabs API
    website: https://elevenlabs.io/
    status: current_restricted
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    modalities: [text_to_speech, speech_to_text, sound_generation]
    api:
      base_url: https://api.elevenlabs.io/v1
      compatibility: elevenlabs_rest_and_websocket
      authentication: ElevenLabs API key
    eligibility:
      account_required: true
      payment_method_required: false
      free_plan_license: noncommercial_with_attribution
    limits:
      plan_price_usd_per_month: 0
      credits_reset: monthly
      flash_or_turbo_tts_characters_per_month: 20000
      multilingual_tts_characters_per_month: 10000
      caveat: Most, but not every, API endpoint is available on Free; each request consumes the account's credits.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      representative_current_models: [eleven_v3, eleven_flash_v2_5]
    sources:
      - {type: official_pricing, title: ElevenAPI pricing, url: https://elevenlabs.io/pricing/api}
      - {type: official_docs, title: API availability on Free, url: https://elevenlabs.io/docs/help-center/technical/how-much-does-it-cost-to-use-the-api}
      - {type: official_docs, title: Billing and Free plan, url: https://elevenlabs.io/docs/overview/administration/billing}

  - id: ovhcloud_ai_endpoints
    name: OVHcloud AI Endpoints
    website: https://www.ovhcloud.com/en/public-cloud/ai-endpoints/
    status: current_restricted
    qualifies: true
    confidence: high
    free_kind: rotating_zero_price
    headline_general_llm_free: false
    scope_note: Qualifies for modality-agnostic hosted inference; it does not currently provide a free general-purpose chat model.
    api:
      unified_openai_base_url: https://oai.endpoints.kepler.ai.cloud.ovh.net/v1
      compatibility: [openai_images, model_specific_rest, grpc_tts, openai_style_guard]
      authentication: Anonymous calls are allowed on documented zero-price endpoints; an OVHcloud access key raises limits.
    limits:
      anonymous_requests_per_minute_per_ip_per_model: 2
      authenticated_requests_per_minute_per_project_per_model: 400
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      exact_zero_price_ids: [Qwen3Guard-Gen-8B, Qwen3Guard-Gen-0.6B, stable-diffusion-xl-base-v10, nvr-tts-de-de, nvr-tts-en-us, nvr-tts-es-es, nvr-tts-it-it]
      discovery_url: https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/
    separate_trial:
      credit_usd: 200
      duration: one month
      payment_method_required: true
      caveat: This finite Public Cloud trial is separate from the explicitly zero-price models.
    sources:
      - {type: official_catalog, title: AI Endpoints catalog, url: https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/}
      - {type: official_model_page, title: SDXL endpoint and rate limits, url: https://www.ovhcloud.com/en-gb/public-cloud/ai-endpoints/catalog/stable-diffusion-xl/}
      - {type: official_model_page, title: Qwen Guard endpoint, url: https://www.ovhcloud.com/de/public-cloud/ai-endpoints/catalog/qwen-guard-gen-8b/}
      - {type: official_trial, title: Public Cloud free trial, url: https://www.ovhcloud.com/en/public-cloud/free-trial/}

  - id: requesty
    name: Requesty LLM Gateway
    website: https://www.requesty.ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: rotating_zero_price
    api:
      base_url: https://router.requesty.ai/v1
      eu_base_url: https://router.eu.requesty.ai/v1
      compatibility: [openai_compatible, anthropic_compatible]
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      requests_per_day: 200
      reset: daily
      caveat: The Free plan is restricted to zero-priced models; PAYG and BYOK catalogs are separate.
    models:
      snapshot_at: "2026-08-21T23:28:44-05:00"
      verification: live_catalog
      discovery_url: https://router.requesty.ai/v1/models
      live_total_models: 668
      zero_price_model_ids: [nvidia/nemotron-3-super-120b-a12b, nvidia/nemotron-3-nano-omni-30b-a3b-reasoning, nvidia/nemotron-3-nano-30b-a3b, nvidia/nemotron-3.5-content-safety, nvidia/nemotron-3-ultra-550b-a55b, nvidia/muse-glimmer-30b, nvidia/nemotron-3.5-lightning-30b-a3b, google/gemma-4-31b-it, poolside/laguna-xs.2, poolside/laguna-m.1, mistral/leanstral-1-5, novita/inclusionai/ling-3.0-tiny]
      caveat: Two old Poolside entries and nemotron-3-nano-30b-a3b reported max_output_tokens=0; runtime callability needs authenticated verification.
    sources:
      - {type: official_pricing, title: Pricing, url: https://www.requesty.ai/pricing}
      - {type: official_docs, title: Documentation index, url: https://docs.requesty.ai/llms.txt}
      - {type: live_catalog, title: Models API, url: https://router.requesty.ai/v1/models}

  - id: inception_platform
    name: Inception Platform
    website: https://platform.inceptionlabs.ai/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      base_url: https://api.inceptionlabs.ai/v1
      compatibility: [openai_chat_completions, fim_completions, edit_completions]
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
      new_account_only: true
    limits:
      free_tokens: 100000000
      recurrence: false
      expiry: not_published
      requests_per_minute: 1000
      input_tokens_per_minute: 1000000
      output_tokens_per_minute: 100000
      caveat: The July 2026 announcement raised the older 10M-token documentation value to 100M; after depletion billing information is required.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      free_coverage: Shared initial allowance across both production models.
      model_ids: [mercury-2, mercury-edit-2]
    sources:
      - {type: official_docs, title: Platform quickstart, url: https://docs.inceptionlabs.ai/get-started/get-started}
      - {type: official_docs, title: Models and pricing, url: https://docs.inceptionlabs.ai/get-started/models}
      - {type: official_docs, title: Rate limits, url: https://docs.inceptionlabs.ai/get-started/rate-limits}
      - {type: official_announcement, title: Mercury 2 free-token increase, url: https://www.inceptionlabs.ai/blog/mercury-2-10x-free-tokens, published_at: "2026-07-29"}

  - id: poolside_direct_api
    name: Poolside direct API
    website: https://poolside.ai/models
    status: current_promotional
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: promotional
    api:
      base_url: https://inference.poolside.ai/v1
      compatibility: openai_compatible
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: not_documented
    limits:
      numeric_rate_limits: not_published
      expiry: not_published
      official_duration: Free for a limited time.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      current_names: [Laguna S 2.1, Laguna XS 2.1]
      confirmed_api_id: poolside/laguna-s-2.1
      likely_second_api_id: poolside/laguna-xs-2.1
      caveat: The second exact ID matches current router catalogs, but Poolside's direct models endpoint requires authentication.
    sources:
      - {type: official_product, title: Poolside models, url: https://poolside.ai/models}
      - {type: official_docs, title: Supported models, url: https://docs.poolside.ai/get-started/supported-models}
      - {type: official_announcement, title: Laguna XS.2 and M.1, url: https://poolside.ai/blog/introducing-laguna-xs2-m1, published_at: "2026-04-28"}

  - id: voyage_ai
    name: Voyage AI
    website: https://www.voyageai.com/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    modalities: [embeddings, multimodal_embeddings, reranking]
    api:
      base_url: https://api.voyageai.com/v1
      compatibility: voyage_rest
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required_for_initial_quota: false
    limits:
      recurrence: false
      free_text_tokens_by_model:
        200000000: [voyage-4-large, voyage-4, voyage-4-lite, voyage-context-4, voyage-code-3]
        50000000: [voyage-multilingual-2, voyage-finance-2, voyage-law-2, voyage-code-2]
      free_multimodal_text_tokens: 200000000
      free_multimodal_pixels: 150000000000
      free_rerank_tokens: 200000000
      batch_api_included: false
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      current_rerank_ids: [rerank-2.5, rerank-2.5-lite, rerank-2, rerank-2-lite]
      caveat: Current pricing text elsewhere mentions voyage-code-4 where the free-token table says voyage-code-3; preserve the table value until corrected.
    sources:
      - {type: official_pricing, title: Pricing and free tokens, url: https://docs.voyageai.com/docs/pricing}
      - {type: official_docs, title: Rate limits, url: https://docs.voyageai.com/docs/rate-limits}

  - id: ai21_studio
    name: AI21 Studio
    website: https://www.ai21.com/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      base_url: https://api.ai21.com/studio/v1
      compatibility: ai21_rest
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required_initially: false
    limits:
      signup_credit_usd: 10
      expires_after_months: 3
      recurrence: false
      coverage: API, SDK, and playground usage.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://docs.ai21.com/docs/models
    sources:
      - {type: official_pricing, title: Usage and cost, url: https://docs.ai21.com/docs/usage-cost}
      - {type: official_docs, title: Models, url: https://docs.ai21.com/docs/models}

  - id: deepgram
    name: Deepgram
    website: https://deepgram.com/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    modalities: [speech_to_text, text_to_speech, voice_agents]
    api:
      base_url: https://api.deepgram.com/v1
      compatibility: deepgram_rest_and_websocket
      authentication: Deepgram API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      signup_credit_usd: 200
      expires_after_years: 1
      recurrence: false
      coverage: All public model endpoints.
    sources:
      - {type: official_pricing, title: Pricing, url: https://deepgram.com/pricing}
      - {type: official_docs, title: Promotional credit expiry, url: https://developers.deepgram.com/guides/deep-dives/managing-projects}

  - id: jina_ai_search_foundation
    name: Jina Search Foundation API
    website: https://jina.ai/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    modalities: [embeddings, multimodal_embeddings, reranking, classification, vision_language, grounded_research]
    api:
      base_url: https://api.jina.ai/v1
      compatibility: openai_embeddings_and_jina_rest
      authentication: Bearer Jina API key
    eligibility:
      account_required: true
      payment_method_required: false
      signup_token_use: noncommercial_only
      signup_token_license: CC-BY-NC
    limits:
      signup_tokens: 10000000
      recurrence: false
      free_token_expiry: not_documented
      conservative_embedding_and_rerank_requests_per_minute: 100
      conservative_embedding_and_rerank_tokens_per_minute: 100000
      caveat: The product table gives the conservative limits above while the versioned OpenAPI page publishes higher Free-tier limits; use the lower figures until response headers or the dashboard resolve the conflict.
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://api.jina.ai/v1/models
      current_catalog_count: 29
      coverage: Current embedding, reranker, classifier, vision-language, and DeepSearch APIs; catalog entries have positive prices and draw down the one-time balance.
    sources:
      - {type: official_api_reference, title: Jina Search Foundation API, url: https://api.jina.ai/docs}
      - {type: official_product, title: Reader and Search API limits, url: https://jina.ai/reader/}

  - id: jina_ai_reader
    name: Jina Reader and anonymous utility APIs
    website: https://jina.ai/reader/
    status: current
    qualifies: true
    confidence: high
    free_kind: ongoing_free
    modalities: [webpage_to_markdown, pdf_extraction, image_captioning, tokenization, segmentation]
    api:
      reader_base_url: https://r.jina.ai
      segmenter_url: https://api.jina.ai/v1/segment
      compatibility: jina_rest
      authentication: None at anonymous limits; an optional Bearer key raises limits.
    eligibility:
      account_required: false
      payment_method_required: false
    limits:
      reader_anonymous_requests_per_minute_per_ip: 20
      segmenter_anonymous_requests_per_minute_per_ip: 20
      segmenter_tokens_charged: 0
      reader_with_free_key_requests_per_minute: 500
      caveat: Supplying a key to Reader charges its token balance; anonymous basic Reader calls remain free.
    models:
      snapshot_at: "2026-08-21"
      verification: live_call
      coverage: Service-level APIs rather than caller-selected model routes.
      implementation_models_named_by_jina: [ReaderLM-v2, jina-vlm]
      observed_calls: Anonymous Reader and Segmenter requests both returned non-empty HTTP 200 responses; Segmenter reported zero tokens used.
    sources:
      - {type: official_product, title: Reader API and current limits, url: https://jina.ai/reader/}
      - {type: official_terms, title: Legal information, url: https://jina.ai/legal/, updated_at: "2026-05-04"}

  - id: mancer_ai
    name: Mancer AI
    operator: Sunlit Software, Inc.
    website: https://mancer.tech/
    status: current
    qualifies: true
    confidence: high
    free_kind: rotating_zero_price
    api:
      base_url: https://neuro.mancer.tech/oai/v1
      compatibility: [openai_chat_completions, openai_completions, openai_models]
      authentication: Bearer API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      numeric_quota: not_published
      official_claim: Free models continue working even with a negative credit balance.
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://neuro.mancer.tech/oai/v1/models
      free_model_ids: [mytholite]
      model_details: {name: MythoLite, base: LLaMA 2 13B, context_tokens: 2560, max_completion_tokens: 150}
      caveat: A stale page fragment calls Rocinante free; the live catalog and current model table identify only mytholite at zero price.
    sources:
      - {type: official_product, title: Mancer AI, url: https://mancer.tech/}
      - {type: official_pricing, title: Pricing FAQ, url: https://mancer.tech/pricing}
      - {type: live_catalog, title: Models API, url: https://neuro.mancer.tech/oai/v1/models}
      - {type: official_api_reference, title: OpenAPI specification, url: https://mancer.tech/resources/api-docs-webui.yml}

  - id: mara_inference_cloud
    name: MARA Inference Cloud
    website: https://cloud.mara.com/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      base_url: https://api.cloud.mara.com/v1
      compatibility: openai_compatible
      authentication: Bearer API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      signup_credit_usd: 5
      expires_after_days: 30
      recurrence: false
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://api.cloud.mara.com/v1/models
      current_ids: [DeepSeek-V3.1, DeepSeek-V3.2, MiniMax-M2.7, gemma-4-31B-it, gpt-oss-120b]
      caveat: Plans distinguish production from preview access, so do not assume every catalog entry is available to trial accounts without an authenticated check.
    sources:
      - {type: official_pricing, title: Plans, url: https://cloud.mara.com/plans}
      - {type: official_product, title: Dashboard and quickstart, url: https://cloud.mara.com/}
      - {type: live_catalog, title: Models API, url: https://api.cloud.mara.com/v1/models}

  - id: assemblyai
    name: AssemblyAI
    website: https://www.assemblyai.com/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    modalities: [speech_to_text, speech_understanding, audio_guardrails]
    api:
      base_url: https://api.assemblyai.com
      streaming_url: wss://streaming.assemblyai.com/v3/ws
      compatibility: assemblyai_rest_and_websocket
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      signup_credit_usd: 50
      expiry: none
      recurrence: false
      exclusions: [LLM Gateway]
    sources:
      - {type: official_help, title: Free signup, url: https://support.assemblyai.com/articles/5370767329-can-i-sign-up-for-free}
      - {type: official_docs, title: Account billing and free credits, url: https://www.assemblyai.com/docs/faq/how-to-get-your-api-key}

  - id: speechmatics
    name: Speechmatics
    website: https://www.speechmatics.com/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    modalities: [speech_to_text, text_to_speech]
    api:
      compatibility: speechmatics_rest_and_realtime_websocket
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      signup_credit_usd: 100
      recurrence: false
      realtime_concurrent_sessions: 2
      languages: 55+
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Batch and real-time STT plus TTS under the shared credit balance.
    sources:
      - {type: official_pricing, title: Speech API pricing, url: https://www.speechmatics.com/pricing}
      - {type: official_announcement, title: Credit-based billing, url: https://www.speechmatics.com/company/articles-and-news/moving-to-credit-based-billing}

  - id: aws_bedrock
    name: Amazon Bedrock
    website: https://aws.amazon.com/bedrock/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      openai_base_url_pattern: https://bedrock-mantle.{region}.api.aws/openai/v1
      compatibility: [openai_responses, openai_chat_completions, anthropic_messages, bedrock_converse, bedrock_invoke]
      authentication: AWS credentials, SigV4, or Bedrock API key depending on endpoint
    eligibility:
      new_aws_customer_only: true
      payment_method_required: true
    limits:
      signup_credit_usd: 100
      additional_earnable_credit_usd: 100
      free_plan_duration_months: 6
      credit_expiry_months_from_account_creation: 12
      recurrence: false
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Dynamic regional multi-provider catalog; models have positive unit prices and draw down general AWS trial credit.
    sources:
      - {type: official_announcement, title: AWS Free Tier credits and six-month plan, url: https://aws.amazon.com/about-aws/whats-new/2025/07/aws-free-tier-credits-month-free-plan/, published_at: "2025-07-16"}
      - {type: official_docs, title: Free Tier FAQ, url: https://docs.aws.amazon.com/awsaccountbilling/latest/aboutv2/free-tier-FAQ.html}
      - {type: official_pricing, title: Bedrock pricing, url: https://aws.amazon.com/bedrock/pricing/}
      - {type: official_docs, title: OpenAI-compatible inference, url: https://docs.aws.amazon.com/bedrock/latest/userguide/inference-chat-completions-mantle.html}

  - id: azure_ai_foundry
    name: Microsoft Foundry Models
    website: https://azure.microsoft.com/products/ai-foundry/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      foundry_resource_base_url: https://{resource}.services.ai.azure.com/api/
      openai_base_url: https://{resource}.openai.azure.com/openai/v1/
      compatibility: [openai_v1, foundry_native]
    eligibility:
      new_customer_only: true
      phone_required: true
      non_prepaid_payment_card_required: true
      automatic_charging: false
    limits:
      general_cloud_credit_usd: 200
      duration_days: 30
      recurrence: false
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Dynamic regional Foundry catalog; positively priced serverless or provisioned inference draws down Azure trial credit.
    sources:
      - {type: official_trial, title: Azure free account, url: https://azure.microsoft.com/free/}
      - {type: official_pricing, title: Foundry Models pricing, url: https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/microsoft/}
      - {type: official_docs, title: Foundry application integration, url: https://learn.microsoft.com/en-us/azure/foundry/how-to/integrate-with-other-apps}

  - id: google_vertex_ai
    name: Google Cloud Vertex AI
    website: https://cloud.google.com/vertex-ai
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      native_url_pattern: https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/publishers/google/models/{model}:generateContent
      openai_base_url_pattern: https://{location}-aiplatform.googleapis.com/v1/projects/{project}/locations/{location}/endpoints/openapi
      authentication: Google Cloud OAuth/IAM
    eligibility:
      new_customer_only: true
      payment_method_required_for_verification: true
    limits:
      general_cloud_credit_usd: 300
      duration_days: 90
      recurrence: false
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      coverage: Eligible first-party Google-managed Vertex services and models only.
      exclusions:
        - Gemini Developer API / AI Studio usage; its independent free tier is a separate record.
        - Generative AI partner models offered as managed API services.
    sources:
      - {type: official_trial, title: Google Cloud free program, url: https://docs.cloud.google.com/free/docs/free-cloud-features}
      - {type: official_pricing, title: Vertex generative AI pricing, url: https://cloud.google.com/vertex-ai/generative-ai/pricing}
      - {type: official_docs, title: OpenAI-compatible Gemini call on Vertex, url: https://docs.cloud.google.com/vertex-ai/generative-ai/docs/samples/generativeaionvertexai-gemini-chat-completions-non-streaming}

  - id: oracle_oci_generative_ai
    name: OCI Generative AI
    website: https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      openai_base_url_pattern: https://inference.generativeai.{region}.oci.oraclecloud.com/openai/v1
      compatibility: [openai_responses, openai_chat_completions, oci_native_inference]
      authentication: OCI Generative AI API key or OCI IAM
    eligibility:
      new_customer_only: true
      mobile_phone_required_for_most_users: true
      payment_card_required_for_most_users: true
    limits:
      general_cloud_credit_usd: 300
      duration_days: 30
      recurrence: false
    caveat: Generative AI is not an OCI Always Free service; only the finite general cloud trial offsets eligible usage.
    sources:
      - {type: official_trial, title: OCI Free Tier, url: https://docs.oracle.com/en-us/iaas/Content/FreeTier/freetier.htm}
      - {type: official_trial, title: Oracle Cloud Free, url: https://www.oracle.com/cloud/free/}
      - {type: official_docs, title: OpenAI-compatible API, url: https://docs.oracle.com/en-us/iaas/Content/generative-ai/openai-compatible-api.htm}

  - id: replicate
    name: Replicate
    website: https://replicate.com/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      base_url: https://api.replicate.com/v1
      compatibility: replicate_native_rest
      authentication: Bearer API token
    eligibility:
      account_required: true
      payment_method_required_initially: false
    limits:
      coverage: Dynamic subset of models can be run free for a small unquantified allowance.
      recurrence: none_documented
      no_card_granted_credit_requests_per_second: 1
      no_card_granted_credit_requests_per_minute: 6
      caveat: Billing setup is required after the model-specific initial allowance is exhausted.
    sources:
      - {type: official_docs, title: Billing, url: https://replicate.com/docs/topics/billing}
      - {type: official_docs, title: Prepaid credit, url: https://replicate.com/docs/topics/billing/prepaid-credit}
      - {type: official_docs, title: Prediction rate limits, url: https://replicate.com/docs/topics/predictions/rate-limits}

  - id: cerebras_inference
    name: Cerebras Inference
    website: https://www.cerebras.ai/inference
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      base_url: https://api.cerebras.ai/v1
      compatibility: openai_compatible
      authentication: API key
    eligibility:
      account_required: true
      verified_payment_method_required: true
    limits:
      signup_credit_usd: 5
      expires_after_days: 30
      recurrence: false
      caveat: Current documentation explicitly says no permanently free tier exists.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      trial_model_limits:
        - {id: gpt-oss-120b, rpm: 5, tpm: 30000, tpd: 1000000}
        - {id: gemma-4-31b, rpm: 5, tpm: 30000, tpd: 1000000}
    history:
      - {date: "2024-08-27", event: Launched with one million free tokens per day; that recurring offer ended.}
    sources:
      - {type: official_docs, title: Current trial and no-free-tier statement, url: https://inference-docs.cerebras.ai/support/rate-limits}
      - {type: official_announcement, title: Inference launch, url: https://www.cerebras.ai/blog/introducing-cerebras-inference-ai-at-instant-speed, published_at: "2024-08-27"}

  - id: clarifai
    name: Clarifai
    website: https://www.clarifai.com/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      native_base_url: https://api.clarifai.com
      openai_base_url: https://api.clarifai.com/v2/ext/openai/v1
      compatibility: [clarifai_native, openai_compatible]
    eligibility:
      phone_verification_required: true
      payment_method_required_initially: false
      payment_method_required_to_recharge: true
    limits:
      signup_credit_usd: 5
      expires_after_days: 30
      recurrence: false
      maximum_welcome_bonuses: 2
      default_requests_per_second: 15
    history:
      - {date: "2026-02-03", event: The recurring Community free plan retired and pay-as-you-go replaced it.}
    sources:
      - {type: official_docs, title: Account billing, url: https://docs.clarifai.com/control/account-billing/}
      - {type: official_docs, title: Inference, url: https://docs.clarifai.com/compute/inference/}
      - {type: official_docs, title: Rate limits, url: https://docs.clarifai.com/resources/api-overview/rate-limits/}
      - {type: official_changelog, title: Community plan retirement, url: https://docs.clarifai.com/product-updates/changelog/release121/, published_at: "2026-02-03"}

  - id: fireworks_ai
    name: Fireworks AI
    website: https://fireworks.ai/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      base_url: https://api.fireworks.ai/inference/v1
      compatibility: openai_compatible
      authentication: API key
    limits:
      signup_credit_usd: 1
      recurrence: false
      caveat: This is a small promotional signup credit, not a recurring free tier.
    sources:
      - {type: official_pricing, title: Pricing, url: https://fireworks.ai/pricing}
      - {type: official_billing_faq, title: Billing and pricing FAQ, url: https://docs.fireworks.ai/faq-new/billing-pricing/how-much-does-fireworks-cost}

  - id: nebius_token_factory
    name: Nebius Token Factory
    website: https://nebius.com/token-factory
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    api:
      base_url: https://api.tokenfactory.nebius.com/v1
      compatibility: openai_compatible
      authentication: API key
    limits:
      signup_credit_usd: 1
      recurrence: false
    source: {type: official_pricing, title: Token Factory prices, url: https://nebius.com/token-factory/prices}

  - id: novita_ai
    name: Novita AI
    website: https://novita.ai/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    limits:
      signup_voucher_amount: dashboard_only
      recurrence: false
      caveat: The finite new-user voucher is followed by prepaid top-ups; promotional vouchers can expire.
    sources:
      - {type: official_quickstart, title: Quickstart, url: https://novita.ai/docs/guides/quickstart}
      - {type: official_pricing, title: Pricing, url: https://novita.ai/pricing}

  - id: hyperbolic
    name: Hyperbolic
    website: https://www.hyperbolic.ai/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    eligibility:
      phone_verification_required: true
    limits:
      signup_credit_usd: 1
      recurrence: false
    source: {type: official_billing_docs, title: Billing and payments, url: https://www.hyperbolic.ai/docs/general/billing-payments}

  - id: waterfall
    name: Waterfall
    website: https://www.getwaterfall.org/
    status: current_community
    qualifies: true
    headline_ongoing_free: conditional
    confidence: high
    free_kind: rotating_zero_price
    api:
      base_url: https://api.getwaterfall.org/v1
      compatibility: openai_compatible
      authentication: Bearer API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      numeric_limit: not_public
      policy: community_rate_limits
      routing: free_smart
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://api.getwaterfall.org/v1/models
      live_total_models: 439
      free_model_ids: [nemotron-30b-free, nemotron-9b-free, nemotron-3-super-120b-free, nemotron-3-nano-30b-free, nemotron-nano-12b-vl-free, nemotron-nano-9b-v2-free, gemma-4-31b-free, gemma-4-26b-free, nemotron-12b-video-free, gemma-4-26b-a4b-it-free, gemma-4-31b-it-free, nemotron-3-nano-30b-a3b-free, nemotron-3-nano-omni-30b-a3b-reasoning-free, nemotron-3-super-120b-a12b-free, nemotron-3-ultra-550b-a55b-free, nemotron-3.5-content-safety-free, nemotron-nano-12b-v2-vl-free, laguna-xs-2.1-free, north-mini-code-free, laguna-s-2.1-free, nemotron-3.5-lightning-free, lfm-2.5-2.6b-free, dots-3-note-preview-free, glm-5.2-free, inkling-free, inkling-small-free, lyria-3-clip-preview-free, lyria-3-pro-preview-free]
    sustainability: Bootstrapped single-maintainer service with no SLA or permanence guarantee.
    sources:
      - {type: official_product, title: Waterfall, url: https://www.getwaterfall.org/}
      - {type: official_pricing, title: Pricing, url: https://www.getwaterfall.org/pricing/}
      - {type: official_docs, title: Documentation, url: https://www.getwaterfall.org/docs/}
      - {type: live_catalog, title: Models API, url: https://api.getwaterfall.org/v1/models}

  - id: logfare
    name: Logfare
    website: https://logfare.ai/
    status: current_community
    qualifies: true
    headline_ongoing_free: conditional
    confidence: high
    free_kind: ongoing_free
    api:
      base_url: https://logfare.ai/v1
      compatibility: [openai_chat_completions, openai_responses, anthropic_messages, embeddings, images, audio]
      authentication: Bearer API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      numeric_limit: none_published
      policy: fair_use
      caveat: Excessive or automated traffic may be throttled or blocked.
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://logfare.ai/v1/models
      standard_model_ids: [gemma-4-26b]
      premium_training_opt_in_ids: [kiro-auto, minimax-m3, moondream3.1, deepseek-v4-pro, glm-5.2, qwen-3.8-2.4t-a95b, kimi-k3, deepseek-v4-flash-0731, qwen-3.8-27b, deepseek-v4-pro-0813]
      other_live_models: [sdxl-lightning, whisper-large-v3-turbo, phoenix-1.0, qwen3-embedding-8b, flux-2-pro, aura-2-en, nova-3, lucid-origin, text-embedding-3-small]
    privacy:
      logging: Request and response bodies, IP, headers, and metadata are logged.
      standard_tier: May be used for internal evaluation after best-effort PII scrubbing.
      premium_tier: Requires prospective opt-in to model-training use.
    sustainability: Independent, self-funded, donation-supported service with no SLA.
    sources:
      - {type: official_product, title: Logfare, url: https://logfare.ai/}
      - {type: official_docs, title: API docs, url: https://logfare.ai/docs}
      - {type: official_terms, title: Terms, url: https://logfare.ai/tos}
      - {type: live_catalog, title: Models API, url: https://logfare.ai/v1/models}

  - id: bazaarlink
    name: BazaarLink
    operator: 集聯科技有限公司
    website: https://bazaarlink.ai/
    status: current
    qualifies: true
    confidence: high
    free_kind: rotating_zero_price
    geography: Taiwan
    api:
      base_url: https://api.bazaarlink.ai/v1
      compatibility: openai_compatible
      authentication: Bearer API key; programmatic agent registration is also supported.
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      requests_per_minute: 10
      requests_per_day: 50
      requests_per_day_after_any_topup: 100
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://api.bazaarlink.ai/v1/models
      free_model_ids: ["qwen/qwen3.7-flash:free", "auto:free"]
    sources:
      - {type: official_product, title: Free models, url: https://bazaarlink.ai/free}
      - {type: official_docs, title: API docs, url: https://bazaarlink.ai/en/docs/api}
      - {type: official_terms, title: Terms, url: https://bazaarlink.ai/en/terms}
      - {type: live_catalog, title: Models API, url: https://api.bazaarlink.ai/v1/models}

  - id: dreamprompting
    name: DreamPrompting
    website: https://dreamprompting.com/
    status: current_community
    qualifies: true
    headline_ongoing_free: conditional
    confidence: medium_high
    free_kind: ongoing_free
    api:
      base_url: https://dreamprompting.com/api/v1
      compatibility: openai_chat_completions
      authentication: Google sign-in and Bearer API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      requests_per_minute_per_ip: 100
      tokens_per_rolling_24_hours: 500000
      requests_per_rolling_24_hours: 5000
      max_input_tokens: 32000
      max_output_tokens: 8192
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      routing: Aggregates upstream free tiers, trials, and the keyless ch.at service.
      representative_ids: [google/gemini-3.1-flash-lite, groq/llama-3.3-70b-versatile, nvidia/meta/llama-3.3-70b-instruct, "openrouter/google/gemma-4-31b-it:free", mistral/mistral-small-latest, cohere/command-a-03-2025, chat/ch.at]
    caveat: Young upstream-dependent service; terms prohibit some unapproved automation while API docs promote agent use.
    sources:
      - {type: official_product, title: DreamPrompting, url: "https://dreamprompting.com/?lang=en"}
      - {type: official_docs, title: API docs, url: https://dreamprompting.com/api-docs}
      - {type: official_catalog, title: Models, url: https://dreamprompting.com/models}
      - {type: official_terms, title: Terms, url: https://dreamprompting.com/terms}

  - id: ch_at
    name: ch.at
    website: https://ch.at/
    status: current_community
    qualifies: true
    headline_ongoing_free: conditional
    confidence: high
    free_kind: community_capacity
    api:
      base_url: https://ch.at/v1
      compatibility: openai_chat_completions
      authentication: none
    eligibility:
      account_required: false
      payment_method_required: false
    limits:
      requests_per_minute_per_ip: 100
      burst: 10
      history_cap: 64KB
    models:
      snapshot_at: "2026-08-21"
      verification: authenticated_call
      operator_selected: true
      caveat: A live unauthenticated generation returned HTTP 200, but the response model is blank and the server ignores the caller's model field; do not publish a stable model ID.
    sustainability: Community service with no SLA.
    sources:
      - {type: official_repository, title: ch.at repository, url: https://github.com/Deep-ai-inc/ch.at}
      - {type: live_endpoint, title: Chat completions endpoint, url: https://ch.at/v1/chat/completions}

  - id: opentyphoon
    name: OpenTyphoon API
    operator: SCB 10X
    website: https://opentyphoon.ai/
    status: current_restricted
    qualifies: true
    confidence: high
    free_kind: development_only
    geography: Thailand
    api:
      base_url: https://api.opentyphoon.ai/v1
      compatibility: openai_compatible
      authentication: Playground account and Bearer API key
    eligibility:
      account_required: true
      payment_method_required: not_stated
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      current_limits:
        - {id: typhoon-v2.5-30b-a3b-instruct, context_tokens: 128000, rps: 5, rpm: 200}
        - {id: typhoon-v2.1-12b-instruct, context_tokens: 56000, rps: 5, rpm: 200}
        - {id: typhoon-ocr, rpm: 20}
        - {id: typhoon-asr-realtime, rpm: 100}
    caveat: Beta research showcase provided as-is without formal support.
    sources:
      - {type: official_docs, title: Documentation, url: https://docs.opentyphoon.ai/en/}
      - {type: official_docs, title: FAQ, url: https://docs.opentyphoon.ai/en/faq/}
      - {type: official_catalog, title: Models and limits, url: https://docs.opentyphoon.ai/en/models/}
      - {type: official_api_reference, title: API reference, url: https://docs.opentyphoon.ai/en/api-reference/}

  - id: alcf_inference_endpoints
    name: Argonne Leadership Computing Facility Inference Endpoints
    website: https://www.alcf.anl.gov/
    status: current_restricted
    qualifies: true
    confidence: high
    free_kind: development_only
    geography: United States
    audience: approved_research
    api:
      service_root: https://inference-api.alcf.anl.gov
      compatibility: [openai_chat, openai_responses, anthropic_messages, completions, embeddings, batches]
      authentication: ALCF account and Globus OAuth
    eligibility:
      approved_project_required: true
      programs: [Directors Discretionary, INCITE, ALCC, NAIRR]
      open_research_cost: generally_no_cost_compute_allocation
      proprietary_research: cost_recovery
    limits:
      numeric_user_limit: allocation_specific
      batch_max_requests: 150000
      cold_start_minutes: 10-15
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      discovery_url: https://inference-api.alcf.anl.gov/resource_server/list-endpoints
      representative_ids: [meta-llama/Meta-Llama-3.1-405B-Instruct, meta-llama/Llama-4-Maverick-17B-128E-Instruct, mistralai/Mistral-Large-Instruct-2407, openai/gpt-oss-120b, argonne/AuroraGPT-IT-v4-0125, google/gemma-4-31B-it, nvidia/nemotron-3-super-120b]
    sources:
      - {type: official_docs, title: Inference endpoints, url: https://docs.alcf.anl.gov/services/inference-endpoints/}
      - {type: official_docs, title: Allocation management, url: https://docs.alcf.anl.gov/account-project-management/allocation-management/}
      - {type: official_program, title: Discretionary allocations, url: https://www.alcf.anl.gov/science/directors-discretionary-allocation-program}
      - {type: official_repository, title: Inference endpoints repository, url: https://github.com/argonne-lcf/inference-endpoints}

  - id: fikra_api
    name: Fikra API
    website: https://fikraapi.co.ke/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    geography: Kenya
    api:
      base_url: https://api.fikraapi.co.ke/v1
      compatibility: openai_chat_completions
      authentication: Bearer API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      signup_tokens: 300000
      recurrence: false
      requests_per_minute: 30
      exhausted_behavior: HTTP 402 until top-up.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      model_ids: [fikra-fast-8b, fikra-pro-20b, fikra-pro-120b]
    sources:
      - {type: official_product, title: Fikra API, url: https://fikraapi.co.ke/}
      - {type: official_docs, title: Documentation, url: https://docs.fikraapi.co.ke/}

  - id: sarvam_ai
    name: Sarvam AI
    website: https://www.sarvam.ai/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    geography: India
    api:
      base_url: https://api.sarvam.ai/v2
      compatibility: sarvam_rest_with_openai_style_chat_parameters
      authentication: API subscription key
    eligibility:
      account_required: true
      payment_method_required: not_explicitly_stated
    limits:
      signup_credit_inr: 100
      expiry: none
      recurrence: false
      starter_requests_per_minute: 60
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://api.sarvam.ai/v2/models
      public_ids: [glm5.2, gemma4, sarvam-105b]
      caveat: glm5.2 and gemma4 require per-key beta whitelisting.
    sources:
      - {type: official_docs, title: Rate limits, url: https://docs.sarvam.ai/api/getting-started/ratelimits}
      - {type: official_catalog, title: Open-source models, url: https://docs.sarvam.ai/api/getting-started/models/open-source}
      - {type: official_pricing, title: API pricing, url: https://web.sarvam.dev/api-pricing}
      - {type: live_catalog, title: Models API, url: https://api.sarvam.ai/v2/models}

  - id: byteplus_modelark
    name: BytePlus ModelArk
    website: https://www.byteplus.com/en/product/modelark
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    geography: Singapore and supported global regions
    api:
      base_url: https://ark.ap-southeast.bytepluses.com/api/v3
      compatibility: [openai_chat_completions, openai_responses]
      authentication: ARK API key
    eligibility:
      enterprise_information_submission_required: true
      payment_method_required: not_explicitly_stated
    limits:
      typical_tokens_per_eligible_model: 500000
      recurrence: once_per_account
      exact_models_and_expiry: Dynamic in Model Activation and Billing Center.
      exclusions: [plugins, knowledge_bases, batch_inference, cache_storage]
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      representative_ids: [seed-1.6, seed-1.6-flash, seed-2-1-turbo, seed-translation, skylark-pro, skylark-vision]
    sources:
      - {type: official_docs, title: Free token package, url: https://docs.byteplus.com/en/docs/modelark/1399514}
      - {type: official_docs, title: API overview, url: https://docs.byteplus.com/api/docs/modelark/1465347}
      - {type: official_terms, title: Free-token campaign terms, url: https://docs.byteplus.com/en/docs/legal/termsandconditions_modelark_free-token_campaign}

  - id: tencent_hunyuan
    name: Tencent Hunyuan
    website: https://cloud.tencent.com/product/hunyuan
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    geography: China
    api:
      base_url: https://api.hunyuan.cloud.tencent.com/v1
      migration_base_url: https://tokenhub.tencentmaas.com/v1
      compatibility: openai_compatible
      authentication: Bearer API key
    eligibility:
      account_required: true
      real_name_verification_required: true
      payment_method_required: false_if_postpaid_not_enabled
    limits:
      activation_tokens: 1000000
      separate_embedding_tokens: 1000000
      expires_after_years: 1
      recurrence: false
      default_concurrency: 5
      exhausted_behavior: Does not auto-switch to paid unless postpaid is explicitly enabled.
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      representative_ids: [Hunyuan-a13b, Hunyuan-role-latest, Hunyuan-translation, Hunyuan-translation-lite, Tencent-HY-Vision-1.5-Instruct, Hunyuan-embedding]
    sources:
      - {type: official_docs, title: Free resource package, url: https://cloud.tencent.com/document/product/1729/97731}
      - {type: official_docs, title: API access, url: https://cloud.tencent.com/document/product/1729/111007}
      - {type: official_docs, title: Migration and current aliases, url: https://cloud.tencent.com/document/product/1729/131925}

  - id: upstage
    name: Upstage
    website: https://www.upstage.ai/
    status: current_trial_and_restricted_program
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    geography: South Korea and supported regions
    api:
      base_url: https://api.upstage.ai/v1
      compatibility: openai_compatible
      authentication: Bearer API key
    general_signup:
      credit_usd: 10
      recurrence: false
      expiry: not_public
      current_model_example: solar-pro3
    institutional_program:
      eligibility: [K-12 schools, universities, university hospitals, nonprofits, NGOs]
      coverage: [solar-pro2, solar-pro3, document-parse]
      duration: up to one year
      final_access_date: "2027-03-31T23:00:00+09:00"
      approval_required: true
    sources:
      - {type: official_guide, title: Console and API guide, url: https://www.upstage.ai/blog/en/guide-1-upstage-console-api}
      - {type: official_pricing, title: API pricing, url: https://www.upstage.ai/pricing/api}
      - {type: official_program, title: AI Initiative 2026, url: https://www.upstage.ai/events/ai-initiative-2026-en}

  - id: maritaca_academic_credits
    name: Maritaca AI Academic Credits
    website: https://www.maritaca.ai/
    status: current_restricted
    qualifies: true
    confidence: high
    free_kind: development_only
    geography: Brazil
    audience: research_and_teaching
    api:
      base_url: https://chat.maritaca.ai/api
      compatibility: openai_compatible
      authentication: Approved account and API key
    eligibility:
      application_required: true
      applicants: [students, faculty, researchers]
      project_types: [teaching, scientific_research]
    limits:
      credit_amount: not_public
      duration: not_public
      tier_0_requests_per_minute: 60
      tier_0_input_tokens_per_minute: 128000
      tier_0_output_tokens_per_minute: 10000
    models:
      snapshot_at: "2026-08-21"
      verification: official_current
      current_ids: [sabia-4-thinking, sabia-4-thinking-br-sp, sabia-4, sabia-4-2026-01-06, sabia-4-br-sp, sabiazinho-4, sabiazinho-4-2026-01-06, sabia-4-small, sabiazim-4, sabiazinho-4-br-sp]
    sources:
      - {type: official_program, title: Academic credits, url: https://www.maritaca.ai/research}
      - {type: official_catalog, title: Models, url: https://docs.maritaca.ai/pt/modelos}
      - {type: official_docs, title: Rate limits, url: https://docs.maritaca.ai/pt/rate-limits}
      - {type: official_pricing, title: Pricing, url: https://docs.maritaca.ai/pt/precos}

  - id: wavespeedai
    name: WaveSpeedAI
    website: https://wavespeed.ai/
    status: current_trial
    qualifies: true
    headline_ongoing_free: false
    confidence: high
    free_kind: trial
    modalities: [language_models, image, video, audio]
    api:
      compatibility: wavespeed_native_rest
      authentication: API key
    eligibility:
      account_required: true
      payment_method_required: false
    limits:
      signup_credit_usd: 1
      recurrence: false
      expiry: not_published
      caveat: Some premium models are unavailable to trial balances; eligible catalog entries have positive prices and draw down the credit.
    sources:
      - {type: official_pricing, title: Pricing and signup credit, url: https://wavespeed.ai/pricing}

conditional_or_needs_verification:
  - id: anyrouter
    name: AnyRouter
    status: conditional
    confidence: high
    api_base_url: https://anyrouter.dev/api/v1
    model_id: anyrouter/free
    zero_token_price: true
    unlock_requirement:
      - Pay $1/month for the Go plan.
      - Or donate a working upstream provider API key, which unlocks Go without a card.
    limits:
      requests_per_day: 1000
      daily_reset: "00:00 UTC"
      go_requests_per_minute: 60
      go_rolling_five_hour_requests: 3000
    current_router_models: [cohere/north-mini-code, deepseek/DeepSeek-V3.1]
    reason_not_headline_free: Requires money or contribution of an upstream credential/capacity.
    sources:
      - {type: official_product, title: Free model, url: https://anyrouter.dev/free}
      - {type: official_model_page, title: anyrouter/free, url: https://anyrouter.dev/model/anyrouter/free}
      - {type: official_docs, title: Rate limits, url: https://docs.anyrouter.dev/features/rate-limits}

  - id: nexusrouter
    name: NexusRouter
    status: needs_authenticated_verification
    confidence: low
    api_base_url: https://api.nexusrouter.net/v1
    compatibility: [openai_compatible, anthropic_compatible]
    public_claims:
      fair_use_tokens_per_day: 1000000
      requests_per_minute: 10
      reset: "07:00 Asia/Jakarta"
      free_model_included: true
    conflict: The current client also contains “Free — 100K tokens/month,” and no exact free model is publicly named.
    recommendation: Do not include in headline counts until the authenticated dashboard and models endpoint resolve the plan and model.
    sources:
      - {type: official_product, title: NexusRouter, url: https://nexusrouter.net/}
      - {type: official_terms, title: Fair use, url: https://nexusrouter.net/fair-use}
      - {type: official_pricing, title: Pricing, url: https://nexusrouter.net/pricing}

  - id: siliconflow
    name: SiliconFlow / SiliconCloud
    status: conflicting_current_sources
    confidence: medium
    api_base_url: https://api.siliconflow.com/v1
    positive_evidence:
      - Generic docs say free models have fixed limits and paid variants use Pro/ prefixes.
      - Chinese docs say real-name-verified users can use models currently labeled free at zero cost.
    negative_evidence:
      - The current global public catalog gives positive prices to formerly cited DeepSeek-V3 and Qwen3-8B models.
      - The older exact free list contains dated Qwen2-era IDs.
      - No current exact zero-price ID was publicly verifiable without authentication.
    china_endpoint_note: The China service may retain a separate real-name-verified free catalog from the global service, but it could not be enumerated anonymously.
    recommendation: Require an authenticated model response or billing-free call before treating an ID as current.
    sources:
      - {type: official_docs, title: Global rate limits, url: https://docs.siliconflow.com/en/userguide/rate-limits/rate-limit-and-upgradation}
      - {type: official_docs, title: China rate limits, url: https://docs.siliconflow.cn/cn/userguide/rate-limits/rate-limit-and-upgradation}
      - {type: official_pricing, title: Current global pricing, url: https://www.siliconflow.com/pricing}
      - {type: official_api_reference, title: Models API, url: https://docs.siliconflow.cn/cn/api-reference/models/get-model-list}

  - id: freemodel_dev
    name: FreeModel.dev
    status: insufficiently_documented_trial
    confidence: low
    public_claim: Signup credits with no card.
    missing: Credit amount, recurrence, exact current catalog, and durable quota are not published clearly enough to validate ongoing free inference.
    source: {type: official_product, url: https://www.freemodel.dev/}

  - id: arouter
    name: ARouter
    status: insufficiently_documented
    confidence: medium
    api_base_url: https://api.arouter.ai/v1
    finding: Documentation describes generic :free routing and possible promotional credits, but exposes no exact public free model, stable quota, or guaranteed starter amount.
    source: {type: official_docs, url: https://docs.arouter.ai/en/faq}

  - id: baseten
    name: Baseten
    website: https://www.baseten.co/
    status: signup_credit_amount_unpublished
    confidence: high
    headline_ongoing_free: false
    free_kind: trial
    api:
      base_url: https://inference.baseten.co/v1
      compatibility: [openai_chat_completions, anthropic_messages]
      authentication: API key
    positive_evidence: Current pricing says new accounts receive credits to experiment for free.
    unresolved: [credit_amount, expiry, card_requirement, recurrence]
    models:
      coverage: Dynamic Model API catalog plus arbitrary self-deployed models; all published token prices are positive.
      representative_ids: [zai-org/GLM-5, deepseek-ai/DeepSeek-V4-Pro, moonshotai/Kimi-K2.6]
    sources:
      - {type: official_pricing, title: Pricing, url: https://www.baseten.co/pricing/}
      - {type: official_docs, title: Inference API overview, url: https://docs.baseten.co/reference/inference-api/overview}

  - id: nscale_serverless_inference
    name: Nscale Serverless Inference
    website: https://www.nscale.com/
    status: promotional_credits_only
    confidence: high
    headline_ongoing_free: false
    free_kind: trial
    api:
      compatibility: openai_compatible
      authentication: Service token
    current_requirement: Current overview says to add at least $5 credit before using the service.
    possible_exception: Current quickstart says early users may be eligible for unspecified promotional credits.
    history: The 2025 launch offered every new user $5, but the current docs no longer make that universal promise.
    recommendation: Do not count without an account-specific promotion.
    sources:
      - {type: official_docs, title: Current overview and minimum credit, url: https://docs.nscale.com/docs/getting-started/overview}
      - {type: official_docs, title: Current quickstart and possible promotions, url: https://docs.nscale.com/docs/getting-started/quickstart}
      - {type: historical_announcement, title: 2025 serverless launch, url: https://www.nscale.com/blog/introducing-nscale-serverless-inference-scalable-ai-without-infrastructure-hassles, published_at: "2025-04-02"}

  - id: bentoml_cloud
    name: BentoML Cloud / Bento Inference Platform
    website: https://www.bentoml.com/
    status: signup_credit_amount_unpublished
    confidence: medium_high
    headline_ongoing_free: false
    free_kind: trial
    positive_evidence: Current pricing promises one-time free compute credit.
    unresolved: [credit_amount, expiry]
    eligibility:
      payment_method_required_for_initial_trial: false
      payment_method_required_for_starter_upgrade: true
    api:
      base_url: deployment_specific
      compatibility: custom_rest_or_user_deployed_openai
    conflict: Older BentoML material mentioned $10, but current pricing no longer confirms that amount.
    sources:
      - {type: official_pricing, title: Pricing, url: https://www.bentoml.com/pricing}
      - {type: historical_product_post, title: Deployment example with older credit claim, url: https://www.bentoml.com/blog/deploying-a-text-to-speech-application-with-bentoml}

  - id: friendli_ai
    name: FriendliAI
    website: https://friendli.ai/
    status: signup_credit_amount_unpublished
    confidence: medium_high
    headline_ongoing_free: false
    free_kind: trial
    api:
      base_url: https://api.friendli.ai
      compatibility: [friendli_rest, openai_compatible]
      authentication: API key
    positive_evidence: Current billing docs confirm promotional signup credit.
    unresolved: [current_amount, expiry, card_requirement]
    exhausted_behavior: Purchased credit with a $10 minimum is required after the promotion.
    sources:
      - {type: official_docs, title: Credits, url: https://friendli.ai/docs/guides/suite/credits}
      - {type: official_pricing, title: Pricing, url: https://friendli.ai/pricing}
      - {type: official_api_reference, title: API introduction, url: https://friendli.ai/docs/openapi/introduction}

  - id: akashml
    name: AkashML
    website: https://akashml.com/
    status: signup_credit_amount_unpublished
    confidence: medium_high
    headline_ongoing_free: false
    free_kind: trial
    api:
      base_url: https://api.akashml.com/v1
      compatibility: [openai_compatible, anthropic_compatible]
      authentication: API key
    positive_evidence: Current FAQ says new users automatically receive free credits.
    unresolved: [credit_amount, expiry, card_requirement, recurrence, exact_trial_model_ids]
    models:
      representative_current_names: [QWEN3.8 27B, DeepSeek V4 Flash 0731, QWEN3.6 35B A3B, Llama 3.3 70B, GPT OSS 120B, GPT OSS 20B]
      caveat: All public model prices are positive and the models endpoint requires authentication.
    sources:
      - {type: official_product, title: AkashML and FAQ, url: https://akashml.com/}
      - {type: official_docs, title: Documentation, url: https://akashml.com/docs}

  - id: eden_ai
    name: Eden AI
    website: https://www.edenai.co/
    status: trial_plus_zero_price_routes_need_runtime_verification
    confidence: medium_high
    headline_ongoing_free: false
    free_kind: trial
    api:
      base_url: https://api.edenai.run/v3
      compatibility: [edenai_unified, openai_compatible_llm]
      authentication: Bearer API token
    trial:
      signup_credit_usd: 10
      recurrence: false
      unresolved: [expiry, payment_method_requirement]
    models:
      snapshot_at: "2026-08-21"
      verification: live_catalog
      discovery_url: https://api.edenai.run/v3/models
      live_total_models: 928
      zero_price_entries: [cloudflare/@cf/google/gemma-2b-it-lora, cloudflare/@cf/google/gemma-7b-it-lora, cloudflare/@cf/meta-llama/llama-2-7b-chat-hf-lora, cloudflare/@cf/mistral/mistral-7b-instruct-v0.2-lora, google/gemma-4-26b-a4b-it, google/gemma-4-31b-it]
      caveat: The catalog reports zero prices, but billing docs do not confirm those routes work at a zero wallet balance; sandbox responses are simulated and do not qualify.
    sources:
      - {type: official_docs, title: Eden AI overview, url: https://www.edenai.co/docs/index.md}
      - {type: official_docs, title: LLM quickstart, url: https://www.edenai.co/docs/v3/quickstart/first-llm-call.md}
      - {type: official_pricing, title: Pricing and signup credit, url: https://www.edenai.co/pricing}
      - {type: live_catalog, title: Models API, url: https://api.edenai.run/v3/models}
      - {type: official_docs, title: Sandbox is simulated, url: https://www.edenai.co/docs/v3/general/sandbox.md}

  - id: sambanova_cloud
    name: SambaNova Cloud
    website: https://cloud.sambanova.ai/
    status: conflicting_current_sources
    confidence: medium_high
    headline_ongoing_free: conditional
    free_kind: ongoing_free
    api:
      base_url: https://api.sambanova.ai/v1
      compatibility: openai_compatible
      authentication: API key
    positive_evidence:
      - Current rate-limit docs define a no-payment-method Free Tier.
      - Official staff posts in 2026 describe accounts without a card as Free Tier accounts.
    negative_evidence:
      - The current plans page says Free users must add a payment method and purchase credits before the first request.
    advertised_free_limits:
      production_model_ids: [DeepSeek-V3.1, Meta-Llama-3.3-70B-Instruct, gpt-oss-120b]
      preview_evaluation_only_ids: [DeepSeek-V3.2, gemma-4-31B-it]
      per_model: {requests_per_minute: 20, requests_per_day: 20, tokens_per_day: 200000}
    recommendation: Require a zero-balance authenticated generation before counting this as an ongoing tier.
    sources:
      - {type: official_docs, title: Model rate limits and Free Tier, url: https://docs.sambanova.ai/docs/en/models/rate-limits}
      - {type: official_plans, title: Contradictory plans page, url: https://cloud.sambanova.ai/plans}
      - {type: official_community_staff, title: No-card Free Tier confirmation, url: https://community.sambanova.ai/t/minimax-2-5-is-now-available-on-the-sambacloud-developer-tier/1601/10, published_at: "2026-03-18"}

  - id: fal_ai_builder_grant
    name: fal Builder Grant
    website: https://fal.ai/builder-grant
    status: application_required
    confidence: high
    headline_ongoing_free: false
    free_kind: selective_program
    eligibility:
      regions: [Europe, Asia]
      intended_use: Direct generative-media applications
      approval_guaranteed: false
      application_frequency: once_per_team_per_3_months
    grant_packs:
      - {name: Starter, credit_usd: 25}
      - {name: Plus, credit_usd: 100, unlock_monthly_usage_usd: 50}
      - {name: Launch, credit_usd: 250, unlock_monthly_usage_usd: 100}
    caveat: Sandbox coupons and free credits explicitly cannot be used through the API or Workflows; only an approved Builder Grant funds API usage.
    sources:
      - {type: official_program, title: Builder Grant, url: https://fal.ai/builder-grant}
      - {type: official_docs, title: Sandbox limitations, url: https://fal.ai/docs/documentation/model-apis/sandbox}
      - {type: official_pricing, title: API pricing, url: https://fal.ai/docs/documentation/model-apis/pricing}

  - id: openai_data_sharing_tokens
    name: OpenAI API complimentary data-sharing tokens
    website: https://platform.openai.com/
    status: conditional_eligibility_and_positive_balance_required
    confidence: high
    headline_ongoing_free: conditional
    free_kind: contribution_based
    eligibility:
      organization_must_be_selected_by_openai: true
      input_output_data_sharing_opt_in_required: true
      positive_account_balance_required: true
      unavailable_to: [Enterprise, Zero Data Retention organizations]
    limits:
      reset: "00:00 UTC daily"
      tiers_1_and_2: {large_model_group_tokens_per_day: 250000, small_model_group_tokens_per_day: 2500000}
      tiers_3_to_5: {large_model_group_tokens_per_day: 1000000, small_model_group_tokens_per_day: 10000000}
      caveat: A request that would cross the quota is billed in full; fine-tuning, evals, and tool use are excluded.
    models:
      representative_large_group: [gpt-5.5-2026-04-23, gpt-5.4-2026-03-05, gpt-4.1-2025-04-14, o3-2025-04-16]
      representative_small_group: [gpt-5.4-mini-2026-03-17, gpt-5.4-nano-2026-03-17, gpt-4.1-mini-2025-04-14, o4-mini-2025-04-16]
    sources:
      - {type: official_help, title: Complimentary tokens for shared API traffic, url: https://help.openai.com/en/articles/10306912-sharing-feedback-and-api-inputs-and-outputs-with-openai}
      - {type: official_docs, title: API data controls, url: https://platform.openai.com/docs/models/default-usage-policies-by-endpoint}

  - id: llm_gateway
    name: LLM Gateway
    website: https://llmgateway.io/
    status: conflicting_current_sources
    confidence: medium
    api_base_url: https://api.llmgateway.io/v1
    public_claim: Pricing says three zero-cost models at 20 requests per minute with no card.
    conflict:
      - Current models page renders zero Free Models.
      - The unauthenticated models endpoint returned 256 models with none marked free.
    recommendation: Do not count until an authenticated catalog or generation confirms a free route.
    sources:
      - {type: official_pricing, url: https://llmgateway.io/pricing}
      - {type: official_catalog, url: https://llmgateway.io/models}
      - {type: live_catalog, url: https://api.llmgateway.io/v1/models}

  - id: baidu_qianfan
    name: Baidu Qianfan
    status: current_trial_terms_incomplete
    confidence: medium
    geography: China
    public_claim: New customers can receive more than one million trial tokens.
    unresolved: [exact_models, per_model_quota, duration, payment_requirement]
    historical_conflict: A 2024 notice promised long-term-free ERNIE Speed/Lite/Tiny routes, but current pricing shows positive prices.
    sources:
      - {type: official_product, url: https://cloud.baidu.com/product/qianfan.html}
      - {type: official_docs, url: https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya}
      - {type: official_notice, url: https://cloud.baidu.com/news/notice_c7a145b9-5c99-4870-939f-e09ba506aab3}

  - id: tera_promotional_credits
    name: Tera
    website: https://www.tera.gw/
    status: application_required
    confidence: high
    headline_ongoing_free: false
    free_kind: selective_program
    api_base_url: https://api.tera.gw/v1
    eligibility:
      application_required: true
      founder_call_required: true
      grants_per_company: 1
    credits:
      solo_developers_usd: 150
      vc_backed_startups_usd: 250
      expires_after_days: 45
    models:
      representative_ids: [gpt-oss-20b, gpt-oss-120b, DeepSeek-V4-Flash, DeepSeek-V3.2, Qwen3-Coder-480B, kimi-k2-thinking, GLM-5, Kimi-K2.6]
    sources:
      - {type: official_program, title: Credits, url: https://www.tera.gw/credits}
      - {type: official_pricing, title: Pricing, url: https://www.tera.gw/pricing}

  - id: runpod_startup_program
    name: Runpod startup credits
    website: https://www.runpod.io/startup-program
    status: application_required
    confidence: high
    headline_ongoing_free: false
    free_kind: selective_program
    offer:
      startup_credit_usd: 1000
      approval_required: true
    caveat: Normal serverless/public-endpoint inference is prepaid and requires a positive balance; this is not an automatic free tier.
    sources:
      - {type: official_program, title: Startup program, url: https://www.runpod.io/startup-program}
      - {type: official_docs, title: Billing, url: https://docs.runpod.io/accounts-billing/billing}

  - id: jina_ai_llm_serp
    name: Jina LLM-as-SERP API
    status: advertised_free_but_callability_unresolved
    confidence: medium
    api_base_url: https://llm-serp.jina.ai
    positive_evidence: Current product page says the API is free and optional keys raise limits without being charged.
    negative_evidence: Two anonymous calls returned HTTP 200 with empty results and zero token usage; no numeric public limit is stated.
    recommendation: Require a non-empty successful response before moving to current providers.
    sources:
      - {type: official_product, url: https://jina.ai/api-dashboard/llm-serp/}
      - {type: company_press_release, url: https://jina.ai/news/llm-as-serp-search-engine-result-pages-from-large-language-models/, published_at: "2025-02-27"}

  - id: neurlap
    name: Neurlap
    website: https://neurlap.ai/
    status: early_access_waitlist
    confidence: low
    free_kind: contributor_credits
    condition: Users share GPU capacity to earn credits.
    missing: [public_signup, stable_catalog, quotas, complete_terms]
    source: {type: official_product, url: https://neurlap.ai/}

historical_or_excluded:
  - id: aimlapi
    name: AI/ML API
    status: free_tier_paused
    confidence: high
    reason: The authoritative current FAQ says the Free Tier is paused; live aliases containing :free do not override that statement.
    source: {type: official_docs, title: Free Tier FAQ, url: https://docs.aimlapi.com/faq/free-tier.md}

  - id: runpod
    name: Runpod Serverless and Public Endpoints
    status: prepaid_only
    confidence: high
    reason: Normal inference requires a positive balance; selective startup credits are recorded separately as conditional.
    sources:
      - {type: official_docs, title: Billing, url: https://docs.runpod.io/accounts-billing/billing}
      - {type: official_docs, title: Public endpoints quickstart, url: https://docs.runpod.io/public-endpoints/quickstart}

  - id: digitalocean_gradient_inference
    name: DigitalOcean Gradient serverless inference
    status: prepaid_only
    confidence: high
    reason: A separate prepaid inference balance is mandatory; the free router preview does not make routed model inference free.
    sources:
      - {type: official_pricing, url: https://docs.digitalocean.com/products/gradient-platform/details/pricing/}
      - {type: official_api_reference, url: https://docs.digitalocean.com/reference/api/reference/serverless-inference/}

  - id: fal_ai_general_api
    name: fal Model APIs
    status: prepaid_only
    confidence: high
    reason: Sandbox coupons cannot fund API or Workflow calls; only the application-based Builder Grant is potentially free.
    sources:
      - {type: official_pricing, url: https://fal.ai/docs/documentation/model-apis/pricing}
      - {type: official_docs, url: https://fal.ai/docs/documentation/model-apis/sandbox}

  - id: featherless_ai
    name: Featherless AI
    status: paid_subscription_only
    confidence: high
    reason: The lowest current plan is paid and monthly credits belong to paid subscriptions.
    sources:
      - {type: official_plans, url: https://featherless.ai/docs/plans}
      - {type: official_pricing, url: https://featherless.ai/docs/request-pricing-and-credits}

  - id: inference_net
    name: Inference.net
    status: free_gateway_not_free_inference
    confidence: high
    reason: The free plan covers gateway requests and tracing; all live model entries have positive inference prices.
    sources:
      - {type: official_pricing, url: https://inference.net/pricing/}
      - {type: live_catalog, url: https://api.inference.net/v1/models}

  - id: venice_api
    name: Venice API
    status: free_web_chat_but_paid_api
    confidence: high
    reason: API calls require spendable DIEM, bundled, or USD balance; the free web-chat plan does not fund API inference.
    sources:
      - {type: official_pricing, url: https://venice.ai/pricing}
      - {type: official_docs, title: Generating an API key, url: https://docs.venice.ai/guides/getting-started/generating-api-key.md}

  - id: infermatic
    name: Infermatic
    status: free_ui_only
    confidence: high
    reason: The $0 plan explicitly has no API access; API access starts on a paid plan.
    source: {type: official_pricing, url: https://infermatic.ai/pricing/}

  - id: modular_model_api
    name: Modular shared Model API
    status: paid_hosted_api
    confidence: high
    reason: Free-forever language refers to self-hosted MAX; hosted shared endpoints have positive token prices.
    source: {type: official_pricing, url: https://www.modular.com/pricing}

  - id: lambda_inference_api
    name: Lambda hosted Inference API
    status: winding_down_paid_service
    confidence: high
    reason: The hosted API is winding down and the current alternative is paid GPU instances.
    source: {type: official_product, url: https://lambda.ai/inference}

  - id: avian_api
    name: Avian API
    status: prepaid_only
    confidence: high
    reason: Current models have positive prices and credits must be prepaid; Start Free is signup wording, not a compute allowance.
    source: {type: official_pricing, url: https://api.avian.io/pricing/}

  - id: nextbit
    name: NextBit
    status: prepaid_only
    confidence: medium_high
    reason: Current docs and terms describe prepaid positive-price API usage with no quantified automatic free allowance.
    source: {type: official_docs, url: https://www.nextbit256.com/docs}

  - id: darkbloom
    name: Darkbloom
    website: https://www.darkbloom.dev/
    status: paid_public_alpha
    confidence: medium_high
    reason: Public alpha access is evaluation-only, but every displayed model has a positive token price.
    sources:
      - {type: official_product, title: Darkbloom, url: https://www.darkbloom.dev/}

  - id: wafer
    name: Wafer
    status: paid_subscription_or_inference
    confidence: high
    reason: Current terms describe paid Wafer Pass and inference; third-party claims of a free route lack first-party support.
    source: {type: official_terms, url: https://www.wafer.ai/terms}

  - id: liquid_ai_hosted_api
    name: Liquid AI hosted API
    status: self_hosted_weights_only
    confidence: high
    reason: Current free offer covers downloadable/on-device models and tooling, not operator-funded remote inference.
    sources:
      - {type: official_pricing, url: https://www.liquid.ai/pricing}
      - {type: official_pricing, title: LEAP, url: https://leap.liquid.ai/pricing}

  - id: petals_public_chat
    name: Petals public chat endpoint
    status: currently_nonfunctional
    confidence: high
    reason: A live call returned MissingBlocksError because no peers held the required model blocks.
    source: {type: official_repository, url: https://github.com/petals-infra/chat.petals.dev}

  - id: webinfer_resource_pool
    name: WebInfer resource pool
    status: insufficiently_documented
    confidence: low
    reason: No stable gateway base URL, quota, catalog endpoint, signup flow, or complete terms are published.
    sources:
      - {type: official_product, url: https://webllm.org/providers/resource-pool}
      - {type: official_product, url: https://webinfer.com/providers}

  - id: cscs_inference
    name: CSCS inference service
    status: project_accounted_not_free
    confidence: high
    reason: Non-SwissAI use consumes project credits converted from allocated node-hours at a published CHF rate.
    source: {type: official_docs, url: https://docs.cscs.ch/services/inference/api/}

  - id: isambard_ai_inference
    name: Isambard-AI inference service
    status: not_yet_available
    confidence: high
    reason: Research allocations exist, but the July 2026 update says the shared inference service is still being prepared.
    sources:
      - {type: official_announcement, url: https://www.bristol.ac.uk/research/centres/bristol-supercomputing/articles/2026/one-year-of-isambard-ai.html}

  - id: alia_spain
    name: ALIA Spain
    status: open_models_no_public_api
    confidence: high
    reason: Publishes open models and resources, but no stable public developer inference endpoint with current auth and quotas.
    source: {type: official_product, url: https://alia.gob.es/eng}

  - id: falcon_tii
    name: Falcon / Technology Innovation Institute
    status: weights_and_playground_only
    confidence: high
    reason: Free weights and a playground are available, but no stable documented public inference API with quotas and auth.
    source: {type: official_product, url: https://falconllm.tii.ae/index.html}

  - id: ai2_playground
    name: Ai2 Playground
    status: web_playground_only
    confidence: high
    reason: Ai2 directs API users to external paid routes; its playground is not a documented general developer API.
    source: {type: official_docs, url: https://docs.allenai.org/quick_start/apis}

  - id: llm_jp
    name: LLM-jp
    status: weights_and_tooling_only
    confidence: high
    reason: Publishes weights, corpora, and tooling rather than a stable hosted inference API.
    source: {type: official_product, url: https://llm-jp.nii.ac.jp/release/}

  - id: opengradient
    name: OpenGradient
    status: paid_x402_inference
    confidence: high
    reason: Hosted inference uses x402 or OPG payment rather than a free tier.
    source: {type: official_docs, url: https://docs.opengradient.ai/about/}

  - id: swarmllm
    name: SwarmLLM
    status: self_hosted_only
    confidence: high
    reason: Peer-to-peer software without an operator-hosted public endpoint.
    source: {type: official_repository, url: https://github.com/enapt/SwarmLLM}

  - id: freellmapi_co
    name: FreeLLMAPI.co
    status: self_hosted_byok_router
    confidence: high
    reason: Users self-host the router and supply upstream provider keys.
    source: {type: official_product, url: https://freellmapi.co/}

  - id: krutrim_cloud
    name: Krutrim Cloud AI Studio
    status: prepaid_only
    confidence: high
    reason: Current AI Studio requires purchased credits and has no verified automatic free allowance.
    source: {type: official_docs, url: https://docs.cloud.olakrutrim.com/basics/ai-studio/billing-for-ai-studio}

  - id: duckk_africa
    name: Duckk Africa
    status: waitlist_without_verified_free_terms
    confidence: low
    reason: OpenAI-compatible marketing and a waitlist exist, but no validated free quota or public production endpoint.
    source: {type: official_product, url: https://www.duckk.org/}

  - id: national_university_of_singapore
    name: National University of Singapore generic AI services (unrelated name match)
    status: not_a_public_general_inference_api
    requested_name_resolution: The user confirmed the requested provider was Nous Research; this record remains only to document the unrelated name-match check.
    current_findings:
      - NUS ChatGPT Edu is institution-restricted to students, faculty, and staff beginning 2026-08-31.
      - AI Sense Maker is an NUS web application, not a general public inference API.
    sources:
      - {type: official_announcement, title: NUS and OpenAI strategic collaboration, url: https://news.nus.edu.sg/nus-powers-education-research-and-administration-to-new-heights-with-ai-through-a-strategic-collaboration-with-openai/, published_at: "2026-08-11"}
      - {type: official_research_site, title: NUS AI Institute, url: https://ai.nus.edu.sg/research/}

  - id: github_models
    name: GitHub Models
    status: retired
    confidence: high
    retired_on: "2026-07-30"
    retired_components: [playground, model_catalog, inference_api, byok]
    historical_api_base_url: https://models.github.ai/inference
    history:
      - {date: "2024-10-29", event: Public preview with rate-limited free model inference}
      - {date: "2025-06-24", event: Paid usage beyond free limits added}
      - {date: "2026-06-16", event: Closed to new customers}
      - {date: "2026-07-30", event: Fully retired}
    sources:
      - {type: official_retirement, title: Full retirement announcement, url: https://github.blog/changelog/2026-07-01-github-models-is-being-fully-retired-on-july-30-2026/}
      - {type: official_announcement, title: Public preview, url: https://github.blog/changelog/2024-10-29-github-models-is-now-available-in-public-preview/}

  - id: chutes
    name: Chutes
    status: paid_only
    confidence: high
    api_base_url: https://llm.chutes.ai/v1
    historical_offer:
      requests_per_day: 200
      tee_access_removed: "2026-02-27"
      non_tee_free_access_ended: "2026-03-15"
      replacement: One month of Base or $5 credit.
    sustainability_explanation: Chutes reported removing about 20B daily tokens of sponsored OpenRouter traffic and 10B daily tokens from its direct free quota; estimated direct cost was about $6/user/month across roughly 11,000 users.
    sources:
      - {type: official_pricing, title: Current pricing, url: https://chutes.ai/pricing}
      - {type: official_announcement, title: February community announcement, url: https://chutes.ai/news/community-announcement-february}
      - {type: official_postmortem, title: Building a sustainable inference platform, url: https://chutes.ai/news/from-volume-to-value-building-a-sustainable-ai-inference-platform-2}

  - id: together_ai
    name: Together AI
    status: prepaid_only
    confidence: high
    reason: Official billing docs say there are no free trials; API access requires at least a $5 credit purchase.
    sources:
      - {type: official_docs, title: Billing and credits, url: https://docs.together.ai/docs/billing-credits}
      - {type: official_pricing, title: Inference pricing, url: https://docs.together.ai/docs/inference/pricing}

  - id: deepinfra
    name: DeepInfra
    status: paid_only
    confidence: medium_high
    reason: The current public catalog has positive per-token or per-execution prices and no permanent recurring free quota.
    sources:
      - {type: official_pricing, url: https://deepinfra.com/pricing}
      - {type: official_api_reference, url: https://docs.deepinfra.com/api-reference/introduction}

  - id: qwen_code_oauth
    name: Qwen OAuth free access for Qwen Code
    status: retired
    retired_on: "2026-04-15"
    note: Model Studio/Qwen Cloud new-user quotas remain separately recorded as a current trial.
    source: {type: official_docs, url: https://qwenlm.github.io/qwen-code-docs/en/users/configuration/auth/}

  - id: public_ai_old_blanket_free
    name: Public AI old blanket-free offer
    status: superseded
    old_claim_date: "2025-09-17"
    old_claim: Free of charge at the time of writing.
    current_reality: Starter credit followed by positive per-token wallet billing.
    source: {type: historical_announcement, url: https://huggingface.co/blog/inference-providers-publicai}

  - id: huggingface_spaces_as_a_class
    name: Hugging Face Spaces / Gradio demos
    status: excluded_provider_class
    reason: Many Spaces expose callable Gradio endpoints, but there is no shared quota, catalog guarantee, authentication policy, or availability commitment; individual Spaces may sleep or disappear.
    inclusion_rule: Include a Space only when its owner documents a public API, quota, access policy, and continuing availability.

  - id: self_hosted_open_weights
    name: Self-hosted open-weight models
    status: out_of_scope
    reason: The model weights may be free, but remote hosted compute is not being provided.

privacy_classification_audit:
  as_of: "2026-08-22"
  checked_at: "2026-08-22T09:00:00-05:00"
  timezone: America/Chicago
  methodology:
    private: Strong plan-specific evidence excludes ordinary prompt/output retention, training, and content-based improvement; narrowly scoped abuse or transient-processing exceptions may remain.
    partially_private: Some protections exist, but retention, operator access, optional features, route-specific upstream terms, or missing documentation prevents a strong privacy claim.
    not_private: Default free-tier terms permit prompt/output retention, model training, content-based product improvement, publication, or similarly broad secondary use.
  interpretation: The badge is a conservative service-level summary, not a security certification. Read the full data-governance profile because model routes and account settings can change the result.
  records:
    openrouter: {classification: partially_private, reason: "OpenRouter defaults are protective, but the selected upstream model provider and optional logging or data-sharing controls determine end-to-end handling."}
    nous_portal: {classification: partially_private, reason: "No usable public policy was found for prompt retention, training, operator access, or deletion."}
    orcarouter: {classification: partially_private, reason: "OrcaRouter says it does not retain or train on content, but independently governed upstream providers receive the requests."}
    nvidia_build: {classification: not_private, reason: "The free developer terms permit storage and broad product, service, and underlying-model improvement uses."}
    hetzner_experiments: {classification: private, reason: "Hetzner documents no ordinary prompt/output retention, no content training right, and EU-hosted processing."}
    groqcloud: {classification: private, reason: "Inputs and outputs are not retained by default or used for training without permission; limited abuse and feature-specific retention exceptions remain."}
    google_gemini_api: {classification: not_private, reason: "Unpaid-service inputs and outputs may be retained and used to improve and train Google models."}
    mistral: {classification: not_private, reason: "Free-mode inputs and outputs are eligible for model improvement and training by default unless the user opts out."}
    huggingface_inference_providers: {classification: partially_private, reason: "Hugging Face does not store routed bodies, but each selected upstream provider has its own independent data policy."}
    cloudflare_workers_ai: {classification: private, reason: "Workers AI excludes Customer Content from storage, training, and improvement unless the customer deliberately enables a feature or consents."}
    cohere: {classification: partially_private, reason: "Content can be logged for 30 days and reviewed for abuse, while training is opt-in and aggregate improvement uses remain."}
    vercel_ai_gateway: {classification: partially_private, reason: "The gateway deletes content after each request, but upstream retention and non-enterprise protections depend on route controls and provider terms."}
    ibm_watsonx_ai_runtime: {classification: private, reason: "IBM says unsaved API prompts and outputs are not accessed, logged, stored, trained on, or used for improvement without permission."}
    opencode_zen: {classification: not_private, reason: "Named free routes may collect prompts and completions for model improvement or training, with route-specific retention."}
    zai: {classification: private, reason: "The API DPA describes real-time processing without storage and requires explicit agreement before content-based development or improvement."}
    llmapi_ai: {classification: partially_private, reason: "The gateway defaults to zero retention and no training, but upstream providers are independently governed and optional logging changes retention."}
    api_airforce: {classification: partially_private, reason: "Content can be retained for safety without a fixed TTL, while training and upstream handling are not fully documented."}
    llm7: {classification: partially_private, reason: "No usable public policy was found for request-content handling."}
    modelscope_inference: {classification: partially_private, reason: "API-specific retention and training protections are not documented clearly enough for a stronger classification."}
    awanllm: {classification: private, reason: "AwanLLM says prompts and generations are not logged and discloses no content-based training or improvement use."}
    arliai: {classification: private, reason: "Arli AI documents transient-only processing with no prompt/output storage, training, or content-based improvement."}
    freeinference_org: {classification: not_private, reason: "Prompts and responses may be logged and reused for service research, improvement, and published derived datasets."}
    fastrouter: {classification: not_private, reason: "The terms grant broad perpetual rights to store and use inputs and outputs for debugging and service improvement."}
    kilo_ai_gateway: {classification: not_private, reason: "The terms grant broad perpetual improvement rights and some routed models can require training permission."}
    scaleway_generative_apis: {classification: private, reason: "Scaleway excludes ordinary prompt collection and model training, hosts models on its own EU infrastructure, and limits incident retention."}
    qwen_cloud: {classification: partially_private, reason: "Alibaba excludes training without consent, but request data is stored regionally and third-party model terms can differ."}
    sea_lion_api: {classification: not_private, reason: "The terms authorize use of inputs and outputs to develop and improve the service without a clear no-training commitment."}
    ndif: {classification: partially_private, reason: "No endpoint-wide public prompt-retention, training, operator-access, or deletion policy was found."}
    ai_horde: {classification: partially_private, reason: "The coordinator is transient, but volunteer workers can inspect or retain plaintext requests outside a uniform contractual control."}
    pollinations: {classification: partially_private, reason: "Current free-route retention, training, and multi-path infrastructure handling are not documented precisely."}
    puter_js: {classification: partially_private, reason: "Puter and the selected upstream both process requests, while AI-specific TTL and training protections are unresolved."}
    public_ai: {classification: partially_private, reason: "No usable public inference-data agreement was found."}
    lightning_ai_model_apis: {classification: partially_private, reason: "Model-API retention, training, publisher access, and plan-specific protections are not documented fully."}
    modal: {classification: private, reason: "Modal does not store ordinary endpoint payloads or train on Customer Data without written consent."}
    beam_cloud: {classification: not_private, reason: "Beam permits storage and expressly permits use of customer data and queries to measure and improve the service."}
    sail_research: {classification: private, reason: "Sail limits persistent content storage to 48 hours and excludes content from training and service improvement."}
    cartesia: {classification: not_private, reason: "Default terms permit indefinite storage and perpetual use of inputs and outputs for labeling, improvement, promotion, and model training."}
    elevenlabs_api: {classification: not_private, reason: "Non-enterprise API data is retained and eligible for audio-model improvement by default unless the user opts out."}
    ovhcloud_ai_endpoints: {classification: partially_private, reason: "Endpoint-specific prompt retention and training commitments were not found."}
    requesty: {classification: not_private, reason: "Free self-service traffic is logged by default and can use upstream models that train on content."}
    inception_platform: {classification: partially_private, reason: "No usable public inference-data policy was found."}
    poolside_direct_api: {classification: partially_private, reason: "No usable public inference-data policy was found for the promotion."}
    voyage_ai: {classification: not_private, reason: "Standard terms permit perpetual storage, model training, and improvement unless an eligible organization opts out."}
    ai21_studio: {classification: partially_private, reason: "Default retention and training treatment are not clear; stronger traceless operation requires explicit configuration."}
    deepgram: {classification: partially_private, reason: "Per-request opt-out can exclude training, but participating accounts permit sampled retention and model improvement."}
    jina_ai_search_foundation: {classification: partially_private, reason: "Jina excludes model training on request content, but retention and broader support or improvement handling remain imprecise."}
    jina_ai_reader: {classification: partially_private, reason: "Jina excludes model training, but anonymous Reader retention and non-training content handling remain unclear."}
    mancer_ai: {classification: not_private, reason: "Default terms permit retention, third-party disclosure, model training, and service improvement unless the user opts out."}
    mara_inference_cloud: {classification: partially_private, reason: "No usable public inference-data policy was found."}
    assemblyai: {classification: not_private, reason: "Standard terms allow indefinite retention in some modes and permit training and broad improvement unless excluded or opted out."}
    speechmatics: {classification: partially_private, reason: "Training is opt-in, but batch content is retained for seven days and portal or support features add storage and access."}
    aws_bedrock: {classification: partially_private, reason: "Bedrock is zero-retention and no-training by default, but named model routes have current abuse-review retention exceptions."}
    azure_ai_foundry: {classification: partially_private, reason: "Base inference is stateless and no-training, but abuse review and stateful features can retain content for human access."}
    google_vertex_ai: {classification: partially_private, reason: "Google Cloud excludes training without permission, but caching, grounding, and session features introduce bounded retention."}
    oracle_oci_generative_ai: {classification: private, reason: "Ordinary inference is not stored or used for general model or service improvement; persistence is customer-selected."}
    replicate: {classification: partially_private, reason: "API data is deleted after one hour by default, but community models, resultant data, and web history introduce additional handling."}
    cerebras_inference: {classification: private, reason: "Cerebras states inference inputs and outputs are not retained or used for content-based training or improvement."}
    clarifai: {classification: not_private, reason: "Inputs and predictions are stored by default and the general terms retain broad service-development rights."}
    fireworks_ai: {classification: partially_private, reason: "Open-model requests are transient and no-training by default, but stored-response and proprietary partner routes can differ."}
    nebius_token_factory: {classification: partially_private, reason: "Training is excluded, but prompts and outputs may be stored by default unless zero-data-retention mode is enabled."}
    novita_ai: {classification: private, reason: "Default inference content is transient and excluded from training and service improvement, subject to narrow legal and support exceptions."}
    hyperbolic: {classification: partially_private, reason: "The AI FAQ promises transient no-training inference, but feedback storage, node operators, and a broader content license remain."}
    waterfall: {classification: partially_private, reason: "No usable public inference-data policy was found."}
    logfare: {classification: not_private, reason: "Every request body is logged and scrubbed content may be used in internal evaluation datasets; premium access can require training opt-in."}
    bazaarlink: {classification: partially_private, reason: "Prompt TTL, route-specific training, product-improvement scope, and upstream handling remain unresolved."}
    dreamprompting: {classification: not_private, reason: "Prompts can be stored or publicly shared and user content may be used to improve the service without a gateway-specific no-training promise."}
    ch_at: {classification: partially_private, reason: "The service claims no logs, but the current upstream model and its data handling are not fully documented."}
    opentyphoon: {classification: partially_private, reason: "No public service agreement defining inference retention, training, access, or deletion was found."}
    alcf_inference_endpoints: {classification: partially_private, reason: "Institutional controls apply, but endpoint-wide content retention, training, administrator access, and deletion terms are not public."}
    fikra_api: {classification: private, reason: "Fikra states prompts and responses are processed in memory, immediately discarded, excluded from logs, and never used for training."}
    sarvam_ai: {classification: not_private, reason: "Free developer inputs and outputs are collected and may support service improvement without a clear default no-training commitment."}
    byteplus_modelark: {classification: partially_private, reason: "Campaign terms establish free tokens but do not resolve content TTL, training, improvement, or model-specific routing."}
    tencent_hunyuan: {classification: partially_private, reason: "Hunyuan free-tier retention, training, improvement, and operator-access commitments were not verified."}
    upstage: {classification: not_private, reason: "Current free-service terms expressly allow storage, service improvement, AI research, and model training with inputs and outputs."}
    maritaca_academic_credits: {classification: partially_private, reason: "No public academic-API agreement defining content retention, training, access, or deletion was found."}
    wavespeedai: {classification: partially_private, reason: "A no-training statement exists, but prompt/output TTL, deletion, and third-party model handling remain unresolved."}

popularity_audit:
  as_of: "2026-08-22"
  checked_at: "2026-08-22T21:30:00-05:00"
  timezone: America/Chicago
  methodology:
    score_range: [0, 100]
    recommendation_weight: score_over_6_to_at_most_17_points
    missing_behavior: not_audited is neutral; not_found applies a small confidence discount because a completed audit found no traceable adoption.
    tiers: {ubiquitous: 80_to_100, well_known: 60_to_79, established: 20_to_59, niche: 0_to_19, unknown: unscored_only}
    components:
      adoption: "0-40: 40 at 10M+ documented users; 35 at 5M+; 30 at 1M+; 20 at 100K+; 10 for smaller numeric adoption."
      developer_ecosystem: "0-30: 30 at 1M+ documented dependent packages/apps/downloads; 25 at 100K+; 20 at 10K+; 10 at 1K+."
      independent_awareness: "0-20 from independent surveys or reproducible search-interest measurements."
      durable_recognition: "0-10 for an established global developer platform or recognized first-party model family."
  records:
    openrouter:
      status: scored
      score: 73
      tier: well_known
      basis: OpenRouter reports 10M+ global users, 250K+ apps, and 200T+ monthly tokens.
      components: {adoption: 40, developer_ecosystem: 25, independent_awareness: 0, durable_recognition: 8}
      signals:
        - {metric: global_users, value: 10000000, observed_at: "2026-08-22", source_url: https://openrouter.ai/}
        - {metric: apps, value: 250000, observed_at: "2026-08-22", source_url: https://openrouter.ai/}
      sources:
        - {type: official_product, title: OpenRouter usage and ecosystem metrics, url: https://openrouter.ai/}
    nous_portal:
      status: partial
      score: 36
      tier: established
      basis: Nous Research's verified Hugging Face organization records 1,982,041 model downloads in the last 30 days across 126 models, and the first-party Hermes Agent repository that documents Nous Portal integration holds 234,456 GitHub stars; Nous stewards the widely recognized first-party Hermes open-model family. Partial because no Nous Portal user, developer, or request-volume count is published and no independent survey or search-interest measure exists, so only the developer-ecosystem and durable-recognition components are scorable.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 6
      signals:
      - metric: hf_org_model_downloads_30d
        value: 1982041
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/api/models?author=NousResearch
      - metric: hf_org_followers
        value: 4431
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/NousResearch
      - metric: github_stars_official_repo
        value: 234456
        observed_at: '2026-08-23'
        source_url: https://github.com/NousResearch/hermes-agent
      - metric: github_org_followers
        value: 8224
        observed_at: '2026-08-23'
        source_url: https://github.com/NousResearch
      sources:
      - type: official_repository
        title: Nous Research organization on Hugging Face (Hermes model family)
        url: https://huggingface.co/NousResearch
      - type: official_repository
        title: Hermes Agent repository documenting the Nous Portal integration
        url: https://github.com/NousResearch/hermes-agent
      - type: code_hosting_stats
        title: Nous Research GitHub organization followers
        url: https://github.com/NousResearch
    orcarouter:
      status: partial
      score: 8
      tier: niche
      basis: The only citable rubric-eligible metric is the operator's own open-source router repository, Continuum-AI-Corp/OrcaRouter-Lite, at 583 GitHub stars, with 56 followers on the Continuum AI organization; both sit below the 1K+ developer-ecosystem anchor, so a sub-threshold 8 is scored. The May 2026 launch release and the OrcaRouter site publish no user, developer, customer, or request-volume figures, no official package exists on npm or PyPI, and no survey or search-interest measure covers the service.
      components:
        adoption: 0
        developer_ecosystem: 8
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: github_stars_official_repo
        value: 583
        observed_at: '2026-08-23'
        source_url: https://github.com/Continuum-AI-Corp/OrcaRouter-Lite
      - metric: github_org_followers
        value: 56
        observed_at: '2026-08-23'
        source_url: https://github.com/Continuum-AI-Corp
      sources:
      - type: official_repository
        title: OrcaRouter Lite open-source router by Continuum AI Corp.
        url: https://github.com/Continuum-AI-Corp/OrcaRouter-Lite
      - type: code_hosting_stats
        title: Continuum AI GitHub organization
        url: https://github.com/Continuum-AI-Corp
      - type: company_press_release
        title: OrcaRouter launches the open LLM API router (2026-05-08)
        url: https://www.prnewswire.com/news-releases/orcarouter-launches-the-open-llm-api-router--zero-markup-mit-licensed-100-models-302766356.html
    nvidia_build:
      status: partial
      score: 20
      tier: established
      basis: NVIDIA documented nearly 200 NIM technology integrations, and NVIDIA is a globally established developer platform; no NIM active-user count was captured.
      components: {adoption: 10, developer_ecosystem: 0, independent_awareness: 0, durable_recognition: 10}
      signals:
        - {metric: technology_partners, value: 200, observed_at: "2026-08-22", source_url: https://nvidianews.nvidia.com/news/nvidia-nim-model-deployment-generative-ai-developers}
      sources:
        - {type: official_press_release, title: NVIDIA NIM partner adoption, url: https://nvidianews.nvidia.com/news/nvidia-nim-model-deployment-generative-ai-developers}
    hetzner_experiments:
      status: scored
      score: 39
      tier: established
      basis: No Hetzner Inference API adoption count exists, so the service itself contributes no adoption points and Hetzner's company-wide customer base is not counted as service adoption; the operator's official hcloud Python SDK sees 260,947 monthly PyPI downloads, Hetzner held 5.0% share in the independent 2024 Stack Overflow cloud-platform survey, and Hetzner Online is a major European hosting and data-center operator running parks in Germany, Finland, Singapore and the United States. Scored on the operating company's developer standing, with the service-scope gap stated.
      components:
        adoption: 0
        developer_ecosystem: 25
        independent_awareness: 7
        durable_recognition: 7
      signals:
      - metric: pypi_hcloud_monthly_downloads
        value: 260947
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/hcloud/recent
      - metric: stackoverflow_2024_cloud_platform_share_percent
        value: 5.0
        observed_at: '2026-08-23'
        source_url: https://survey.stackoverflow.co/2024/technology
      - metric: github_org_followers_hetznercloud
        value: 830
        observed_at: '2026-08-23'
        source_url: https://github.com/hetznercloud
      sources:
      - type: package_registry_stats
        title: PyPI download stats for the official Hetzner Cloud python library
        url: https://pypistats.org/packages/hcloud
      - type: independent_survey
        title: 2024 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2024/technology
      - type: official_about
        title: About Hetzner Online
        url: https://www.hetzner.com/unternehmen/ueber-uns/
    groqcloud:
      status: scored
      score: 65
      tier: well_known
      basis: Groq officially reports serving more than five million developers, and the groq Python SDK sees 26.6M monthly PyPI downloads; no independent survey lists Groq and no durable-recognition claim is made.
      components:
        adoption: 35
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: groqcloud_developers
        value: 5000000
        qualifier: at_least
        observed_at: '2026-08-22'
        source_url: https://groq.com/newsroom/groq-raises-usd650m-to-scale-its-ai-inference-cloud-business
      - metric: pypi_groq_monthly_downloads
        value: 26574183
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/groq
      sources:
      - type: official_press_release
        title: Groq raises $650M and serves more than five million developers (2026-06-22)
        url: https://groq.com/newsroom/groq-raises-usd650m-to-scale-its-ai-inference-cloud-business
      - type: official_adoption
        title: 1 million developers on GroqCloud milestone (2025-03)
        url: https://groq.com/blog/thank-you-1m-developers-building-with-groqcloud
      - type: package_registry_stats
        title: PyPI download stats for groq
        url: https://pypistats.org/packages/groq
    google_gemini_api:
      status: scored
      score: 70
      tier: well_known
      basis: 'Google reported millions of Gemini API and AI Studio developers, the official google-genai SDK shows 293.9M monthly PyPI downloads, and Google operates an established global developer platform. Upgraded from partial: the SDK download metric adds an independently sourced developer-ecosystem component alongside the existing adoption signal.'
      components:
        adoption: 30
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 10
      signals:
      - metric: developers
        value: 1000000
        qualifier: at_least
        observed_at: '2026-08-22'
        source_url: https://developers.googleblog.com/en/looking-back-at-the-first-year-of-the-gemini-era/
      - metric: pypi_monthly_downloads_google_genai
        value: 293873058
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/api/packages/google-genai/recent
      sources:
      - type: official_adoption
        title: First year of the Gemini era (millions of developers)
        url: https://developers.googleblog.com/en/looking-back-at-the-first-year-of-the-gemini-era/
      - type: package_downloads
        title: pypistats recent downloads for the official google-genai package
        url: https://pypistats.org/api/packages/google-genai/recent
    mistral:
      status: scored
      score: 80
      tier: ubiquitous
      basis: Mistral confirmed 1M+ Le Chat downloads in 14 days, the mistralai PyPI package sees 47.6M monthly downloads, Mistral models hold 10.4% share in the independent 2025 Stack Overflow survey, and Mistral stewards a widely recognized first-party model family distributed on major clouds.
      components:
        adoption: 30
        developer_ecosystem: 30
        independent_awareness: 10
        durable_recognition: 10
      signals:
      - metric: le_chat_app_downloads_first_14_days
        value: 1000000
        qualifier: at_least
        observed_at: '2026-08-22'
        source_url: https://www.eweek.com/news/mistral-ai-le-chat-app-downloads/
      - metric: pypi_mistralai_monthly_downloads
        value: 47593598
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/mistralai
      - metric: stackoverflow_2025_llm_usage_share_percent
        value: 10.4
        observed_at: '2026-08-22'
        source_url: https://survey.stackoverflow.co/2025/technology
      sources:
      - type: reputable_reporting
        title: Mistral confirms Le Chat surpassed 1M downloads in 14 days (2025-02)
        url: https://www.eweek.com/news/mistral-ai-le-chat-app-downloads/
      - type: package_registry_stats
        title: PyPI download stats for mistralai
        url: https://pypistats.org/packages/mistralai
      - type: independent_survey
        title: 2025 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2025/technology
      - type: official_docs
        title: Amazon Bedrock models at a glance listing the Mistral AI model family
        url: https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards.html
    huggingface_inference_providers:
      status: scored
      score: 80
      tier: ubiquitous
      basis: Hugging Face reports 13M users and a hub client that powers 200K dependent libraries, alongside its durable role as a major open-model platform.
      components: {adoption: 40, developer_ecosystem: 30, independent_awareness: 0, durable_recognition: 10}
      signals:
        - {metric: platform_users, value: 13000000, observed_at: "2026-08-22", source_url: https://huggingface.co/blog/huggingface/state-of-os-hf-spring-2026}
        - {metric: dependent_libraries, value: 200000, observed_at: "2026-08-22", source_url: https://huggingface.co/blog/huggingface-hub-v1}
      sources:
        - {type: official_ecosystem_report, title: State of Open Source on Hugging Face Spring 2026, url: https://huggingface.co/blog/huggingface/state-of-os-hf-spring-2026}
        - {type: official_changelog, title: huggingface_hub v1.0, url: https://huggingface.co/blog/huggingface-hub-v1}
    cloudflare_workers_ai:
      status: scored
      score: 80
      tier: ubiquitous
      basis: Cloudflare officially reports 2M+ developers building on the Workers platform that hosts Workers AI, the Workers-AI-specific AI SDK provider package sees 926K monthly npm downloads, Cloudflare holds 20.1% share in the independent 2025 Stack Overflow cloud-platform survey, and Cloudflare is a globally established developer platform.
      components:
        adoption: 30
        developer_ecosystem: 25
        independent_awareness: 15
        durable_recognition: 10
      signals:
      - metric: workers_platform_developers
        value: 2000000
        qualifier: at_least
        observed_at: '2026-08-22'
        source_url: https://blog.cloudflare.com/welcome-to-developer-week-2024/
      - metric: npm_workers_ai_provider_monthly_downloads
        value: 925963
        observed_at: '2026-08-22'
        source_url: https://www.npmjs.com/package/workers-ai-provider
      - metric: stackoverflow_2025_cloud_platform_share_percent
        value: 20.1
        observed_at: '2026-08-22'
        source_url: https://survey.stackoverflow.co/2025/technology
      sources:
      - type: official_blog
        title: Cloudflare Developer Week 2024 citing two million developers on the platform (2024-03-31)
        url: https://blog.cloudflare.com/welcome-to-developer-week-2024/
      - type: package_registry_stats
        title: npm download stats for workers-ai-provider
        url: https://www.npmjs.com/package/workers-ai-provider
      - type: independent_survey
        title: 2025 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2025/technology
    cohere:
      status: partial
      score: 38
      tier: established
      basis: Cohere's official Python SDK shows very heavy documented usage (35.2M monthly PyPI downloads) and Cohere stewards the widely recognized first-party Command / Embed / Rerank / Aya model families, but no citable numeric active-developer or customer count was found (revenue and valuation reports are excluded as proxies), so adoption and independent awareness are unscored.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 8
      signals:
      - metric: pypi_monthly_downloads_cohere
        value: 35221292
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/api/packages/cohere/recent
      sources:
      - type: package_downloads
        title: pypistats recent downloads for the official cohere package
        url: https://pypistats.org/api/packages/cohere/recent
      - type: official_docs
        title: Cohere models documentation (first-party Command/Embed/Rerank/Aya families)
        url: https://docs.cohere.com/v2/docs/models
    vercel_ai_gateway:
      status: scored
      score: 48
      tier: established
      basis: The Vercel AI SDK (npm ai), the gateway's primary client, sees 86.6M monthly downloads and Vercel holds 10.6% share in the independent 2025 Stack Overflow cloud-platform survey as an established global developer platform and Next.js steward; no gateway-scoped or platform-user adoption count is published, and Next.js framework user figures were not counted as gateway adoption.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 10
        durable_recognition: 8
      signals:
      - metric: npm_ai_sdk_monthly_downloads
        value: 86551852
        observed_at: '2026-08-22'
        source_url: https://www.npmjs.com/package/ai
      - metric: stackoverflow_2025_cloud_platform_share_percent
        value: 10.6
        observed_at: '2026-08-22'
        source_url: https://survey.stackoverflow.co/2025/technology
      sources:
      - type: package_registry_stats
        title: npm download stats for ai (Vercel AI SDK)
        url: https://www.npmjs.com/package/ai
      - type: independent_survey
        title: 2025 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2025/technology
      - type: official_product
        title: Next.js by Vercel
        url: https://nextjs.org/
    ibm_watsonx_ai_runtime:
      status: scored
      score: 43
      tier: established
      basis: No official watsonx client count was found; the service-specific ibm-watsonx-ai SDK sees 1.5M monthly PyPI downloads, IBM Cloud held 1.2% share in the independent 2025 Stack Overflow cloud-platform survey, and IBM is a globally established platform that stewards the first-party Granite model family.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 3
        durable_recognition: 10
      signals:
      - metric: pypi_ibm_watsonx_ai_monthly_downloads
        value: 1506468
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/ibm-watsonx-ai
      - metric: stackoverflow_2025_cloud_platform_share_percent
        value: 1.2
        observed_at: '2026-08-22'
        source_url: https://survey.stackoverflow.co/2025/technology
      sources:
      - type: package_registry_stats
        title: PyPI download stats for ibm-watsonx-ai
        url: https://pypistats.org/packages/ibm-watsonx-ai
      - type: independent_survey
        title: 2025 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2025/technology
      - type: official_product
        title: IBM Granite first-party model family
        url: https://www.ibm.com/granite
    opencode_zen:
      status: scored
      score: 70
      tier: well_known
      basis: OpenCode officially reports 16M monthly developers and 950 contributors on its own site, the opencode-ai npm package that ships the Zen client sees 9,519,371 monthly downloads, and the primary official repository holds 200,402 stars (the site rounds this to 195K). The 16M figure is OpenCode-product-scoped rather than Zen-API-scoped — Zen is the operator's first-party inference gateway inside that CLI — and no independent survey share or durable-recognition claim applies.
      components:
        adoption: 40
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: official_monthly_developers
        value: 16000000
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://opencode.ai/
      - metric: npm_opencode_ai_monthly_downloads
        value: 9519371
        observed_at: '2026-08-23'
        source_url: https://api.npmjs.org/downloads/point/last-month/opencode-ai
      - metric: github_stars_official_repo
        value: 200402
        observed_at: '2026-08-23'
        source_url: https://github.com/anomalyco/opencode
      - metric: official_contributors
        value: 950
        observed_at: '2026-08-23'
        source_url: https://opencode.ai/
      sources:
      - type: official_product
        title: OpenCode home page citing 16M monthly developers and 195K GitHub stars
        url: https://opencode.ai/
      - type: package_registry_stats
        title: npm download stats for opencode-ai
        url: https://www.npmjs.com/package/opencode-ai
      - type: official_repository
        title: OpenCode primary repository
        url: https://github.com/anomalyco/opencode
    zai:
      status: partial
      score: 61
      tier: well_known
      basis: Reputable IPO coverage of operator Zhipu documents 2.9M users on its MaaS open platform (2026-01-09), the official zai-sdk Python package shows 120K monthly downloads, and the GLM family is a widely recognized first-party open-model family. Partial because the platform-user figure covers the operator's combined MaaS platform rather than the international Z.AI API service specifically, and no independent survey or search-interest measure was captured.
      components:
        adoption: 30
        developer_ecosystem: 25
        independent_awareness: 0
        durable_recognition: 6
      signals:
      - metric: maas_open_platform_users
        value: 2900000
        observed_at: '2026-01-09'
        source_url: https://news.cgtn.com/news/2026-01-09/4-key-takeaways-Zhipu-becomes-first-Chinese-AI-firm-to-go-public-1JN0K7CEJaw/p.html
      - metric: pypi_monthly_downloads_zai_sdk
        value: 120301
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/api/packages/zai-sdk/recent
      sources:
      - type: reputable_reporting
        title: CGTN coverage of Zhipu IPO metrics (2.9M MaaS platform users, 12,000 enterprise clients)
        url: https://news.cgtn.com/news/2026-01-09/4-key-takeaways-Zhipu-becomes-first-Chinese-AI-firm-to-go-public-1JN0K7CEJaw/p.html
      - type: package_downloads
        title: pypistats recent downloads for the official zai-sdk package
        url: https://pypistats.org/api/packages/zai-sdk/recent
      - type: official_repository
        title: Z.ai organization model releases (GLM family)
        url: https://huggingface.co/zai-org
    llmapi_ai:
      status: not_found
      score: 0
      tier: unknown
      reason: 'Searched on 2026-08-23 and found no citable rubric-eligible evidence: llmapi.ai publishes no user, developer, customer, or request-volume figure (its home page quantifies only ''400+ models''); no official GitHub organization or repository for llmapi.ai exists (the github.com/llmapi-io organization belongs to the unrelated llmapi.io self-host project); no official npm package (llmapi, llmapi-ai) or PyPI package for the service exists; and no developer survey, search-interest measure, or third-party metric covers it. Only a third-party provider-request discussion thread on cline/cline mentions the service, which is not a rubric-eligible metric.'
    api_airforce:
      status: not_found
      score: 0
      tier: unknown
      reason: 'Searched on 2026-08-23 and found no citable rubric-eligible evidence: api.airforce publishes only service-capability figures (99.4% uptime, 185ms average latency, 620+ models) and no user, developer, customer, or request-volume count; the github.com/api-airforce organization has 0 public repositories and 0 followers; no official npm or PyPI package exists (the npm ''airforce'' package is unrelated); and the service appears in no developer survey or reproducible search-interest measure. The only related repository found is a third-party account auto-registration tool (lza6/AirForce-API-Auto-Register-System, 35 stars), which is neither official nor an adoption metric.'
    llm7:
      status: partial
      score: 5
      tier: niche
      basis: The official llm7.io gateway repository shows 202 GitHub stars (created 2025-04) and the llm7 PyPI package about 16 downloads in the last month; both sit below the 1K+ developer ecosystem anchor, so a sub-threshold 5 is scored, and no official adoption counts, survey share, or durable-recognition evidence exists.
      components:
        adoption: 0
        developer_ecosystem: 5
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: github_stars_official_repo
        value: 202
        observed_at: '2026-08-22'
        source_url: https://github.com/chigwell/llm7.io
      - metric: pypi_llm7_downloads_last_month
        value: 16
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/llm7
      sources:
      - type: official_repository
        title: llm7.io gateway repository
        url: https://github.com/chigwell/llm7.io
      - type: package_registry
        title: llm7 PyPI download statistics
        url: https://pypistats.org/packages/llm7
    modelscope_inference:
      status: partial
      score: 80
      tier: ubiquitous
      basis: Alibaba Cloud's official developer community states that ModelScope serves over 14 million developers across more than 50,000 AI models (2026-07-17), the official modelscope PyPI package sees 5,665,582 monthly downloads, and ModelScope is operated by Alibaba Cloud — a globally established cloud platform that also distributes the first-party Qwen family. Partial because the 14M developer count covers the whole ModelScope community rather than the API-Inference service specifically, and no independent survey or search-interest measure was captured.
      components:
        adoption: 40
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 10
      signals:
      - metric: platform_developers
        value: 14000000
        qualifier: at_least
        observed_at: '2026-07-17'
        source_url: https://developer.aliyun.com/modelscope/
      - metric: platform_models
        value: 50000
        qualifier: at_least
        observed_at: '2026-07-17'
        source_url: https://developer.aliyun.com/modelscope/
      - metric: pypi_modelscope_monthly_downloads
        value: 5665582
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/modelscope/recent
      - metric: github_stars_official_repo
        value: 9098
        observed_at: '2026-08-23'
        source_url: https://github.com/modelscope/modelscope
      - metric: github_org_followers
        value: 6506
        observed_at: '2026-08-23'
        source_url: https://github.com/modelscope
      sources:
      - type: official_product
        title: Alibaba Cloud developer community ModelScope page citing 14M+ developers and 50K+ models
        url: https://developer.aliyun.com/modelscope/
      - type: package_registry_stats
        title: PyPI download stats for modelscope
        url: https://pypistats.org/packages/modelscope
      - type: official_repository
        title: ModelScope primary repository
        url: https://github.com/modelscope/modelscope
    awanllm:
      status: not_found
      score: 0
      tier: unknown
      reason: 'As of 2026-08-22 no citable rubric-eligible evidence exists: AwanLLM publishes no user, customer, or request-volume figures, operates no official repository or packages, appears in no developer survey, and third-party directories list it without usable metrics (G2 shows zero reviews; only an unofficial community API wrapper exists).'
    arliai:
      status: partial
      score: 20
      tier: established
      basis: Arli AI's verified first-party Hugging Face organization records 13,716 model downloads in the last 30 days (501,806 all-time across 78 models) and 763 followers; only the developer ecosystem component is scorable (10K+ current downloads threshold) because no official user counts, survey share, or search-interest measurements exist.
      components:
        adoption: 0
        developer_ecosystem: 20
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: hf_org_model_downloads_30d
        value: 13716
        observed_at: '2026-08-22'
        source_url: https://huggingface.co/api/models?author=ArliAI
      - metric: hf_org_model_downloads_all_time
        value: 501806
        observed_at: '2026-08-22'
        source_url: https://huggingface.co/api/models?author=ArliAI
      - metric: hf_org_followers
        value: 763
        observed_at: '2026-08-22'
        source_url: https://huggingface.co/ArliAI
      sources:
      - type: official_repository
        title: Arli AI organization on Hugging Face (verified)
        url: https://huggingface.co/ArliAI
    freeinference_org:
      status: not_found
      score: 0
      tier: unknown
      reason: 'Searched on 2026-08-23 and found no citable rubric-eligible evidence: freeinference.org publishes no user, developer, request, or token-served figure; the operator, the Harvard SEAS MadSys Lab, publishes no adoption count for the service; no official GitHub organization or repository exists for freeinference.org (the github.com/freeinference user account has 0 followers, and repository searches for ''freeinference'' and ''madsys'' return only unrelated third-party projects with 0-1 stars); no official npm or PyPI package exists; and the service appears in no developer survey or reproducible search-interest measure.'
    fastrouter:
      status: not_found
      score: 0
      tier: unknown
      reason: 'Searched on 2026-08-23 and found no citable rubric-eligible evidence: fastrouter.ai publishes customer logos but no numeric user, developer, customer, request, or token figure; the official github.com/fastrouter organization has 0 followers and 4 repositories whose highest star count is 1 (its docs repository has 0 stars); no official npm package (fastrouter, fastrouter-ai, @fastrouter/sdk) or PyPI package exists; and the service appears in no developer survey or reproducible search-interest measure. Third-party directory listings (SourceForge, VoltAgent, theresanaiforthat) carry no usable metrics.'
    kilo_ai_gateway:
      status: scored
      score: 60
      tier: well_known
      basis: Kilo officially reports 3M+ Kilo Coders and 40T+ tokens processed on kilo.ai, and the first-party Kilo Code extension records 1,449,845 installs on the VS Code Marketplace, with 191,870 monthly npm downloads for @kilocode/cli and 26,975 GitHub stars. Both headline figures are Kilo-product-scoped — the gateway is the first-party inference layer those clients call — and no independent survey share or durable-recognition claim applies.
      components:
        adoption: 30
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: official_kilo_coders
        value: 3000000
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://kilo.ai/
      - metric: official_tokens_processed
        value: 40000000000000
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://kilo.ai/
      - metric: vscode_marketplace_installs
        value: 1449845
        observed_at: '2026-08-23'
        source_url: https://marketplace.visualstudio.com/items?itemName=kilocode.Kilo-Code
      - metric: npm_kilocode_cli_monthly_downloads
        value: 191870
        observed_at: '2026-08-23'
        source_url: https://api.npmjs.org/downloads/point/last-month/@kilocode/cli
      - metric: github_stars_official_repo
        value: 26975
        observed_at: '2026-08-23'
        source_url: https://github.com/Kilo-Org/kilocode
      sources:
      - type: official_product
        title: Kilo home page citing 3M+ Kilo Coders and 40T+ tokens processed
        url: https://kilo.ai/
      - type: marketplace_stats
        title: Kilo Code on the Visual Studio Marketplace
        url: https://marketplace.visualstudio.com/items?itemName=kilocode.Kilo-Code
      - type: package_registry_stats
        title: npm download stats for @kilocode/cli
        url: https://www.npmjs.com/package/@kilocode/cli
    scaleway_generative_apis:
      status: scored
      score: 32
      tier: established
      basis: No Generative APIs adoption count is published and Scaleway discloses no company-wide customer number, so adoption is unscored; the official scaleway Python SDK sees 227,797 monthly PyPI downloads (with 66,620 monthly npm downloads for @scaleway/sdk), Scaleway held 0.9% share in the independent 2024 Stack Overflow cloud-platform survey, and Scaleway is an established European cloud provider within the iliad Group. Scored on the operating company's developer standing, with the service-scope gap stated.
      components:
        adoption: 0
        developer_ecosystem: 25
        independent_awareness: 2
        durable_recognition: 5
      signals:
      - metric: pypi_scaleway_monthly_downloads
        value: 227797
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/scaleway/recent
      - metric: npm_scaleway_sdk_monthly_downloads
        value: 66620
        observed_at: '2026-08-23'
        source_url: https://api.npmjs.org/downloads/point/last-month/@scaleway/sdk
      - metric: stackoverflow_2024_cloud_platform_share_percent
        value: 0.9
        observed_at: '2026-08-23'
        source_url: https://survey.stackoverflow.co/2024/technology
      - metric: github_org_followers
        value: 311
        observed_at: '2026-08-23'
        source_url: https://github.com/scaleway
      sources:
      - type: package_registry_stats
        title: PyPI download stats for the official Scaleway Python SDK
        url: https://pypistats.org/packages/scaleway
      - type: independent_survey
        title: 2024 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2024/technology
      - type: official_about
        title: About Scaleway (iliad Group European cloud provider)
        url: https://www.scaleway.com/en/about-us/
    qwen_cloud:
      status: scored
      score: 43
      tier: established
      basis: Alibaba states officially that its open-weight Qwen models passed 3 billion downloads in six months across 460+ released models with 300,000+ community derivatives (2026-08-15), the service-specific dashscope SDK sees 3,194,979 monthly PyPI downloads, the Qwen Hugging Face organization records 337,309,074 model downloads in 30 days and 99,505 followers, Alibaba Cloud held 1.2% share in the independent 2024 Stack Overflow cloud-platform survey, and Alibaba Cloud is a globally established cloud platform stewarding the widely recognized first-party Qwen family. Adoption is unscored because no Model Studio / Qwen API user, developer, or customer count is published — the 234M Qwen app users and the reported 8x Model Studio customer growth are consumer-app and ratio figures, not an API-scoped count.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 3
        durable_recognition: 10
      signals:
      - metric: pypi_dashscope_monthly_downloads
        value: 3194979
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/dashscope/recent
      - metric: hf_qwen_org_model_downloads_30d
        value: 337309074
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/api/models?author=Qwen
      - metric: hf_qwen_org_followers
        value: 99505
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/Qwen
      - metric: official_qwen_cumulative_downloads_six_months
        value: 3000000000
        qualifier: at_least
        observed_at: '2026-08-15'
        source_url: https://fortune.com/2026/08/15/alibaba-qwen-open-ai-models-3-billion-downloads-meta-google/
      - metric: official_qwen_derivative_models
        value: 300000
        qualifier: at_least
        observed_at: '2026-08-15'
        source_url: https://fortune.com/2026/08/15/alibaba-qwen-open-ai-models-3-billion-downloads-meta-google/
      - metric: github_org_followers_qwenlm
        value: 18812
        observed_at: '2026-08-23'
        source_url: https://github.com/QwenLM
      - metric: stackoverflow_2024_cloud_platform_share_percent
        value: 1.2
        observed_at: '2026-08-23'
        source_url: https://survey.stackoverflow.co/2024/technology
      sources:
      - type: package_registry_stats
        title: PyPI download stats for the official Alibaba Cloud dashscope SDK
        url: https://pypistats.org/packages/dashscope
      - type: official_repository
        title: Qwen organization on Hugging Face
        url: https://huggingface.co/Qwen
      - type: reputable_reporting
        title: Alibaba-provided figures on Qwen downloads and derivative models (2026-08-15)
        url: https://fortune.com/2026/08/15/alibaba-qwen-open-ai-models-3-billion-downloads-meta-google/
      - type: independent_survey
        title: 2024 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2024/technology
      - type: official_docs
        title: Alibaba Cloud Model Studio product documentation
        url: https://www.alibabacloud.com/help/en/model-studio/what-is-model-studio
    sea_lion_api:
      status: partial
      score: 25
      tier: established
      basis: AI Singapore's verified Hugging Face organization records 44,163 model downloads in the last 30 days across 77 models with 398 followers, and SEA-LION is a recognized first-party regional open-model family that Google DeepMind features in its Gemma "Gemmaverse" programme. Partial because no SEA-LION API user, key, or request count is published, no independent survey or search-interest measure covers it, and the widely repeated ~235,000 cumulative-download figure could not be confirmed against a first-party source.
      components:
        adoption: 0
        developer_ecosystem: 20
        independent_awareness: 0
        durable_recognition: 5
      signals:
      - metric: hf_org_model_downloads_30d
        value: 44163
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/api/models?author=aisingapore
      - metric: hf_org_followers
        value: 398
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/aisingapore
      - metric: github_stars_official_repo
        value: 422
        observed_at: '2026-08-23'
        source_url: https://github.com/aisingapore/sealion
      - metric: github_org_followers
        value: 331
        observed_at: '2026-08-23'
        source_url: https://github.com/aisingapore
      sources:
      - type: official_repository
        title: AI Singapore organization on Hugging Face (verified)
        url: https://huggingface.co/aisingapore
      - type: official_repository
        title: SEA-LION model repository
        url: https://github.com/aisingapore/sealion
      - type: partner_page
        title: Google DeepMind Gemmaverse page featuring AI Singapore's SEA-LION
        url: https://deepmind.google/models/gemma/gemmaverse/sea-lion/
    ndif:
      status: scored
      score: 30
      tier: established
      basis: NDIF officially reports 110+ published papers at top venues and a 1,200+ member research community, and its official NNsight client sees 24,377 monthly PyPI downloads with 1,041 GitHub stars on the primary repository. No independent survey share applies, and NDIF is an NSF-funded academic fabric rather than a global developer platform or model-family steward, so durable recognition is unscored.
      components:
        adoption: 10
        developer_ecosystem: 20
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: official_community_members
        value: 1200
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://ndif.us/
      - metric: official_published_papers
        value: 110
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://ndif.us/
      - metric: pypi_nnsight_monthly_downloads
        value: 24377
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/nnsight/recent
      - metric: github_stars_official_repo
        value: 1041
        observed_at: '2026-08-23'
        source_url: https://github.com/ndif-team/nnsight
      sources:
      - type: official_product
        title: NDIF home page citing 110+ papers and a 1,200+ member community
        url: https://ndif.us/
      - type: package_registry_stats
        title: PyPI download stats for nnsight
        url: https://pypistats.org/packages/nnsight
      - type: official_repository
        title: NNsight primary repository
        url: https://github.com/ndif-team/nnsight
    ai_horde:
      status: scored
      score: 30
      tier: established
      basis: The AI Horde's own public statistics endpoint documents 6,198,825 text requests in the trailing month and 291,762,534 text requests totalling 67.3B tokens since launch, and its public user listing enumerates at least 250,000 registered accounts before the API's pagination ceiling; the official horde_sdk package sees 4,343 monthly PyPI downloads and the coordinator repository holds 1,532 stars. Request volume at millions per month is mapped conservatively to the 100K+ adoption rung; no independent survey or durable-recognition evidence applies.
      components:
        adoption: 20
        developer_ecosystem: 10
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: text_requests_last_30_days
        value: 6198825
        observed_at: '2026-08-23'
        source_url: https://aihorde.net/api/v2/stats/text/totals
      - metric: text_requests_all_time
        value: 291762534
        observed_at: '2026-08-23'
        source_url: https://aihorde.net/api/v2/stats/text/totals
      - metric: text_tokens_all_time
        value: 67297802984
        observed_at: '2026-08-23'
        source_url: https://aihorde.net/api/v2/stats/text/totals
      - metric: registered_users_enumerable_floor
        value: 250000
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://aihorde.net/api/v2/users?page=10000
        note: The public user listing returns results through page 10000 at 25 per page and none beyond, so this is a pagination-limited floor rather than a published total.
      - metric: pypi_horde_sdk_monthly_downloads
        value: 4343
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/horde_sdk/recent
      - metric: github_stars_official_repo
        value: 1532
        observed_at: '2026-08-23'
        source_url: https://github.com/Haidra-Org/AI-Horde
      sources:
      - type: official_statistics
        title: AI Horde public text-generation totals endpoint
        url: https://aihorde.net/api/v2/stats/text/totals
      - type: package_registry_stats
        title: PyPI download stats for horde_sdk
        url: https://pypistats.org/packages/horde_sdk
      - type: official_repository
        title: AI Horde coordinator repository
        url: https://github.com/Haidra-Org/AI-Horde
    pollinations:
      status: partial
      score: 15
      tier: niche
      basis: The official Pollinations repository holds 4,976 GitHub stars with 627 followers on the organization, and its first-party README documents 500+ community projects built on the APIs. The star count sits between the methodology's 1K+ (10) and 10K+ (20) developer-ecosystem rungs, so 15 is interpolated. Partial because no first-party active-user, developer, or request-volume figure could be confirmed — the widely repeated "10K weekly active developers / 1.5M daily requests" claim appears only in third-party tool directories, not in any Pollinations source — and no survey or search-interest measure applies.
      components:
        adoption: 0
        developer_ecosystem: 15
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: github_stars_official_repo
        value: 4976
        observed_at: '2026-08-23'
        source_url: https://github.com/pollinations/pollinations
      - metric: official_community_projects
        value: 500
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://github.com/pollinations/pollinations
      - metric: github_org_followers
        value: 627
        observed_at: '2026-08-23'
        source_url: https://github.com/pollinations
      sources:
      - type: official_repository
        title: Pollinations repository citing 500+ community projects
        url: https://github.com/pollinations/pollinations
      - type: code_hosting_stats
        title: Pollinations GitHub organization followers
        url: https://github.com/pollinations
    puter_js:
      status: scored
      score: 35
      tier: established
      basis: Puter officially reports 80K+ developers, 130K+ apps powered, and 400K+ installations on its developer site, and the open-source Puter repository holds 43,179 GitHub stars with 1,621 followers on the organization. The 80K developer count falls below the 100K adoption rung, and the 130K+ apps figure scores the 100K+ developer-ecosystem rung; no independent survey share or durable-recognition claim applies.
      components:
        adoption: 10
        developer_ecosystem: 25
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: official_developers
        value: 80000
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://developer.puter.com/
      - metric: official_apps_powered
        value: 130000
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://developer.puter.com/
      - metric: official_installations
        value: 400000
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://developer.puter.com/
      - metric: github_stars_official_repo
        value: 43179
        observed_at: '2026-08-23'
        source_url: https://github.com/HeyPuter/puter
      - metric: github_org_followers
        value: 1621
        observed_at: '2026-08-23'
        source_url: https://github.com/HeyPuter
      sources:
      - type: official_product
        title: Puter.js developer site citing 80K+ developers, 130K+ apps powered, and 400K+ installations
        url: https://developer.puter.com/
      - type: official_repository
        title: Puter open-source repository
        url: https://github.com/HeyPuter/puter
    public_ai:
      status: partial
      score: 3
      tier: niche
      basis: The only citable rubric-eligible metrics are the operator's own code-hosting footprints — 163 followers on the forpublicai GitHub organization, 79 stars on its largest repository, and 69 followers on the publicai Hugging Face organization — all far below the 1K+ developer-ecosystem anchor, so a sub-threshold 3 is scored. Partial because publicai.co, platform.publicai.co, the Public AI Substack launch posts, and the Hugging Face inference-provider announcement publish no user, developer, request, or token-served count; the models the utility serves (Apertus, SEA-LION, EuroLLM) are stewarded by its partners rather than by Public AI, so no durable-recognition points are assigned.
      components:
        adoption: 0
        developer_ecosystem: 3
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: github_org_followers
        value: 163
        observed_at: '2026-08-23'
        source_url: https://github.com/forpublicai
      - metric: github_stars_largest_official_repo
        value: 79
        observed_at: '2026-08-23'
        source_url: https://github.com/forpublicai/chat.publicai.co
      - metric: hf_org_followers
        value: 69
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/publicai
      sources:
      - type: code_hosting_stats
        title: Public AI Inference Utility GitHub organization
        url: https://github.com/forpublicai
      - type: official_repository
        title: Public AI Inference Utility organization on Hugging Face
        url: https://huggingface.co/publicai
      - type: partner_page
        title: Public AI listed as a Hugging Face inference provider
        url: https://huggingface.co/docs/inference-providers/providers/publicai
    lightning_ai_model_apis:
      status: scored
      score: 55
      tier: established
      basis: Lightning AI's own merger announcement states the platform is used by over 400,000 developers, startups, and enterprises, and its flagship first-party package (PyPI lightning) sees 6.29M monthly downloads; the Model APIs service itself is far smaller (litAI SDK 642 monthly downloads, 52 repo stars), so the ecosystem component is scored at company scope and the basis says so. Durable recognition is partial credit for stewardship of the PyTorch Lightning developer framework and operation of a merged AI cloud, not for a model family.
      components:
        adoption: 20
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 5
      signals:
      - metric: platform_developers_startups_enterprises
        value: 400000
        qualifier: at_least
        observed_at: '2026-01-21'
        source_url: https://www.businesswire.com/news/home/20260121371691/en/Lightning-AI-and-Voltage-Park-Complete-Merger-to-Create-the-First-Cloud-Built-for-AI
      - metric: pypi_lightning_monthly_downloads
        value: 6291056
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/lightning/recent
      - metric: pypi_litai_monthly_downloads_service_scope
        value: 642
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/litai/recent
      - metric: github_stars_litai_service_sdk
        value: 52
        observed_at: '2026-08-23'
        source_url: https://github.com/Lightning-AI/LitAI
      - metric: github_org_followers_lightning_ai
        value: 6857
        observed_at: '2026-08-23'
        source_url: https://github.com/Lightning-AI
      sources:
      - type: company_press_release
        title: Lightning AI and Voltage Park complete merger (over 400,000 developers, startups and enterprises), 2026-01-21
        url: https://www.businesswire.com/news/home/20260121371691/en/Lightning-AI-and-Voltage-Park-Complete-Merger-to-Create-the-First-Cloud-Built-for-AI
      - type: package_registry_stats
        title: PyPI download stats for the lightning package
        url: https://pypistats.org/api/packages/lightning/recent
      - type: package_registry_stats
        title: PyPI download stats for the litai Model APIs SDK
        url: https://pypistats.org/api/packages/litai/recent
      - type: official_product
        title: PyTorch Lightning framework by Lightning AI
        url: https://lightning.ai/pytorch-lightning/
    modal:
      status: partial
      score: 30
      tier: established
      basis: Only one quantitative signal was captured; the modal PyPI client sees 80.5M monthly downloads, but Modal publishes no official developer or customer count and no independent survey or durable-recognition evidence was found.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: pypi_modal_monthly_downloads
        value: 80466285
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/modal
      sources:
      - type: package_registry_stats
        title: PyPI download stats for modal
        url: https://pypistats.org/packages/modal
    beam_cloud:
      status: partial
      score: 20
      tier: established
      basis: Only ecosystem evidence is citable; Beam's official beam-client Python SDK sees 57,444 monthly PyPI downloads and its open-source beta9 runtime has 1,750 GitHub stars, but Beam publishes no developer, team, or customer count, appears in no developer survey, and has no durable-recognition claim.
      components:
        adoption: 0
        developer_ecosystem: 20
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: pypi_beam_client_monthly_downloads
        value: 57444
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/beam-client/recent
      - metric: github_stars_beta9_official_runtime
        value: 1750
        observed_at: '2026-08-23'
        source_url: https://github.com/beam-cloud/beta9
      - metric: github_org_followers_beam_cloud
        value: 87
        observed_at: '2026-08-23'
        source_url: https://github.com/beam-cloud
      sources:
      - type: package_registry_stats
        title: PyPI download stats for beam-client
        url: https://pypistats.org/api/packages/beam-client/recent
      - type: official_repository
        title: beam-cloud/beta9 serverless GPU runtime
        url: https://github.com/beam-cloud/beta9
    sail_research:
      status: partial
      score: 20
      tier: established
      basis: The only citable signal is Sail Research's own SDK; the sail PyPI package (documented at docs.sailresearch.com, repository sailresearchco/sail) sees 39,041 monthly downloads. No user, customer, or request-volume figure is published, the sailresearchco GitHub organization has 1 follower across 6 mostly forked repositories, and there is no survey or durable-recognition evidence.
      components:
        adoption: 0
        developer_ecosystem: 20
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: pypi_sail_monthly_downloads
        value: 39041
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/sail/recent
      - metric: github_org_followers_sailresearchco
        value: 1
        observed_at: '2026-08-23'
        source_url: https://github.com/sailresearchco
      sources:
      - type: package_registry
        title: sail PyPI project (Python SDK for the Sail platform, docs.sailresearch.com)
        url: https://pypi.org/project/sail/
      - type: package_registry_stats
        title: PyPI download stats for sail
        url: https://pypistats.org/api/packages/sail/recent
    cartesia:
      status: partial
      score: 30
      tier: established
      basis: One quantitative signal plus qualitative recognition; the official cartesia Python SDK sees 790,319 monthly PyPI downloads (the TypeScript client adds 286,231 npm downloads, not counted separately) and Cartesia stewards the first-party Sonic text-to-speech and Ink speech-to-text model families. Cartesia's own pages describe only "thousands of customers" with no numeric count, and no independent survey share exists.
      components:
        adoption: 0
        developer_ecosystem: 25
        independent_awareness: 0
        durable_recognition: 5
      signals:
      - metric: pypi_cartesia_monthly_downloads
        value: 790319
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/cartesia/recent
      - metric: npm_cartesia_js_monthly_downloads
        value: 286231
        observed_at: '2026-08-23'
        source_url: https://www.npmjs.com/package/@cartesia/cartesia-js
      - metric: github_org_followers_cartesia_ai
        value: 243
        observed_at: '2026-08-23'
        source_url: https://github.com/cartesia-ai
      sources:
      - type: package_registry_stats
        title: PyPI download stats for the official cartesia package
        url: https://pypistats.org/api/packages/cartesia/recent
      - type: package_registry_stats
        title: npm download stats for the official @cartesia/cartesia-js client
        url: https://www.npmjs.com/package/@cartesia/cartesia-js
      - type: official_product
        title: Cartesia first-party Sonic and Ink model families
        url: https://www.cartesia.ai/launch
    elevenlabs_api:
      status: scored
      score: 65
      tier: well_known
      basis: ElevenLabs officially reports millions of users and adoption by employees at over 60% of Fortune 500 companies, the elevenlabs PyPI SDK sees 10.7M monthly downloads, and the company stewards a widely recognized first-party voice model family; no independent survey share was found.
      components:
        adoption: 30
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 5
      signals:
      - metric: platform_users
        value: 1000000
        qualifier: at_least
        observed_at: '2026-08-22'
        source_url: https://elevenlabs.io/blog/series-c
      - metric: pypi_elevenlabs_monthly_downloads
        value: 10712499
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/elevenlabs
      sources:
      - type: official_blog
        title: ElevenLabs Series C post citing millions of users and 60%+ Fortune 500 reach (2025-01-30)
        url: https://elevenlabs.io/blog/series-c
      - type: package_registry_stats
        title: PyPI download stats for elevenlabs
        url: https://pypistats.org/packages/elevenlabs
    ovhcloud_ai_endpoints:
      status: scored
      score: 22
      tier: established
      basis: No AI Endpoints adoption count exists and OVHcloud's 1.6M group-wide customers were not counted as service adoption; the OVH GitHub organization has 1,580 followers, OVH held 3.0% share in the independent 2024 Stack Overflow cloud-platform survey, and OVHcloud is Europe's leading cloud provider operating in 140 countries.
      components:
        adoption: 0
        developer_ecosystem: 10
        independent_awareness: 5
        durable_recognition: 7
      signals:
      - metric: github_org_followers_ovh
        value: 1580
        observed_at: '2026-08-22'
        source_url: https://github.com/ovh
      - metric: stackoverflow_2024_cloud_platform_share_percent
        value: 3.0
        observed_at: '2026-08-22'
        source_url: https://survey.stackoverflow.co/2024/technology
      sources:
      - type: code_hosting_stats
        title: OVHcloud GitHub organization followers
        url: https://github.com/ovh
      - type: independent_survey
        title: 2024 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2024/technology
      - type: official_about
        title: OVHcloud about page citing 1.6M customers and Europe's leading cloud provider
        url: https://www.ovhcloud.com/en/about-us/
    requesty:
      status: scored
      score: 30
      tier: established
      basis: Requesty publishes first-party adoption figures of 70,000+ developers and 90+ billion tokens routed daily, and its official @requesty/ai-sdk npm client sees 20,755 monthly downloads; the developer count falls below the 100K adoption threshold, and no independent survey or durable-recognition evidence exists.
      components:
        adoption: 10
        developer_ecosystem: 20
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: official_developers
        value: 70000
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://www.requesty.ai/
      - metric: official_tokens_routed_per_day
        value: 90000000000
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://www.requesty.ai/
      - metric: npm_requesty_ai_sdk_monthly_downloads
        value: 20755
        observed_at: '2026-08-23'
        source_url: https://www.npmjs.com/package/@requesty/ai-sdk
      - metric: github_org_followers_requestyai
        value: 10
        observed_at: '2026-08-23'
        source_url: https://github.com/requestyai
      sources:
      - type: official_product
        title: Requesty homepage stating 70,000+ developers and 90+ billion tokens processed daily
        url: https://www.requesty.ai/
      - type: package_registry_stats
        title: npm download stats for the official @requesty/ai-sdk client
        url: https://www.npmjs.com/package/@requesty/ai-sdk
    inception_platform:
      status: partial
      score: 5
      tier: niche
      basis: Only one indirect signal is citable; Inception stewards the first-party Mercury diffusion LLM family, which Microsoft distributes on Azure AI Foundry, giving partial durable recognition. Inception publishes no user, developer, or customer count, operates no official GitHub organization or package (the API is OpenAI-compatible with no first-party SDK), Mercury weights are not published, and no independent survey lists it.
      components:
        adoption: 0
        developer_ecosystem: 0
        independent_awareness: 0
        durable_recognition: 5
      signals:
      - metric: first_party_model_family_on_major_cloud
        value: 1
        qualifier: mercury_on_azure_ai_foundry
        observed_at: '2026-08-13'
        source_url: https://www.inceptionlabs.ai/blog/mercury-azure-foundry
      sources:
      - type: official_announcement
        title: Mercury diffusion LLM available on Azure AI Foundry (2026-08-13)
        url: https://www.inceptionlabs.ai/blog/mercury-azure-foundry
      - type: official_docs
        title: Inception platform models and pricing
        url: https://docs.inceptionlabs.ai/get-started/models
    poolside_direct_api:
      status: partial
      score: 36
      tier: established
      basis: The scorable evidence is ecosystem distribution of poolside's own weights; the verified poolside Hugging Face organization records 1,851,060 model downloads in the last 30 days across 33 repositories, led by Laguna-S-2.1 variants, and Laguna is a recognized first-party coding-model family also carried by third-party routers. Poolside publishes no developer, customer, or request-volume figure and appears in no developer survey.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 6
      signals:
      - metric: hf_org_model_downloads_30d
        value: 1851060
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/api/models?author=poolside
      - metric: hf_top_model_downloads_30d_laguna_s_2_1_nvfp4
        value: 563149
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/poolside/Laguna-S-2.1-NVFP4
      - metric: hf_likes_laguna_s_2_1
        value: 982
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/poolside/Laguna-S-2.1
      - metric: github_org_followers_poolsideai
        value: 233
        observed_at: '2026-08-23'
        source_url: https://github.com/poolsideai
      sources:
      - type: official_repository
        title: poolside organization on Hugging Face (Laguna model family)
        url: https://huggingface.co/poolside
      - type: official_announcement
        title: Introducing Laguna XS.2 and M.1 (2026-04-28)
        url: https://poolside.ai/blog/introducing-laguna-xs2-m1
    voyage_ai:
      status: partial
      score: 38
      tier: established
      basis: One strong quantitative signal plus operator recognition; the official voyageai Python SDK sees 3.81M monthly PyPI downloads (its npm client adds 900,054, not counted separately), and Voyage AI is a MongoDB company whose embedding and reranking models are distributed through MongoDB Atlas and the AWS and Azure marketplaces, so durable recognition is credited to the operator. No Voyage-specific user or customer count and no independent survey share were found.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 8
      signals:
      - metric: pypi_voyageai_monthly_downloads
        value: 3807142
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/voyageai/recent
      - metric: npm_voyageai_monthly_downloads
        value: 900054
        observed_at: '2026-08-23'
        source_url: https://www.npmjs.com/package/voyageai
      - metric: hf_org_model_downloads_30d_voyageai
        value: 227956
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/api/models?author=voyageai
      sources:
      - type: package_registry_stats
        title: PyPI download stats for the official voyageai package
        url: https://pypistats.org/api/packages/voyageai/recent
      - type: company_press_release
        title: MongoDB announces acquisition of Voyage AI; models remain available via Voyage APIs and the AWS and Azure marketplaces (2025-02-24)
        url: https://www.mongodb.com/company/newsroom/press-releases/mongodb-announces-acquisition-of-voyage-ai
    ai21_studio:
      status: partial
      score: 30
      tier: established
      basis: One quantitative signal plus qualitative recognition; the ai21 PyPI SDK sees 250K monthly downloads and AI21 Labs stewards the first-party Jamba model family distributed on Amazon Bedrock, but no official user or customer count and no independent survey share were found.
      components:
        adoption: 0
        developer_ecosystem: 25
        independent_awareness: 0
        durable_recognition: 5
      signals:
      - metric: pypi_ai21_monthly_downloads
        value: 249681
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/ai21
      sources:
      - type: package_registry_stats
        title: PyPI download stats for ai21
        url: https://pypistats.org/packages/ai21
      - type: official_docs
        title: Amazon Bedrock models at a glance listing the AI21 Labs Jamba model family
        url: https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards.html
    deepgram:
      status: scored
      score: 50
      tier: established
      basis: Deepgram officially reports 200,000+ developers building with its voice models and the deepgram-sdk PyPI package sees 3.2M monthly downloads; no independent survey share or durable-recognition claim is made.
      components:
        adoption: 20
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: deepgram_developers
        value: 200000
        qualifier: at_least
        observed_at: '2026-08-22'
        source_url: https://deepgram.com/learn/deepgram-accelerates-into-2025
      - metric: pypi_deepgram_sdk_monthly_downloads
        value: 3161955
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/deepgram-sdk
      sources:
      - type: official_press_release
        title: "Deepgram accelerates into 2025 empowering 200,000+ developers (2025-01-29)"
        url: https://deepgram.com/learn/deepgram-accelerates-into-2025
      - type: package_registry_stats
        title: PyPI download stats for deepgram-sdk
        url: https://pypistats.org/packages/deepgram-sdk
    jina_ai_search_foundation:
      status: partial
      score: 37
      tier: established
      basis: Scored on Jina AI's model-distribution ecosystem; the verified jinaai Hugging Face organization records 9,489,146 model downloads in the last 30 days across 123 repositories, led by jina-embeddings-v3 at 2,884,656, and the jina-embeddings and jina-reranker families are widely redistributed first-party models, including through the Elasticsearch Open Inference API. Jina publishes no user, developer, or customer count and appears in no developer survey. Scored on the same methodology as the sibling jina_ai_reader record with record-specific evidence.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 7
      signals:
      - metric: hf_org_model_downloads_30d_jinaai
        value: 9489146
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/api/models?author=jinaai
      - metric: hf_model_downloads_30d_jina_embeddings_v3
        value: 2884656
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/jinaai/jina-embeddings-v3
      - metric: github_org_followers_jina_ai
        value: 4202
        observed_at: '2026-08-23'
        source_url: https://github.com/jina-ai
      sources:
      - type: official_repository
        title: Jina AI organization on Hugging Face (embedding and reranker model families)
        url: https://huggingface.co/jinaai
      - type: partner_page
        title: Elasticsearch Open Inference API adds support for Jina AI embeddings and rerank models (2025-02-20)
        url: https://www.businesswire.com/news/home/20250220781575/en/Elasticsearch-Open-Inference-API-now-Supports-Jina-AI-Embeddings-and-Rerank-Model
    jina_ai_reader:
      status: partial
      score: 27
      tier: established
      basis: Scored on the Reader service's own primary official repository, jina-ai/reader, which has 11,898 GitHub stars; the operator-level durable recognition matches the sibling jina_ai_search_foundation record (widely redistributed first-party Jina model families, including ReaderLM). No user, request-volume, or survey evidence exists for the anonymous Reader endpoints, and the Reader API ships no package client, so only one ecosystem metric is counted.
      components:
        adoption: 0
        developer_ecosystem: 20
        independent_awareness: 0
        durable_recognition: 7
      signals:
      - metric: github_stars_official_reader_repo
        value: 11898
        observed_at: '2026-08-23'
        source_url: https://github.com/jina-ai/reader
      - metric: hf_model_downloads_30d_readerlm_v2
        value: 1057
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/jinaai/ReaderLM-v2
      - metric: hf_likes_readerlm_v2
        value: 812
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/jinaai/ReaderLM-v2
      sources:
      - type: official_repository
        title: jina-ai/reader, the official Reader API repository
        url: https://github.com/jina-ai/reader
      - type: official_repository
        title: ReaderLM-v2 first-party model card on Hugging Face
        url: https://huggingface.co/jinaai/ReaderLM-v2
    mancer_ai:
      status: not_found
      score: 0
      tier: unknown
      reason: Searched on 2026-08-23 and found no citable rubric-eligible evidence. Checked mancer.tech (homepage, pricing FAQ, docs, live models endpoint) for any user, customer, or request-volume figure; GitHub for an official organization for Mancer or its operator Sunlit Software, Inc. (the SunlitSoftware account has 0 followers and no public repositories, and no Mancer-owned repository exists); PyPI and npm for an official SDK (the only PyPI package named mancer is an unrelated Polish command framework by a different author); the OpenRouter provider page for Mancer, which lists models but publishes no token or request totals; the 2025 Stack Overflow Developer Survey; and third-party AI directories, which list the service without usable metrics.
    mara_inference_cloud:
      status: not_found
      score: 0
      tier: unknown
      reason: Searched on 2026-08-23 and found no citable rubric-eligible evidence. Checked cloud.mara.com (plans, dashboard, playground, live models endpoint) and mara.com company posts and investor material for any developer, customer, or request-volume figure for the inference service; GitHub for an official MARA Holdings organization or inference-cloud repository; PyPI and npm for a first-party SDK (the service exposes only an OpenAI-compatible API with no MARA-published client); and the 2025 Stack Overflow Developer Survey. MARA Holdings is a listed bitcoin-mining and energy company rather than an established developer platform, so no durable-recognition credit applies either.
    assemblyai:
      status: partial
      score: 30
      tier: established
      basis: Only one quantitative signal was captured; the assemblyai PyPI SDK sees 2.3M monthly downloads, but no current official developer or customer count is stated on AssemblyAI's own pages and no independent survey or durable-recognition evidence was found.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: pypi_assemblyai_monthly_downloads
        value: 2302820
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/assemblyai
      sources:
      - type: package_registry_stats
        title: PyPI download stats for assemblyai
        url: https://pypistats.org/packages/assemblyai
    speechmatics:
      status: partial
      score: 38
      tier: established
      basis: The official speechmatics-rt Python SDK sees 259,903 monthly PyPI downloads (its real-time npm client adds 138,811, not counted separately) and Speechmatics stewards its own first-party STT and TTS model family. Adoption is scored low and marked partial because the only first-party numeric figures are scope-mismatched - 2 million-plus on-device laptop end users of an OEM deployment and 30 million-plus minutes of transcription - rather than API developers or customers; no independent survey share was found.
      components:
        adoption: 10
        developer_ecosystem: 25
        independent_awareness: 0
        durable_recognition: 3
      signals:
      - metric: pypi_speechmatics_rt_monthly_downloads
        value: 259903
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/speechmatics-rt/recent
      - metric: npm_speechmatics_real_time_client_monthly_downloads
        value: 138811
        observed_at: '2026-08-23'
        source_url: https://www.npmjs.com/package/@speechmatics/real-time-client
      - metric: on_device_end_users_oem_deployment
        value: 2000000
        qualifier: at_least
        observed_at: '2026-01-07'
        source_url: https://www.speechmatics.com/company/articles-and-news/speechmatics-in-2025-the-numbers-that-shaped-voice-ais-breakthrough-year
      - metric: healthcare_minutes_processed_2025
        value: 30000000
        qualifier: at_least
        observed_at: '2026-01-07'
        source_url: https://www.speechmatics.com/company/articles-and-news/speechmatics-in-2025-the-numbers-that-shaped-voice-ais-breakthrough-year
      sources:
      - type: package_registry_stats
        title: PyPI download stats for the official speechmatics-rt SDK
        url: https://pypistats.org/api/packages/speechmatics-rt/recent
      - type: official_repository
        title: speechmatics/speechmatics-python-sdk
        url: https://github.com/speechmatics/speechmatics-python-sdk
      - type: official_blog
        title: 'Speechmatics in 2025: the numbers that shaped voice AI (2026-01-07)'
        url: https://www.speechmatics.com/company/articles-and-news/speechmatics-in-2025-the-numbers-that-shaped-voice-ais-breakthrough-year
    aws_bedrock:
      status: scored
      score: 70
      tier: well_known
      basis: Amazon officially reports tens of thousands of Bedrock customers, the Bedrock-specific @aws-sdk/client-bedrock-runtime npm client sees 39.4M monthly downloads, AWS holds 43.3% share in the independent 2025 Stack Overflow cloud-platform survey, and Bedrock is operated by a globally established cloud platform.
      components:
        adoption: 10
        developer_ecosystem: 30
        independent_awareness: 20
        durable_recognition: 10
      signals:
      - metric: bedrock_customers
        value: 10000
        qualifier: at_least
        observed_at: '2026-08-22'
        source_url: https://press.aboutamazon.com/2024/12/amazon-bedrock-empowers-customers-to-accelerate-generative-ai-adoption-with-more-than-100-new-models-and-powerful-new-capabilities-for-inference-and-working-with-data
      - metric: npm_aws_sdk_client_bedrock_runtime_monthly_downloads
        value: 39380105
        observed_at: '2026-08-22'
        source_url: https://www.npmjs.com/package/@aws-sdk/client-bedrock-runtime
      - metric: stackoverflow_2025_cloud_platform_share_percent
        value: 43.3
        observed_at: '2026-08-22'
        source_url: https://survey.stackoverflow.co/2025/technology
      sources:
      - type: official_press_release
        title: Amazon Bedrock tens of thousands of customers and 4.7x customer growth (2024-12-04)
        url: https://press.aboutamazon.com/2024/12/amazon-bedrock-empowers-customers-to-accelerate-generative-ai-adoption-with-more-than-100-new-models-and-powerful-new-capabilities-for-inference-and-working-with-data
      - type: package_registry_stats
        title: npm download stats for @aws-sdk/client-bedrock-runtime
        url: https://www.npmjs.com/package/@aws-sdk/client-bedrock-runtime
      - type: independent_survey
        title: 2025 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2025/technology
    azure_ai_foundry:
      status: scored
      score: 65
      tier: well_known
      basis: Microsoft officially reports more than 70,000 Azure AI Foundry customers, the Foundry-specific azure-ai-projects SDK sees 14.7M monthly PyPI downloads, Azure holds 26.3% share in the independent 2025 Stack Overflow cloud-platform survey, and Foundry is operated by a globally established cloud platform.
      components:
        adoption: 10
        developer_ecosystem: 30
        independent_awareness: 15
        durable_recognition: 10
      signals:
      - metric: foundry_customers
        value: 70000
        qualifier: at_least
        observed_at: '2026-08-22'
        source_url: https://azure.microsoft.com/en-us/blog/azure-ai-foundry-your-ai-app-and-agent-factory/
      - metric: pypi_azure_ai_projects_monthly_downloads
        value: 14688510
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/azure-ai-projects
      - metric: stackoverflow_2025_cloud_platform_share_percent
        value: 26.3
        observed_at: '2026-08-22'
        source_url: https://survey.stackoverflow.co/2025/technology
      sources:
      - type: official_blog
        title: "Azure AI Foundry has grown to more than 70,000 customers (2025-05-19)"
        url: https://azure.microsoft.com/en-us/blog/azure-ai-foundry-your-ai-app-and-agent-factory/
      - type: package_registry_stats
        title: PyPI download stats for azure-ai-projects
        url: https://pypistats.org/packages/azure-ai-projects
      - type: independent_survey
        title: 2025 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2025/technology
    google_vertex_ai:
      status: scored
      score: 75
      tier: well_known
      basis: The Vertex-specific google-cloud-aiplatform SDK sees 149.7M monthly PyPI downloads, Google Cloud holds 24.6% share in the independent 2025 Stack Overflow cloud-platform survey, and the service is operated by a globally established cloud platform. Adoption is scored one band below the million-user tier because Google publishes usage-scale metrics rather than an absolute Vertex customer or developer count - 330 Google Cloud customers each processing over one trillion tokens in twelve months, 16 billion tokens per minute of direct first-party API use, and billions of Vertex API calls per month.
      components:
        adoption: 20
        developer_ecosystem: 30
        independent_awareness: 15
        durable_recognition: 10
      signals:
      - metric: google_cloud_customers_above_one_trillion_tokens_12mo
        value: 330
        observed_at: '2026-04-22'
        source_url: https://cloud.google.com/blog/topics/google-cloud-next/welcome-to-google-cloud-next26
      - metric: first_party_model_tokens_per_minute_direct_api
        value: 16000000000
        observed_at: '2026-04-22'
        source_url: https://cloud.google.com/blog/topics/google-cloud-next/welcome-to-google-cloud-next26
      - metric: pypi_google_cloud_aiplatform_monthly_downloads
        value: 149729616
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/google-cloud-aiplatform/recent
      - metric: stackoverflow_2025_cloud_platform_share_percent_google_cloud
        value: 24.6
        observed_at: '2026-08-23'
        source_url: https://survey.stackoverflow.co/2025/technology
      sources:
      - type: official_blog
        title: Welcome to Google Cloud Next '26 (330 customers above one trillion tokens; 16B tokens/minute of direct API use), 2026-04-22
        url: https://cloud.google.com/blog/topics/google-cloud-next/welcome-to-google-cloud-next26
      - type: official_blog
        title: Welcome to Google Cloud Next '25 (40x growth in Gemini use on Vertex AI, billions of API calls per month), 2025-04-09
        url: https://cloud.google.com/blog/topics/google-cloud-next/welcome-to-google-cloud-next25
      - type: package_registry_stats
        title: PyPI download stats for google-cloud-aiplatform (Vertex AI SDK)
        url: https://pypistats.org/api/packages/google-cloud-aiplatform/recent
      - type: independent_survey
        title: 2025 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2025/technology
    oracle_oci_generative_ai:
      status: scored
      score: 45
      tier: established
      basis: No official OCI Generative AI customer count was found; the whole-OCI oci Python SDK that includes the Generative AI client sees 10.5M monthly PyPI downloads, OCI held 2.9% share in the independent 2024 Stack Overflow cloud-platform survey, and the service is operated by Oracle, a globally established cloud platform.
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 5
        durable_recognition: 10
      signals:
      - metric: pypi_oci_monthly_downloads_whole_cloud_sdk
        value: 10549490
        observed_at: '2026-08-22'
        source_url: https://pypistats.org/packages/oci
      - metric: stackoverflow_2024_cloud_platform_share_percent
        value: 2.9
        observed_at: '2026-08-22'
        source_url: https://survey.stackoverflow.co/2024/technology
      sources:
      - type: package_registry_stats
        title: PyPI download stats for oci (whole-OCI SDK including the Generative AI Inference client)
        url: https://pypistats.org/packages/oci
      - type: independent_survey
        title: 2024 Stack Overflow Developer Survey technology section
        url: https://survey.stackoverflow.co/2024/technology
      - type: official_press_release
        title: Oracle announces general availability of OCI Generative AI (2024-01-23)
        url: https://www.oracle.com/news/announcement/oracle-announces-availability-oci-generative-ai-service-2024-01-23/
    replicate:
      status: scored
      score: 60
      tier: well_known
      basis: Replicate's own funding post documents 2 million signups and 30,000 paying customers, and the official replicate npm client sees 2.42M monthly downloads (the Python client adds 1.51M, not counted separately); no newer official count is published, so the adoption figure is dated 2023-12-05 and is a floor. No independent survey share or durable-recognition claim applies.
      components:
        adoption: 30
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: platform_signups
        value: 2000000
        observed_at: '2023-12-05'
        source_url: https://replicate.com/blog/series-b
      - metric: paying_customers
        value: 30000
        observed_at: '2023-12-05'
        source_url: https://replicate.com/blog/series-b
      - metric: npm_replicate_monthly_downloads
        value: 2422663
        observed_at: '2026-08-23'
        source_url: https://www.npmjs.com/package/replicate
      - metric: pypi_replicate_monthly_downloads
        value: 1512907
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/replicate/recent
      - metric: github_org_followers_replicate
        value: 11535
        observed_at: '2026-08-23'
        source_url: https://github.com/replicate
      sources:
      - type: official_blog
        title: Replicate Series B post (2 million signups, 30,000 paying customers), 2023-12-05
        url: https://replicate.com/blog/series-b
      - type: package_registry_stats
        title: npm download stats for the official replicate client
        url: https://www.npmjs.com/package/replicate
    cerebras_inference:
      status: scored
      score: 40
      tier: established
      basis: Cerebras officially reports being the number one inference provider on Hugging Face with over 5 million monthly requests, and its official cerebras-cloud-sdk Python client sees 1.29M monthly PyPI downloads. The request-volume figure is well below any published user-count band, no independent survey lists Cerebras, and no durable-recognition credit is taken because Cerebras is an AI compute vendor without a first-party model family, matching how the comparable Groq record is scored.
      components:
        adoption: 10
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: hugging_face_monthly_inference_requests
        value: 5000000
        qualifier: at_least
        observed_at: '2025-09-30'
        source_url: https://www.cerebras.ai/press-release/series-g
      - metric: pypi_cerebras_cloud_sdk_monthly_downloads
        value: 1292327
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/cerebras-cloud-sdk/recent
      - metric: npm_cerebras_cloud_sdk_monthly_downloads
        value: 487258
        observed_at: '2026-08-23'
        source_url: https://www.npmjs.com/package/@cerebras/cerebras_cloud_sdk
      - metric: github_org_followers_cerebras
        value: 1051
        observed_at: '2026-08-23'
        source_url: https://github.com/Cerebras
      sources:
      - type: company_press_release
        title: Cerebras Series G announcement (#1 inference provider on Hugging Face with over 5 million monthly requests), 2025-09-30
        url: https://www.cerebras.ai/press-release/series-g
      - type: package_registry_stats
        title: PyPI download stats for the official cerebras-cloud-sdk package
        url: https://pypistats.org/api/packages/cerebras-cloud-sdk/recent
    clarifai:
      status: scored
      score: 40
      tier: established
      basis: Clarifai's own press-release boilerplate documents more than 500,000 users across 170 countries and more than 1.5 million AI models built, and its primary official clarifai Python SDK sees 68,436 monthly PyPI downloads (the lower-level clarifai-grpc client adds 123,898 and is not counted separately to avoid double counting). No independent survey share exists, and Clarifai is not a globally established cloud platform nor the steward of a widely recognized model family, so durable recognition is zero.
      components:
        adoption: 20
        developer_ecosystem: 20
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: platform_users
        value: 500000
        qualifier: at_least
        observed_at: '2025-12-01'
        source_url: https://www.prnewswire.com/news-releases/clarifai-selected-inference-provider-for-arcee-ais-new-trinity-family-of-us-built-open-weight-models-302629285.html
      - metric: ai_models_built_on_platform
        value: 1500000
        qualifier: at_least
        observed_at: '2025-12-01'
        source_url: https://www.prnewswire.com/news-releases/clarifai-selected-inference-provider-for-arcee-ais-new-trinity-family-of-us-built-open-weight-models-302629285.html
      - metric: pypi_clarifai_monthly_downloads
        value: 68436
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/clarifai/recent
      - metric: pypi_clarifai_grpc_monthly_downloads
        value: 123898
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/clarifai-grpc/recent
      sources:
      - type: company_press_release
        title: Clarifai press release with About Clarifai boilerplate (1.5M+ models built, 500,000+ users, 170 countries), 2025-12-01
        url: https://www.prnewswire.com/news-releases/clarifai-selected-inference-provider-for-arcee-ais-new-trinity-family-of-us-built-open-weight-models-302629285.html
      - type: package_registry_stats
        title: PyPI download stats for the official clarifai package
        url: https://pypistats.org/api/packages/clarifai/recent
    fireworks_ai:
      status: scored
      score: 60
      tier: well_known
      basis: Fireworks officially reports serving more than 40 trillion tokens every day, and its official fireworks-ai Python SDK sees 4.22M monthly PyPI downloads. The adoption component is mapped from documented request volume rather than a published user count, so it is placed one band below the top user tier; no independent survey lists Fireworks, and it hosts third-party open models rather than stewarding its own recognized model family, so durable recognition is zero.
      components:
        adoption: 30
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: tokens_served_per_day
        value: 40000000000000
        qualifier: at_least
        observed_at: '2026-07-15'
        source_url: https://fireworks.ai/blog/series-d-announcement
      - metric: pypi_fireworks_ai_monthly_downloads
        value: 4219417
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/api/packages/fireworks-ai/recent
      - metric: github_org_followers_fw_ai
        value: 225
        observed_at: '2026-08-23'
        source_url: https://github.com/fw-ai
      sources:
      - type: official_announcement
        title: Fireworks Series D announcement (more than 40 trillion tokens served every day), 2026-07-15
        url: https://fireworks.ai/blog/series-d-announcement
      - type: package_registry_stats
        title: PyPI download stats for the official fireworks-ai package
        url: https://pypistats.org/api/packages/fireworks-ai/recent
    nebius_token_factory:
      status: partial
      score: 35
      tier: established
      basis: 'Only one quantitative signal is available for the operator and none for the service: the official Nebius Python SDK (github.com/nebius/pysdk, authored by @nebius.com engineers, covering the whole Nebius AI Cloud API including its `ai` service group) sees 2,149,526 monthly PyPI downloads, and Nebius is a NASDAQ-listed AI cloud that launched Token Factory as a production inference platform in November 2025. No Token Factory or AI Studio developer, customer, or user count is published anywhere Nebius reports; Q2 2026 revenue was deliberately excluded as an adoption proxy, and Nebius does not appear in the 2025 Stack Overflow cloud-platform list.'
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 5
      signals:
      - metric: pypi_nebius_sdk_monthly_downloads_whole_cloud_sdk
        value: 2149526
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/packages/nebius
      sources:
      - type: package_registry_stats
        title: PyPI download stats for nebius (official whole-cloud SDK including the ai service group)
        url: https://pypistats.org/packages/nebius
      - type: official_repository
        title: Nebius Python SDK
        url: https://github.com/nebius/pysdk
      - type: company_press_release
        title: Nebius launches Nebius Token Factory to deliver production AI inference at scale (2025-11-05)
        url: https://nebius.com/newsroom/nebius-launches-nebius-token-factory-to-deliver-production-ai-inference-at-scale
    novita_ai:
      status: scored
      score: 30
      tier: established
      basis: Novita's own press release states it serves more than 350K developers processing more than 1T tokens per day, and its official novita-client Python SDK sees 4,730 monthly PyPI downloads. No independent developer survey or search-interest measure covers Novita (its February 2026 Ramp top-vendor listing is corroborating spend data, not a survey share, so it was not scored), and Novita stewards no first-party model family.
      components:
        adoption: 20
        developer_ecosystem: 10
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: developers
        value: 350000
        qualifier: at_least
        observed_at: '2026-03-24'
        source_url: https://www.prnewswire.com/news-releases/novita-ai-named-a-top-ai-infrastructure-vendor-on-ramp-302722905.html
      - metric: tokens_per_day
        value: 1000000000000
        qualifier: at_least
        observed_at: '2026-03-24'
        source_url: https://www.prnewswire.com/news-releases/novita-ai-named-a-top-ai-infrastructure-vendor-on-ramp-302722905.html
      - metric: pypi_novita_client_monthly_downloads
        value: 4730
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/packages/novita-client
      sources:
      - type: company_press_release
        title: Novita AI named a top AI infrastructure vendor on Ramp (2026-03-24, cites 350K+ developers and 1T+ tokens/day)
        url: https://www.prnewswire.com/news-releases/novita-ai-named-a-top-ai-infrastructure-vendor-on-ramp-302722905.html
      - type: package_registry_stats
        title: PyPI download stats for novita-client (official Novita AI Python SDK)
        url: https://pypistats.org/packages/novita-client
    hyperbolic:
      status: scored
      score: 25
      tier: established
      basis: 'Hyperbolic''s own site states 250,000+ builders use its AI infrastructure, corroborated by a dated official blog milestone of 195,000+ developers served in April 2025. Its developer ecosystem is small: the most starred repository in the official HyperbolicLabs GitHub organization, Hyperbolic-AgentKit, has 111 stars (the organization itself has 109 followers) and no official Hyperbolic package is published on PyPI or npm outside a third-party AI SDK provider, so a sub-threshold ecosystem score is used. No independent survey share and no durable-recognition claim.'
      components:
        adoption: 20
        developer_ecosystem: 5
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: platform_builders
        value: 250000
        qualifier: at_least
        observed_at: '2026-08-23'
        source_url: https://www.hyperbolic.ai/
      - metric: developers_served
        value: 195000
        qualifier: at_least
        observed_at: '2025-04-28'
        source_url: https://www.hyperbolic.ai/blog/monthly-recap-april-2025
      - metric: github_stars_primary_official_repo
        value: 111
        observed_at: '2026-08-23'
        source_url: https://github.com/HyperbolicLabs/Hyperbolic-AgentKit
      sources:
      - type: official_product
        title: Hyperbolic home page citing 250,000+ builders
        url: https://www.hyperbolic.ai/
      - type: official_blog
        title: 'Hyperbolic Monthly Recap: April 2025 (195,000+ developers served)'
        url: https://www.hyperbolic.ai/blog/monthly-recap-april-2025
      - type: code_hosting_stats
        title: HyperbolicLabs/Hyperbolic-AgentKit repository stars
        url: https://github.com/HyperbolicLabs/Hyperbolic-AgentKit
    waterfall:
      status: not_found
      score: 0
      tier: unknown
      reason: Searched on 2026-08-23 and found no citable rubric-eligible evidence. Checked Waterfall's own site, pricing, and documentation pages (getwaterfall.org) for any user, developer, request, or token-volume claim; searched the web for "getwaterfall.org" adoption and milestone announcements; queried the GitHub API for a `getwaterfall` organization or user (none exists) and searched GitHub repositories for "getwaterfall.org" (0 results) and for a Waterfall LLM-gateway project (only unrelated async-waterfall and Minecraft projects); searched npm and PyPI for a Waterfall client package (none); and confirmed absence from the 2025 Stack Overflow Developer Survey. The dataset already records it as a bootstrapped single-maintainer service.
    logfare:
      status: not_found
      score: 0
      tier: unknown
      reason: Searched on 2026-08-23 and found no citable rubric-eligible evidence. Checked logfare.ai's home page, docs, and terms for any user, developer, or request-volume figure (none published); searched the web for "logfare.ai" adoption claims (only third-party free-API listicles, which carry no metrics); queried the GitHub API for `logfare` and `logfare-ai` organizations or users (neither exists) and searched GitHub repositories for "logfare.ai" and "logfare llm" (0 results); searched npm and PyPI for a Logfare client package (none); and confirmed absence from the 2025 Stack Overflow Developer Survey.
    bazaarlink:
      status: partial
      score: 5
      tier: niche
      basis: 'One weak ecosystem signal exists: BazaarLink''s official open-source LLMprobe-engine repository (github.com/Bazaarlinkorg/LLMprobe-engine, AGPL-3.0 (c) BazaarLink, published as @bazaarlink/probe-engine in its README) has 105 GitHub stars, well below the 1K+ developer-ecosystem anchor, so a sub-threshold 5 is scored. The @bazaarlink/probe-engine package is not actually published on npm, the Bazaarlinkorg organization has 1 follower, and BazaarLink publishes no user, developer, or request counts, appears in no developer survey, and has no durable-recognition basis.'
      components:
        adoption: 0
        developer_ecosystem: 5
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: github_stars_official_repo
        value: 105
        observed_at: '2026-08-23'
        source_url: https://github.com/Bazaarlinkorg/LLMprobe-engine
      - metric: github_org_followers
        value: 1
        observed_at: '2026-08-23'
        source_url: https://github.com/Bazaarlinkorg
      sources:
      - type: official_repository
        title: BazaarLink LLMprobe-engine repository
        url: https://github.com/Bazaarlinkorg/LLMprobe-engine
      - type: official_product
        title: About BazaarLink (operator 集聯科技有限公司, Taiwan)
        url: https://bazaarlink.ai/en/about
    dreamprompting:
      status: not_found
      score: 0
      tier: unknown
      reason: Searched on 2026-08-23 and found no citable rubric-eligible evidence. Checked dreamprompting.com's home, about, API docs, and models pages for any user, developer, or request-volume figure (none published); searched the web for "dreamprompting.com" adoption claims; queried the GitHub API for a `dreamprompting` organization or user (none exists) and searched GitHub repositories for "dreamprompting.com" and "dreamprompting" (only one unrelated project with 0 stars); searched npm and PyPI for a DreamPrompting client package (none); and confirmed absence from the 2025 Stack Overflow Developer Survey.
    ch_at:
      status: partial
      score: 10
      tier: niche
      basis: 'Only one quantitative signal exists: the service''s official repository, Deep-ai-inc/ch.at (created 2025-06-18), has 1,101 GitHub stars, crossing the 1K+ developer-ecosystem anchor. ch.at is an anonymous keyless community endpoint that publishes no user, request, or token-volume count, ships no package on npm or PyPI, appears in no developer survey, and has no durable-recognition basis.'
      components:
        adoption: 0
        developer_ecosystem: 10
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: github_stars_official_repo
        value: 1101
        observed_at: '2026-08-23'
        source_url: https://github.com/Deep-ai-inc/ch.at
      sources:
      - type: official_repository
        title: ch.at repository (Universal Basic Chat)
        url: https://github.com/Deep-ai-inc/ch.at
    opentyphoon:
      status: partial
      score: 30
      tier: established
      basis: One strong ecosystem signal plus qualitative recognition. The official Typhoon Hugging Face organization (typhoon-ai, which lists https://opentyphoon.ai/ as its website) records 783,521 model downloads in the last 30 days across 76 models and 390 followers, clearing the 100K+ anchor. SCB 10X publishes no OpenTyphoon API developer or user count and Typhoon appears in no independent developer survey, so adoption and independent awareness are unscored; durable recognition reflects Typhoon being SCB 10X's first-party Thai model family (typhoon-ocr, typhoon2.5), regionally rather than globally recognized.
      components:
        adoption: 0
        developer_ecosystem: 25
        independent_awareness: 0
        durable_recognition: 5
      signals:
      - metric: hf_org_model_downloads_30d
        value: 783521
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/api/models?author=typhoon-ai
      - metric: hf_org_followers
        value: 390
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/typhoon-ai
      - metric: github_stars_typhoon_ocr_repo
        value: 142
        observed_at: '2026-08-23'
        source_url: https://github.com/scb-10x/typhoon-ocr
      sources:
      - type: official_repository
        title: Typhoon organization on Hugging Face (linked from opentyphoon.ai)
        url: https://huggingface.co/typhoon-ai
      - type: official_repository
        title: scb-10x/typhoon-ocr open-source Thai OCR model repository
        url: https://github.com/scb-10x/typhoon-ocr
      - type: official_product
        title: About Typhoon (SCB 10X open-source Thai AI initiative)
        url: https://opentyphoon.ai/about
    alcf_inference_endpoints:
      status: partial
      score: 25
      tier: established
      basis: 'Scored as a national research platform. The ALCF 2025 Annual Report documents 2,103 facility users and 483 active projects (1,208 academic, 686 government, 212 industry users; 274 publications), which is a real but facility-wide count rather than an inference-endpoint count, so it earns the smaller-numeric-adoption band and keeps the record partial. The argonne-lcf GitHub organization has 256 followers (the argonne-lcf/inference-endpoints repository itself has only 29 stars), below the 1K+ anchor, so a sub-threshold ecosystem score is used. Durable recognition is full: ALCF is a US DOE Office of Science user facility operating the Aurora exascale supercomputer and allocating through INCITE, ALCC, Director''s Discretionary, and NAIRR.'
      components:
        adoption: 10
        developer_ecosystem: 5
        independent_awareness: 0
        durable_recognition: 10
      signals:
      - metric: facility_users
        value: 2103
        observed_at: '2026-08-23'
        source_url: https://ar25.alcf.anl.gov/year-in-review/about-alcf
      - metric: active_projects
        value: 483
        observed_at: '2026-08-23'
        source_url: https://ar25.alcf.anl.gov/year-in-review/about-alcf
      - metric: github_org_followers
        value: 256
        observed_at: '2026-08-23'
        source_url: https://github.com/argonne-lcf
      - metric: github_stars_inference_endpoints_repo
        value: 29
        observed_at: '2026-08-23'
        source_url: https://github.com/argonne-lcf/inference-endpoints
      sources:
      - type: official_annual_report
        title: 2025 ALCF Annual Report, About ALCF (2,103 users, 483 active projects, 274 publications)
        url: https://ar25.alcf.anl.gov/year-in-review/about-alcf
      - type: official_repository
        title: argonne-lcf GitHub organization
        url: https://github.com/argonne-lcf
      - type: official_product
        title: Aurora exascale supercomputer at ALCF
        url: https://www.alcf.anl.gov/aurora
      - type: official_docs
        title: ALCF inference endpoints documentation
        url: https://docs.alcf.anl.gov/services/inference-endpoints/
    fikra_api:
      status: not_found
      score: 0
      tier: unknown
      reason: Searched on 2026-08-23 and found no citable rubric-eligible evidence. Fikra API launched in June 2026 and has independent regional press coverage (Disrupt Africa, 2026-06-30; iAfrica; BusinessTech Africa), but none of it states a user, developer, customer, or request count — only that the service "attracted its first users within days of launching" — and press coverage is not a survey share or search-interest measure. Checked fikraapi.co.ke and docs.fikraapi.co.ke for adoption figures (none); queried the GitHub API for `fikraapi` and `fikra-api` organizations or users (neither exists) and searched GitHub repositories for "fikraapi.co.ke" (0 results) and "fikra inference api" (one unrelated 1-star project); searched npm (no Fikra client) and PyPI, where the only `fikra` package is an unrelated offline "Lacesse Edge AI SDK" with 11 downloads in the last month, not the hosted Fikra API client; confirmed absence from the 2025 Stack Overflow Developer Survey.
    sarvam_ai:
      status: scored
      score: 61
      tier: well_known
      basis: Sarvam reported more than 1 million registered developers on its developer platform at its Epoch 2026 conference (2026-07-30), alongside 325M+ annual AI conversation minutes through Samvaad, and its official sarvamai Python SDK sees 180,724 monthly PyPI downloads. No independent developer survey covers Sarvam. Durable recognition reflects the first-party Sarvam model family (Sarvam-30B and Sarvam-105B, trained on IndiaAI Mission compute and open-sourced) under India's sovereign-model program; the recognition is national rather than global. The official npm sarvamai package (40,231 monthly downloads) and the sarvamai GitHub organization (741 followers) were recorded but not double-counted.
      components:
        adoption: 30
        developer_ecosystem: 25
        independent_awareness: 0
        durable_recognition: 6
      signals:
      - metric: registered_developers
        value: 1000000
        qualifier: at_least
        observed_at: '2026-07-30'
        source_url: https://inc42.com/buzz/sarvam-to-build-trillion-plus-ai-model-in-india-launches-inference-service/
      - metric: samvaad_conversation_minutes_per_year
        value: 325000000
        qualifier: at_least
        observed_at: '2026-07-30'
        source_url: https://inc42.com/buzz/sarvam-to-build-trillion-plus-ai-model-in-india-launches-inference-service/
      - metric: pypi_sarvamai_monthly_downloads
        value: 180724
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/packages/sarvamai
      - metric: npm_sarvamai_monthly_downloads
        value: 40231
        observed_at: '2026-08-23'
        source_url: https://www.npmjs.com/package/sarvamai
      - metric: github_org_followers
        value: 741
        observed_at: '2026-08-23'
        source_url: https://github.com/sarvamai
      sources:
      - type: reputable_reporting
        title: 'Inc42: Sarvam to build trillion-parameter model, developer platform passes 1 Mn registered developers (2026-07-30)'
        url: https://inc42.com/buzz/sarvam-to-build-trillion-plus-ai-model-in-india-launches-inference-service/
      - type: package_registry_stats
        title: PyPI download stats for sarvamai (official Sarvam Python SDK)
        url: https://pypistats.org/packages/sarvamai
      - type: official_repository
        title: Sarvam AI organization on Hugging Face (Sarvam first-party model family)
        url: https://huggingface.co/sarvamai
      - type: official_docs
        title: Sarvam open-source models documentation
        url: https://docs.sarvam.ai/api/getting-started/models/open-source
    byteplus_modelark:
      status: partial
      score: 33
      tier: established
      basis: 'One quantitative signal plus operator recognition. The official BytePlus Python SDK (github.com/byteplus-sdk/byteplus-python-sdk-v2), which ships the ModelArk clients `byteplussdkark` and `byteplussdkarkruntime`, sees 183,054 monthly PyPI downloads, clearing the 100K+ anchor. BytePlus publishes no ModelArk customer, enterprise, or developer count and ModelArk appears in no independent developer survey, so adoption and independent awareness are unscored. Durable recognition is for the operator: BytePlus is ByteDance''s enterprise cloud arm and ModelArk distributes ByteDance''s first-party Seed/Skylark model family. The China-market volcengine-python-sdk (534,696 monthly downloads) was deliberately excluded because it belongs to the separate Volcano Engine service, not the BytePlus ModelArk offer being scored.'
      components:
        adoption: 0
        developer_ecosystem: 25
        independent_awareness: 0
        durable_recognition: 8
      signals:
      - metric: pypi_byteplus_python_sdk_v2_monthly_downloads
        value: 183054
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/packages/byteplus-python-sdk-v2
      sources:
      - type: package_registry_stats
        title: PyPI download stats for byteplus-python-sdk-v2 (includes byteplussdkark and byteplussdkarkruntime ModelArk clients)
        url: https://pypistats.org/packages/byteplus-python-sdk-v2
      - type: official_repository
        title: byteplus-sdk/byteplus-python-sdk-v2 official SDK repository
        url: https://github.com/byteplus-sdk/byteplus-python-sdk-v2
      - type: official_docs
        title: BytePlus ModelArk overview and first-party model list
        url: https://docs.byteplus.com/en/docs/ModelArk/1330310
    tencent_hunyuan:
      status: partial
      score: 40
      tier: established
      basis: 'One strong ecosystem signal plus full operator recognition. Tencent''s official Hugging Face organization records 1,279,484 downloads in the last 30 days across 96 Hunyuan-family repositories (top entries HunyuanOCR, Hunyuan3D-2, Hy3-preview, Hunyuan-A13B-Instruct), clearing the 1M+ anchor; tencent/Hunyuan3D-2 alone has 3,465,932 all-time downloads, matching Tencent''s Davos 2026 claim of three million-plus. Tencent publishes no Hunyuan API developer or enterprise-customer count, and Tencent Cloud does not appear in the 2025 Stack Overflow cloud-platform results, so adoption and independent awareness are unscored. Durable recognition is full: Tencent is a globally established platform and Hunyuan is a widely recognized first-party model family. The service-specific tencentcloud-sdk-python-hunyuan package (3,044 monthly downloads) and the Tencent-Hunyuan GitHub organization (3,181 followers) were recorded but not double-counted.'
      components:
        adoption: 0
        developer_ecosystem: 30
        independent_awareness: 0
        durable_recognition: 10
      signals:
      - metric: hf_hunyuan_family_model_downloads_30d
        value: 1279484
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/api/models?author=tencent
      - metric: hf_hunyuan3d_2_downloads_all_time
        value: 3465932
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/tencent/Hunyuan3D-2
      - metric: github_org_followers_tencent_hunyuan
        value: 3181
        observed_at: '2026-08-23'
        source_url: https://github.com/Tencent-Hunyuan
      - metric: pypi_tencentcloud_sdk_python_hunyuan_monthly_downloads
        value: 3044
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/packages/tencentcloud-sdk-python-hunyuan
      sources:
      - type: official_repository
        title: Tencent organization on Hugging Face (Hunyuan first-party model family)
        url: https://huggingface.co/tencent
      - type: official_repository
        title: tencent/Hunyuan3D-2 model card with all-time download count
        url: https://huggingface.co/tencent/Hunyuan3D-2
      - type: official_docs
        title: Tencent Hunyuan product overview on Tencent Cloud
        url: https://cloud.tencent.com/document/product/1729/104753
    upstage:
      status: partial
      score: 30
      tier: established
      basis: One quantitative signal plus model-family recognition. Upstage's official Hugging Face organization records 142,414 downloads in the last 30 days across 25 Solar-family models (Solar-Open2-250B, SOLAR-10.7B-Instruct, Solar-Open-100B, solar-pro-preview-instruct), clearing the 100K+ anchor. Upstage publishes no API user, developer, or customer count ("hundreds of global companies" is unquantified) and appears in no independent developer survey, so adoption and independent awareness are unscored. Durable recognition is for the first-party Solar model family distributed through Amazon Bedrock Marketplace, SageMaker JumpStart, and AWS Marketplace. The official langchain-upstage package (49,761 monthly PyPI downloads) and the UpstageAI GitHub organization (258 followers) were recorded but not double-counted.
      components:
        adoption: 0
        developer_ecosystem: 25
        independent_awareness: 0
        durable_recognition: 5
      signals:
      - metric: hf_org_model_downloads_30d
        value: 142414
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/api/models?author=upstage
      - metric: pypi_langchain_upstage_monthly_downloads
        value: 49761
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/packages/langchain-upstage
      - metric: github_org_followers
        value: 258
        observed_at: '2026-08-23'
        source_url: https://github.com/UpstageAI
      sources:
      - type: official_repository
        title: Upstage organization on Hugging Face (Solar first-party model family)
        url: https://huggingface.co/upstage
      - type: official_press_release
        title: Upstage releases Solar Pro on AWS (Amazon Bedrock Marketplace, SageMaker JumpStart, AWS Marketplace)
        url: https://www.upstage.ai/news/solar-pro-aws
      - type: package_registry_stats
        title: PyPI download stats for langchain-upstage
        url: https://pypistats.org/packages/langchain-upstage
    maritaca_academic_credits:
      status: partial
      score: 8
      tier: niche
      basis: 'Only one weak ecosystem signal exists: the official MariTalk API client repository (github.com/maritaca-ai/maritalk-api) has 330 GitHub stars and its maritalk PyPI package sees 525 downloads in the last month, both far below the 1K+ anchor, so a sub-threshold 5 is scored (the maritaca-ai GitHub organization has 283 followers and the maritaca-ai Hugging Face organization only 576 downloads in 30 days). Maritaca publishes no user, client, or academic-program participant count, appears in no developer survey, and its recognition is limited: durable recognition reflects the first-party Sabiá Portuguese model family, which is recognized in Brazil rather than globally.'
      components:
        adoption: 0
        developer_ecosystem: 5
        independent_awareness: 0
        durable_recognition: 3
      signals:
      - metric: github_stars_official_api_client_repo
        value: 330
        observed_at: '2026-08-23'
        source_url: https://github.com/maritaca-ai/maritalk-api
      - metric: pypi_maritalk_downloads_last_month
        value: 525
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/packages/maritalk
      - metric: github_org_followers
        value: 283
        observed_at: '2026-08-23'
        source_url: https://github.com/maritaca-ai
      - metric: hf_org_model_downloads_30d
        value: 576
        observed_at: '2026-08-23'
        source_url: https://huggingface.co/api/models?author=maritaca-ai
      sources:
      - type: official_repository
        title: maritaca-ai/maritalk-api official API client repository
        url: https://github.com/maritaca-ai/maritalk-api
      - type: package_registry_stats
        title: PyPI download stats for maritalk
        url: https://pypistats.org/packages/maritalk
      - type: official_catalog
        title: Maritaca AI Sabiá first-party model catalog
        url: https://docs.maritaca.ai/pt/modelos
    wavespeedai:
      status: partial
      score: 20
      tier: established
      basis: 'Only one quantitative signal was captured: the official WaveSpeedAI JavaScript SDK (npm wavespeed, repo WaveSpeedAI/wavespeed-javascript) sees 12,690 downloads in the last month (2,514 in the last week), clearing the 10K+ anchor. WaveSpeedAI publishes no developer, user, or generation-volume count — its site quantifies only catalog size (1000+ models) and uptime — appears in no independent developer survey, and stewards no first-party model family, so the other three components are unscored. The official wavespeed PyPI package (5,551 monthly downloads) and the WaveSpeedAI GitHub organization (372 followers) were recorded but not double-counted.'
      components:
        adoption: 0
        developer_ecosystem: 20
        independent_awareness: 0
        durable_recognition: 0
      signals:
      - metric: npm_wavespeed_monthly_downloads
        value: 12690
        observed_at: '2026-08-23'
        source_url: https://www.npmjs.com/package/wavespeed
      - metric: pypi_wavespeed_monthly_downloads
        value: 5551
        observed_at: '2026-08-23'
        source_url: https://pypistats.org/packages/wavespeed
      - metric: github_org_followers
        value: 372
        observed_at: '2026-08-23'
        source_url: https://github.com/WaveSpeedAI
      sources:
      - type: package_registry_stats
        title: npm download stats for wavespeed (official WaveSpeedAI JavaScript SDK)
        url: https://www.npmjs.com/package/wavespeed
      - type: official_repository
        title: WaveSpeedAI/wavespeed-javascript official SDK repository
        url: https://github.com/WaveSpeedAI/wavespeed-javascript
      - type: package_registry_stats
        title: PyPI download stats for wavespeed
        url: https://pypistats.org/packages/wavespeed
value_estimate_audit:
  as_of: "2026-08-22"
  checked_at: "2026-08-22T21:30:00-05:00"
  timezone: America/Chicago
  default_currency: USD
  methodology:
    - Value means the documented free allowance multiplied by the provider's current paid unit price when the same paid service exists.
    - A provider-funded monetary credit is valued at face value; model rows show the maximum share a single eligible model could consume and are not additive.
    - Daily, weekly, and monthly figures are equivalent-period comparisons, not guaranteed spend or cash. Recurring monthly credits use 365.25 days per year; fixed-duration trials are spread across their stated validity window.
    - A total-token quota without an input/output split is an up-to value: assign the shared envelope to the higher-priced direction first, subject to documented directional ceilings. This is deterministic best-case math, not a typical-traffic assumption.
    - Continuous rate limits are converted to daily, weekly, and monthly token envelopes only when token throughput or an official per-request token maximum makes the envelope finite; every applicable request and token ceiling is applied and the tightest wins.
    - Shared quotas are counted once. A subtotal or at-least qualifier means unpriced models, tools, conditional credits, or ambiguous independent quotas were excluded.
    - Not quantifiable is a research result, not zero value. It is used when the quota, model identity, paid rate, recurrence, or allocation scope is too uncertain for a defensible calculation.
  records:
    openrouter:
      status: quantified
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 53.08416
        weekly: 371.58912
        monthly: 1615.74912
      basis: Best case for the documented no-purchase free tier (20 requests/minute, 50 requests/day across all :free routes; the daily cap binds) saturated by the most expensive currently paid-priced model available on a :free route, thinkingmachines/inkling:free, whose live-catalog specs (262,144-token context with max output equal to the full context) are the official per-request ceiling, priced at OpenRouter's own paid route for the exact same model ($1.00 input / $4.05 output per 1M tokens, snapshot 2026-08-22). Model rows are alternative uses of the one shared request pool and are not additive. Assumes uninterrupted saturation with no latency, availability, or fair-use loss. Free models without a paid sibling on OpenRouter (7 routes incl. cloaked stealth/ox-alpha), the zero-priced Lyria music previews, and the openrouter/free random router were excluded from the comparison. Accounts that have ever purchased at least $10 of credits get 1,000 requests/day instead of 50 (20x this envelope); that conditional variant is excluded from the headline.
      model_values:
      - id: thinkingmachines/inkling:free
        allowance: 50 shared free-tier requests/day, each up to the 262,144-token context (output may fill it)
        paid_rate: $1.00 input + $4.05 output per 1M tokens (OpenRouter paid route for the same model)
        currency: USD
        daily: 53.08416
        weekly: 371.58912
        monthly: 1615.74912
      - id: nvidia/nemotron-3-ultra-550b-a55b:free
        allowance: Alternative use of the same 50 requests/day; 1,000,000-token context, 65,536-token max output
        paid_rate: $0.60 input + $3.60 output per 1M tokens (OpenRouter paid route)
        basis: Compared candidate, not additive with other rows
        currency: USD
        daily: 39.8304
      - id: z-ai/glm-5.2:free
        allowance: Alternative use of the same 50 requests/day; 256,000-token context with max output equal to the full context
        paid_rate: $0.966 input + $3.036 output per 1M tokens (OpenRouter paid route)
        basis: Compared candidate, not additive with other rows
        currency: USD
        daily: 38.8608
      - id: thinkingmachines/inkling-small:free
        allowance: Alternative use of the same 50 requests/day; 262,144-token context with max output equal to the full context
        paid_rate: $0.45 input + $1.20 output per 1M tokens (OpenRouter paid route)
        basis: Compared candidate, not additive with other rows
        currency: USD
        daily: 15.7286
      calculation:
        method: best_case_rate_limit_envelope
        model_id: thinkingmachines/inkling:free
        currency: USD
        limits:
          requests_per_minute: 20
          requests_per_day: 50
          max_input_tokens_per_request: 262144
          max_output_tokens_per_request: 262144
          max_total_tokens_per_request: 262144
        prices:
          input_per_million: 1.0
          output_per_million: 4.05
        result:
          daily:
            input_tokens: 0
            output_tokens: 13107200
            total_tokens: 13107200
            value: 53.08416
          weekly:
            input_tokens: 0
            output_tokens: 91750400
            total_tokens: 91750400
            value: 371.58912
          monthly:
            input_tokens: 0
            output_tokens: 398950400
            total_tokens: 398950400
            value: 1615.74912
      sources:
      - type: official_docs
        title: OpenRouter API rate limits (free model usage)
        url: https://openrouter.ai/docs/api-reference/limits
      - type: live_catalog
        title: Models endpoint with free-route specs and same-model paid prices
        url: https://openrouter.ai/api/v1/models
    nous_portal:
      status: partial
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 691.2
        weekly: 4838.4
        monthly: 21038.4
      basis: Saturation of the OBSERVED free-plan portal envelope (50 requests and 500k tokens per minute, treated as a shared input+output pool) assigned entirely to output of the costliest current zero-price catalog model that has a paid counterpart on the same Nous API, meituan/longcat-2.0, at Nous' own current listed paid rate ($0.24 input / $0.96 output per 1M tokens, a discounted current price; undiscounted list is $0.30/$1.20). Partial because the limits are observed on one portal surface while another surface says only "Standard rate limits" and no public limits document resolves the conflict; the signed-in account remains authoritative. stealth/ox-alpha, the only other zero-price model without a same-catalog paid counterpart, cannot be priced like-for-like and does not affect the maximum. Assumes uninterrupted saturation with no latency, concurrency, availability, or fair-use loss.
      model_values:
      - id: meituan/longcat-2.0:free
        allowance: Shared observed 500k tokens/minute, 50 requests/minute; 131,072 max output tokens per request
        paid_rate: $0.24 input / $0.96 output per 1M tokens (Nous' own current paid listing for meituan/longcat-2.0 on the same API)
        currency: USD
        daily: 691.2
        weekly: 4838.4
        monthly: 21038.4
      - id: stepfun/step-3.7-flash:free
        allowance: Same shared envelope; alternative maximum, not additive
        paid_rate: $0.16 input / $0.92 output per 1M tokens (Nous' own current paid listing for stepfun/step-3.7-flash)
        basis: Alternative maximum for the same shared envelope; not additive.
        currency: USD
        daily: 662.4
        weekly: 4636.8
        monthly: 20161.8
      - id: tencent/hy3:free
        allowance: Same shared envelope; alternative maximum, not additive
        paid_rate: $0.1056 input / $0.4224 output per 1M tokens (Nous' own current paid listing for tencent/hy3)
        basis: Alternative maximum for the same shared envelope; not additive.
        currency: USD
        daily: 304.13
        weekly: 2128.9
        monthly: 9256.9
      calculation:
        method: best_case_rate_limit_envelope
        model_id: meituan/longcat-2.0:free
        currency: USD
        pricing_snapshot: '2026-08-22'
        limits:
          requests_per_minute: 50
          tokens_per_minute: 500000
          max_output_tokens_per_request: 131072
          max_total_tokens_per_request: 1048576
        limits_notes: Requests and tokens per minute are OBSERVED portal free-plan numbers, not published documentation; a conflicting portal surface states only "Standard rate limits". Per-request ceilings are the free variant's live catalog specs (context 1,048,576; max completion 131,072) and never bind; the 500k shared tokens-per-minute pool is the tightest constraint and is allocated output-first.
        prices:
          input_per_million: 0.24
          output_per_million: 0.96
        result:
          daily:
            input_tokens: 0
            output_tokens: 720000000
            total_tokens: 720000000
            value: 691.2
      sources:
      - type: live_catalog
        title: Nous inference API models endpoint (zero-price models and same-catalog paid counterpart rates)
        url: https://inference-api.nousresearch.com/v1/models
      - type: official_product
        title: Nous Portal (observed free-plan limits surface)
        url: https://portal.nousresearch.com/
    orcarouter: {status: not_quantifiable, reason: "Free request limits conflict across first-party surfaces and the free aliases do not have stable like-for-like paid rates."}
    nvidia_build:
      status: partial
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 140.08
        weekly: 980.58
        monthly: 4263.78
      basis: Best-case rate-limit envelope for a single flagship model, meta/llama-3.3-70b-instruct, using NVIDIA's own published per-model limits on the build.nvidia.com model page ("Up to 40 rpm" and "10,000 requests per day") together with NVIDIA's documented per-request ceilings (131,072-token context length on the model page; max_tokens parameter maximum of 4,096 in the NVIDIA API reference). NVIDIA sells no per-token rate for this hosted route, so the paid comparison is the current OpenRouter price for the exact model ($0.10 input / $0.32 output per 1M tokens, snapshot 2026-08-22). Assumes uninterrupted saturation with maximal-context requests. This is one model's envelope only; the other 100+ catalog models have their own displayed limits and are excluded rather than summed, because the account-versus-model scope of the limits is not documented.
      reason: Partial because NVIDIA's own limit text says rates "may vary by model and traffic from other users may cause throttling", the phrasing is "up to", the allowance is development/prototyping-only, and the paid reference is a router price rather than an NVIDIA rate for the same service.
      model_values:
      - id: meta/llama-3.3-70b-instruct
        allowance: Up to 40 rpm and 10,000 requests/day (build.nvidia.com model page), 131,072-token context, 4,096-token max output per request
        paid_rate: $0.10 input + $0.32 output per 1M tokens (OpenRouter price for the exact model, 2026-08-22)
        basis: Single-flagship best case; envelope not extrapolated to the rest of the catalog.
        currency: USD
        daily: 140.08
        weekly: 980.58
        monthly: 4263.78
      calculation:
        method: best_case_rate_limit_envelope
        model_id: meta/llama-3.3-70b-instruct
        currency: USD
        limits:
          requests_per_minute: 40
          requests_per_day: 10000
          max_total_tokens_per_request: 131072
          max_output_tokens_per_request: 4096
        prices:
          input_per_million: 0.1
          output_per_million: 0.32
        result:
          daily:
            input_tokens: 1269760000
            output_tokens: 40960000
            total_tokens: 1310720000
            value: 140.0832
          weekly:
            input_tokens: 8888320000
            output_tokens: 286720000
            total_tokens: 9175040000
            value: 980.5824
          monthly:
            input_tokens: 38648320000
            output_tokens: 1246720000
            total_tokens: 39895040000
            value: 4263.7824
        tightest_constraints: 'Requests: 10,000/day binds (40 rpm would allow 57,600/day). Tokens: each request is capped by the 131,072-token context length, with the API-schema max_tokens ceiling of 4,096 assigned to the costlier output direction first and the remainder priced as input. Limit provenance: rateLimits JSON on the build.nvidia.com model page; max_tokens maximum 4096 from the NVIDIA API reference schema; contextLength 131072 from the model page specifications block.'
      sources:
      - type: official_product
        title: 'NVIDIA Build model page for Llama 3.3 70B Instruct (rateLimits: Up to 40 rpm, 10,000 requests per day; contextLength 131072; limits may vary by model and traffic)'
        url: https://build.nvidia.com/meta/llama-3_3-70b-instruct
      - type: official_api_reference
        title: 'NVIDIA API reference for meta/llama-3.3-70b-instruct (max_tokens: maximum 4096)'
        url: https://docs.api.nvidia.com/nim/reference/meta-llama-3_3-70b-instruct-infer
      - type: router_pricing
        title: OpenRouter models API price for meta-llama/llama-3.3-70b-instruct (snapshot 2026-08-22)
        url: https://openrouter.ai/api/v1/models
    hetzner_experiments:
      status: quantified
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 1884.35
        weekly: 13190.45
        monthly: 57354.89
      basis: Continuous saturation of the documented per-API-key 60-second window (10 requests, 4M input tokens, 100k output tokens) with each request's total tokens capped by the served models' documented 262,144-token context, priced for the costlier served model, Qwen3.8-27B, at the cheapest current OpenRouter listing for the exact model because neither Hetzner (free while experimental) nor Alibaba Cloud Model Studio's international catalog publishes a paid rate for these exact models. The context-derived per-request ceiling (10 x 262,144 = 2,621,440 tokens/minute) is tighter than the 4M input-TPM limit and was used. Assumes uninterrupted saturation of one API key with no latency, concurrency, availability, or fair-use loss; the value is a paid-equivalent ceiling, not expected usage.
      model_values:
      - id: Qwen3.8-27B
        allowance: Per key/60s, 10 requests, 4M input + 100k output tokens, 262,144-token context per request
        paid_rate: $0.40 input / $3.00 output per 1M tokens (cheapest current OpenRouter provider listing for qwen/qwen3.8-27b; provider range $0.40-0.50 / $3.00-3.40)
        currency: USD
        daily: 1884.35
        weekly: 13190.45
        monthly: 57354.89
      - id: Qwen/Qwen3.6-35B-A3B-FP8
        allowance: Same shared per-key envelope; alternative maximum, not additive
        paid_rate: $0.07 input / $0.70 output per 1M tokens (cheapest current OpenRouter provider listing for qwen/qwen3.6-35b-a3b; Hetzner serves an FP8 quantization of the same base model)
        basis: Alternative maximum if the shared envelope were spent on this model instead; not additive with the Qwen3.8-27B row.
        currency: USD
        daily: 354.96
        weekly: 2484.73
        monthly: 10804.13
      calculation:
        method: best_case_rate_limit_envelope
        model_id: Qwen3.8-27B
        currency: USD
        pricing_snapshot: '2026-08-22'
        limits:
          requests_per_minute: 10
          input_tokens_per_minute: 4000000
          output_tokens_per_minute: 100000
          max_total_tokens_per_request: 262144
        limits_notes: Hetzner documents the window as per 60 seconds per API key, mapped 1:1 to per-minute. The 262,144 max-total-per-request ceiling is the served model's documented context window in Hetzner's own model table (model-spec tier); Hetzner documents no separate per-request input/output maxima.
        prices:
          input_per_million: 0.4
          output_per_million: 3.0
        result:
          daily:
            input_tokens: 3630873600
            output_tokens: 144000000
            total_tokens: 3774873600
            value: 1884.3494
      sources:
      - type: official_docs
        title: Hetzner Inference API models and rate limits
        url: https://docs.hetzner.com/general/company-and-policy/experiments/inference/
      - type: router_pricing
        title: OpenRouter qwen/qwen3.8-27b
        url: https://openrouter.ai/qwen/qwen3.8-27b
      - type: router_pricing
        title: OpenRouter qwen/qwen3.6-35b-a3b
        url: https://openrouter.ai/qwen/qwen3.6-35b-a3b
    groqcloud:
      status: quantified
      cadence: recurring
      qualifier: subtotal
      total: {currency: USD, daily: 2.3062, weekly: 16.1434, monthly: 70.195}
      basis: Sum of independently published daily free-model quotas at Groq's current Developer paid rates. Each shared total-token envelope is allocated to costlier output tokens for a deterministic best-case upper bound; Compound systems without a public unit rate are excluded.
      model_values:
        - {id: canopylabs/orpheus-arabic-saudi, allowance: 3600 characters/day, paid_rate: $40 per 1M characters, currency: USD, daily: 0.144, weekly: 1.008, monthly: 4.38}
        - {id: canopylabs/orpheus-v1-english, allowance: 3600 characters/day, paid_rate: $22 per 1M characters, currency: USD, daily: 0.0792, weekly: 0.5544, monthly: 2.41}
        - {id: meta-llama/llama-prompt-guard-2-22m, allowance: 500K tokens/day, paid_rate: $0.03 per 1M input or output tokens, currency: USD, daily: 0.015, weekly: 0.105, monthly: 0.4566}
        - {id: meta-llama/llama-prompt-guard-2-86m, allowance: 500K tokens/day, paid_rate: $0.04 per 1M input or output tokens, currency: USD, daily: 0.02, weekly: 0.14, monthly: 0.6088}
        - {id: openai/gpt-oss-120b, allowance: 200K total tokens/day, paid_rate: $0.15 input + $0.60 output per 1M tokens, basis: Best case assigns the shared envelope to output, currency: USD, daily: 0.12, weekly: 0.84, monthly: 3.6525}
        - {id: openai/gpt-oss-20b, allowance: 200K total tokens/day, paid_rate: $0.075 input + $0.30 output per 1M tokens, basis: Best case assigns the shared envelope to output, currency: USD, daily: 0.06, weekly: 0.42, monthly: 1.8263}
        - {id: openai/gpt-oss-safeguard-20b, allowance: 200K total tokens/day, paid_rate: $0.075 input + $0.30 output per 1M tokens, basis: Best case assigns the shared envelope to output, currency: USD, daily: 0.06, weekly: 0.42, monthly: 1.8263}
        - {id: qwen/qwen3.6-27b, allowance: 200K total tokens/day, paid_rate: $0.60 input + $3.00 output per 1M tokens, basis: Best case assigns the shared envelope to output, currency: USD, daily: 0.60, weekly: 4.20, monthly: 18.2625}
        - {id: whisper-large-v3, allowance: 8 audio hours/day, paid_rate: $0.111 per audio hour, currency: USD, daily: 0.888, weekly: 6.216, monthly: 27.02}
        - {id: whisper-large-v3-turbo, allowance: 8 audio hours/day, paid_rate: $0.04 per audio hour, currency: USD, daily: 0.32, weekly: 2.24, monthly: 9.74}
      sources:
        - {type: official_docs, title: Groq model pricing, url: https://console.groq.com/docs/models}
        - {type: official_docs, title: Groq free-plan rate limits, url: https://console.groq.com/docs/rate-limits}
    google_gemini_api:
      status: partial
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 119.6
        weekly: 837.22
        monthly: 3640.42
      basis: 'Modeled best-case rate-limit envelope for gemini-2.5-flash, the only current free-tier model whose per-model free limits Google ever published. Google''s current rate-limits page (updated 2026-08-18) states quotas are viewable only in the AI Studio dashboard, so the request-rate ceilings are modeled from Google''s last publicly published free-tier table (rate-limits page as archived 2025-12-01: 10 RPM, 250,000 input TPM, 250 RPD for Gemini 2.5 Flash). Per-request ceilings are the model''s current documented specs (1,048,576 input-token limit, 65,536 output-token limit), and prices are Google''s own current paid text rates for the same model ($0.30 input / $2.50 output per 1M). Assumes uninterrupted saturation with maximal-context requests and no latency, concurrency, availability, or fair-use loss. Costlier free-tier models exist (gemini-3.5-flash at $1.50/$9.00, gemini-3.7/3.6-flash at $0.75/$3.75) but have never had published per-model free limits at any date, so they are excluded rather than modeled.'
      reason: Partial because the RPM/TPM/RPD inputs are Google's last published values (2025-12-01 archive), not current public documentation; third-party trackers report subsequent unpublished quota reductions, so current dashboard values may be lower than this modeled ceiling.
      model_values:
      - id: gemini-2.5-flash
        allowance: 'Modeled free-tier limits: 10 RPM, 250K input TPM, 250 RPD (Google''s last published table, 2025-12-01)'
        paid_rate: $0.30 input (text/image/video) + $2.50 output per 1M tokens (Google's current paid tier)
        basis: Best case fills every request to the documented 1,048,576-token input limit and 65,536-token output limit; audio input pricing ($1.00/1M) excluded.
        currency: USD
        daily: 119.6
        weekly: 837.22
        monthly: 3640.42
      calculation:
        method: modeled_request_envelope
        model_id: gemini-2.5-flash
        currency: USD
        assumptions:
        - quantity: requests_per_minute
          value: 10
          basis: Google's last publicly published free-tier limit for Gemini 2.5 Flash; the current page defers numbers to the AI Studio dashboard
          source_url: https://web.archive.org/web/20251201061102/https://ai.google.dev/gemini-api/docs/rate-limits
        - quantity: requests_per_day
          value: 250
          basis: Same archived first-party free-tier table (RPD column for Gemini 2.5 Flash)
          source_url: https://web.archive.org/web/20251201061102/https://ai.google.dev/gemini-api/docs/rate-limits
        - quantity: input_tokens_per_minute
          value: 250000
          basis: Same archived first-party free-tier table; Google defines TPM as input tokens per minute
          source_url: https://web.archive.org/web/20251201061102/https://ai.google.dev/gemini-api/docs/rate-limits
        limits:
          requests_per_minute: 10
          requests_per_day: 250
          input_tokens_per_minute: 250000
          max_input_tokens_per_request: 1048576
          max_output_tokens_per_request: 65536
        prices:
          input_per_million: 0.3
          output_per_million: 2.5
        result:
          daily:
            input_tokens: 262144000
            output_tokens: 16384000
            total_tokens: 278528000
            value: 119.6032
          weekly:
            input_tokens: 1835008000
            output_tokens: 114688000
            total_tokens: 1949696000
            value: 837.2224
          monthly:
            input_tokens: 7979008000
            output_tokens: 498688000
            total_tokens: 8477696000
            value: 3640.4224
        tightest_constraints: 'Requests: 250 RPD binds (10 RPM allows 14,400/day). Input: 250 requests x 1,048,576 max input = 262.144M/day binds (250K TPM would allow 360M/day). Output: 250 requests x 65,536 max output = 16.384M/day. Daily RPD quota resets at midnight Pacific.'
      sources:
      - type: official_pricing
        title: "Gemini API pricing (free-of-charge models and paid rates, snapshot 2026-08-22)"
        url: https://ai.google.dev/gemini-api/docs/pricing
      - type: official_docs
        title: Gemini API rate limits (current page states limits are dashboard-specific)
        url: https://ai.google.dev/gemini-api/docs/rate-limits
      - type: official_docs_archived
        title: Last published free-tier rate-limit table (archived 2025-12-01)
        url: https://web.archive.org/web/20251201061102/https://ai.google.dev/gemini-api/docs/rate-limits
      - type: official_docs
        title: "Gemini 2.5 Flash model specifications (1,048,576 input / 65,536 output token limits)"
        url: https://ai.google.dev/gemini-api/docs/models/gemini-2.5-flash
    mistral:
      status: quantified
      cadence: recurring
      qualifier: at_least
      total: {currency: USD, daily: 0.3285, weekly: 2.30, monthly: 10}
      basis: Face value of the recurring API credit. Each eligible paid model can consume the shared balance; always-zero-priced experimental models add unquantified value.
      model_values:
        - {id: Any eligible paid Mistral API model, allowance: $10 shared credit/month, paid_rate: Current Mistral catalog rate, currency: USD, daily: 0.3285, weekly: 2.30, monthly: 10}
      sources:
        - {type: official_pricing, title: Mistral pricing, url: https://mistral.ai/pricing}
    huggingface_inference_providers:
      status: quantified
      cadence: recurring
      qualifier: exact
      total: {currency: USD, daily: 0.0033, weekly: 0.023, monthly: 0.10}
      basis: Face value of Hugging Face's recurring routed-inference credit; the eligible provider catalog is dynamic and the shared balance is not additive per model.
      model_values:
        - {id: Any eligible routed model, allowance: $0.10 shared credit/month, paid_rate: Routed provider pass-through rate, currency: USD, daily: 0.0033, weekly: 0.023, monthly: 0.10}
      sources:
        - {type: official_docs, title: Inference Providers pricing, url: https://huggingface.co/docs/inference-providers/main/en/pricing}
    cloudflare_workers_ai:
      status: quantified
      cadence: recurring
      qualifier: exact
      total: {currency: USD, daily: 0.11, weekly: 0.77, monthly: 3.35}
      basis: 10,000 shared Neurons per day multiplied by the Workers Paid overage rate of $0.011 per 1,000 Neurons. A model can consume the pool at its documented neuron rate, so model maxima are not additive.
      model_values:
        - {id: Any model eligible for the free Neuron allocation, allowance: 10K shared Neurons/day, paid_rate: $0.011 per 1K Neurons, currency: USD, daily: 0.11, weekly: 0.77, monthly: 3.35}
      sources:
        - {type: official_pricing, title: Workers AI pricing, url: https://developers.cloudflare.com/workers-ai/platform/pricing/}
    cohere:
      status: partial
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 11.52
        weekly: 80.66
        monthly: 350.72
      basis: Best-case rate-limit envelope for the trial key's documented shared cap of 1,000 API calls per month, filled entirely with command-r-plus-08-2024 chat calls at the model's documented per-request ceilings (128K context window, 4K maximum output tokens per Cohere's models documentation) and priced at Cohere's current published rate for that exact model ($2.50 input / $10.00 output per 1M tokens, stated "for existing customers"). The 20 req/min trial chat limit never binds against the monthly cap. Daily and weekly figures are pro-rated shares of the monthly call bucket, not independent daily quotas. Assumes every call carries a maximal 128K-token request; the 1,000-call bucket is shared across all endpoints and models, so per-model rows are alternative maxima and are not additive.
      reason: Partial because the Command R+ price is scoped to existing customers on the pricing page; the priciest trial-eligible models cannot anchor the envelope (Command A+ and North Mini Code are listed by Cohere at $0 API price, and Command A Reasoning/Translate/Vision have no published token price, only "contact sales"), and embed/rerank endpoint allowances are excluded from the total.
      model_values:
      - id: command-r-plus-08-2024
        allowance: 1,000 shared trial API calls/month at 128K context / 4K max output per call
        paid_rate: $2.50 input + $10.00 output per 1M tokens (Cohere pricing FAQ, existing-customer rate)
        basis: Headline best case; shared monthly call bucket assigned entirely to this model with 4,096-token output and 123,904-token input per call.
        currency: USD
        daily: 11.52
        weekly: 80.66
        monthly: 350.72
      - id: command-r-08-2024
        allowance: Same shared 1,000 calls/month (alternative maximum, not additive)
        paid_rate: $0.15 input + $0.60 output per 1M tokens (current standard pricing table)
        basis: 128K context / 4K max output per Cohere models documentation.
        currency: USD
        monthly: 21.04
      - id: command-r7b-12-2024
        allowance: Same shared 1,000 calls/month (alternative maximum, not additive)
        paid_rate: $0.0375 input + $0.15 output per 1M tokens (current standard pricing table)
        basis: 128K context / 4K max output per Cohere models documentation.
        currency: USD
        monthly: 5.26
      - id: command-a-plus-05-2026
        allowance: Same shared 1,000 calls/month (alternative maximum, not additive)
        paid_rate: Cohere's pricing page lists Command A+ API access at $0 (open-weights release), so trial calls to it add no paid-equivalent value
        currency: USD
        monthly: 0
      calculation:
        method: best_case_rate_limit_envelope
        model_id: command-r-plus-08-2024
        currency: USD
        limits:
          requests_per_month: 1000
          requests_per_day: 32.85420944558522
          requests_per_minute: 20
          max_total_tokens_per_request: 128000
          max_output_tokens_per_request: 4096
        prices:
          input_per_million: 2.5
          output_per_million: 10.0
        result:
          daily:
            input_tokens: 4070768
            output_tokens: 134571
            total_tokens: 4205339
            value: 11.5226
          weekly:
            input_tokens: 28495376
            output_tokens: 941996
            total_tokens: 29437372
            value: 80.6584
          monthly:
            input_tokens: 123904000
            output_tokens: 4096000
            total_tokens: 128000000
            value: 350.72
        tightest_constraints: The documented 1,000-calls-per-month trial cap binds (20 req/min would allow 28,800/day); requests_per_day above is the monthly cap divided by 365.25/12 days so the calculator reproduces exactly 1,000 requests per month. Per-request tokens are capped by the model's 128K context window with the 4K output maximum assigned to the costlier output direction first.
      sources:
      - type: official_docs
        title: 'Cohere rate limits (trial: 1,000 API calls/month, chat 20 req/min per model)'
        url: https://docs.cohere.com/v2/docs/rate-limits
      - type: official_docs
        title: Cohere models (context windows and maximum output tokens)
        url: https://docs.cohere.com/v2/docs/models
      - type: official_pricing
        title: Cohere pricing (Command R/R7B standard rates; Command R+ existing-customer rate; Command A+ and North Mini Code listed free; trial keys free)
        url: https://cohere.com/pricing
    vercel_ai_gateway:
      status: quantified
      cadence: recurring
      qualifier: at_least
      total: {currency: USD, daily: 0.1643, weekly: 1.15, monthly: 5}
      basis: Face value of the recurring shared AI Gateway credit. Native zero-price routes can add value but are volatile and excluded from the total.
      model_values:
        - {id: Any eligible AI Gateway model, allowance: $5 shared credit/month, paid_rate: Current gateway catalog rate, currency: USD, daily: 0.1643, weekly: 1.15, monthly: 5}
      sources:
        - {type: official_pricing, title: AI Gateway pricing, url: https://vercel.com/docs/ai-gateway/pricing}
    ibm_watsonx_ai_runtime:
      status: partial
      cadence: recurring
      qualifier: subtotal
      total: {currency: USD, daily: 0.4964, weekly: 3.4749, monthly: 15.1095}
      basis: The Lite plan's three independently documented monthly allowances are priced at IBM's current US rates. The 300K shared foundation-token pool is valued once against Granite 4H Small by assigning the envelope to costlier output tokens; other models and regional price differences are excluded.
      reason: This is a conservative subtotal because IBM's broader catalog uses model-specific token multipliers and regional prices.
      model_values:
        - {id: ibm/granite-4h-small, allowance: 300K shared foundation tokens/month, paid_rate: "$0.0636 input + $0.265 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: USD, daily: 0.0026, weekly: 0.0183, monthly: 0.0795}
        - {id: Machine learning capacity, allowance: 20 CUH/month, paid_rate: $0.55 per CUH, currency: USD, daily: 0.3614, weekly: 2.5298, monthly: 11}
        - {id: Text extraction, allowance: 100 pages/month, paid_rate: $0.0403 per page, currency: USD, daily: 0.1324, weekly: 0.9269, monthly: 4.03}
      sources:
        - {type: official_pricing, title: IBM watsonx.ai pricing, url: https://www.ibm.com/products/watsonx-ai/pricing}
    opencode_zen: {status: not_quantifiable, reason: "Free promotional routes have no stable numeric allowance, expiry, or common paid comparison."}
    zai:
      status: not_quantifiable
      reason: 'The zero-price routes (glm-4.7-flash, glm-4.6v-flash, glm-4.5-flash) have documented per-request ceilings (200K context / 128K max output for GLM-4.7-Flash) and an exact-model paid reference exists (Z.AI''s own accelerated GLM-4.7-FlashX at $0.07 input / $0.40 output per 1M, and OpenRouter''s z-ai/glm-4.7-flash at $0.06/$0.40), but no finite period envelope survives either the documented or the modeled tier: neither docs.z.ai nor the operator''s Chinese platform documentation publishes any numeric request-rate, token-rate, or concurrency ceiling for these routes — both state that per-model rates are visible only in the signed-in rate-limit console — so per-request ceilings alone leave requests per period unbounded. Modeling the missing request-rate quantity would have no first-party published example to cite (third-party reports of a 1-request concurrency cap are leads, not evidence, and a concurrency cap without a first-party generation-speed figure still yields no token envelope), which the modeled tier does not permit.'
      sources:
      - type: official_pricing
        title: Z.AI pricing (free flash models; GLM-4.7-FlashX paid rate)
        url: https://docs.z.ai/guides/overview/pricing
      - type: official_docs
        title: GLM-4.7 model guide (200K context, 128K max output; Flash marked free)
        url: https://docs.z.ai/guides/llm/glm-4.7
      - type: official_docs
        title: Operator rate-limit documentation deferring numbers to the signed-in console
        url: https://docs.bigmodel.cn/cn/api/rate-limit
      - type: router_pricing
        title: OpenRouter price for the exact model z-ai/glm-4.7-flash (snapshot 2026-08-22)
        url: https://openrouter.ai/api/v1/models
    llmapi_ai: {status: not_quantifiable, reason: "A request-rate limit is published, but response sizes and a like-for-like paid rate are not."}
    api_airforce:
      status: quantified
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 14.9914
        weekly: 104.9395
        monthly: 456.2995
      basis: Documented free-plan envelope of 1 request/minute and 1,000 requests/day multiplied by the 4,096-token per-request completion cap that the live catalog documents for every free-tier chat model, priced at Api.Airforce's own paid per-token USD rates (customer_price_table, micro-USD, snapshot 2026-08-22). The headline assigns the shared request pool to rnj-1, the most expensive chat model carrying the catalog's free-tier marker (tier "free", 51 models); model rows are alternative maxima for the same shared pool and are not additive. Assumes uninterrupted saturation with no latency, concurrency, availability, or fair-use loss, and counts completion tokens only because no per-request input ceiling is documented. The provider's confirmed but unpublished per-model daily token cap can only reduce the realizable value, so the figures are strict upper bounds. Free-tier markers in the rotating catalog are volatile, and authenticated runtime verification of premium-looking free-tier entries such as rnj-1 and mistral-large-2512 has not been performed.
      model_values:
      - id: rnj-1
        allowance: 1000 shared requests/day x 4096-token max completion
        paid_rate: $3.66 per 1M input or output tokens (Api.Airforce paid rate)
        currency: USD
        daily: 14.9914
        weekly: 104.9395
        monthly: 456.2995
      - id: kimi-k2.7-code
        allowance: Same shared 1000 requests/day
        paid_rate: $0.85 input + $3.50 output per 1M tokens (Api.Airforce paid rate; catalog status degraded)
        currency: USD
        daily: 14.336
      - id: gpt-oss-120b
        allowance: Same shared 1000 requests/day
        paid_rate: $0.11 input + $0.43 output per 1M tokens (Api.Airforce paid rate)
        currency: USD
        daily: 1.7613
      - id: glm-4.7-flash
        allowance: Same shared 1000 requests/day
        paid_rate: $0.37 per 1M input or output tokens (Api.Airforce paid rate)
        currency: USD
        daily: 1.5155
      - id: mistral-large-2512
        allowance: Same shared 1000 requests/day
        paid_rate: $0.20 per 1M input or output tokens (Api.Airforce paid rate)
        currency: USD
        daily: 0.8192
      - id: qwen3-30b-a3b-fp8
        allowance: Same shared 1000 requests/day
        paid_rate: $0.20 per 1M input or output tokens (Api.Airforce paid rate)
        currency: USD
        daily: 0.8192
      - id: gpt-oss-20b
        allowance: Same shared 1000 requests/day
        paid_rate: $0.03 input + $0.12 output per 1M tokens (Api.Airforce paid rate)
        currency: USD
        daily: 0.4915
      calculation:
        method: best_case_rate_limit_envelope
        model_id: rnj-1
        currency: USD
        limits:
          requests_per_minute: 1
          requests_per_day: 1000
          max_output_tokens_per_request: 4096
        prices:
          input_per_million: 3.66
          output_per_million: 3.66
        result:
          daily:
            input_tokens: 0
            output_tokens: 4096000
            total_tokens: 4096000
            value: 14.99136
          weekly:
            input_tokens: 0
            output_tokens: 28672000
            total_tokens: 28672000
            value: 104.93952
          monthly:
            input_tokens: 0
            output_tokens: 124672000
            total_tokens: 124672000
            value: 456.29952
      sources:
      - type: official_pricing
        title: Api.Airforce pricing with free-plan request limits
        url: https://api.airforce/pricing/
      - type: live_catalog
        title: "Models API with free-tier markers, 4096-token completion caps, and paid micro-USD rates"
        url: https://api.airforce/v1/models
    llm7:
      status: partial
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 0.06
        weekly: 0.42
        monthly: 1.82625
      basis: 'The free-token tier documents a finite envelope of 1,000,000 tokens per rolling 24 hours (input plus output; 2 rps / 40 rpm / 100 rph), and anonymous access documents 500,000 tokens per 24 hours, so the daily token quota binds. The paid reference is modeled: the free default/fast routers do not disclose which model serves a request, so the envelope is priced at LLM7''s own published per-token rate for gpt-oss:20b, the most expensive text model the live catalog flags usage_based_only=false (callable outside usage-based billing), with the whole shared envelope assigned to costlier output. LLM7''s catalog prices are its own same-service Pro/balance rates, which take precedence over higher third-party router prices for the same models. The anonymous tier is worth half the headline ($0.03/day) and is not additive with it. Assumes uninterrupted saturation.'
      model_values:
      - id: gpt-oss:20b
        allowance: 1,000,000 shared free-token-tier tokens per 24 hours (input + output combined)
        paid_rate: $0.04 input + $0.06 output per 1M tokens (LLM7's own catalog rate, 2026-08-22)
        basis: Modeled most expensive free-eligible model; other free-eligible turbo models price lower
        currency: USD
        daily: 0.06
        weekly: 0.42
        monthly: 1.82625
      calculation:
        method: modeled_request_envelope
        model_id: gpt-oss:20b
        currency: USD
        assumptions:
        - quantity: paid_reference_model
          value: gpt-oss:20b
          basis: The free default/fast routers do not disclose the served model; gpt-oss:20b is the highest-priced chat model flagged usage_based_only=false (free-tier callable) in the live catalog
          source_url: https://api.llm7.io/v1/models
        - quantity: output_price_per_million
          value: 0.06
          basis: LLM7's own published catalog rate for gpt-oss:20b output tokens
          source_url: https://api.llm7.io/v1/models
        - quantity: input_price_per_million
          value: 0.04
          basis: LLM7's own published catalog rate for gpt-oss:20b input tokens
          source_url: https://api.llm7.io/v1/models
        limits:
          tokens_per_day: 1000000
          requests_per_minute: 40
        prices:
          input_per_million: 0.04
          output_per_million: 0.06
        result:
          daily:
            input_tokens: 0
            output_tokens: 1000000
            total_tokens: 1000000
            value: 0.06
          weekly:
            input_tokens: 0
            output_tokens: 7000000
            total_tokens: 7000000
            value: 0.42
          monthly:
            input_tokens: 0
            output_tokens: 30437500
            total_tokens: 30437500
            value: 1.82625
      sources:
      - type: official_docs
        title: LLM7 limits (per-tier token and request quotas)
        url: https://docs.llm7.io/limits
      - type: live_catalog
        title: "LLM7 models endpoint (tiers, usage_based_only flags, own token prices)"
        url: https://api.llm7.io/v1/models
      - type: official_docs
        title: LLM7 models and routers guide (free default/fast vs paid pro)
        url: https://docs.llm7.io/guides/models
    modelscope_inference:
      status: quantified
      cadence: recurring
      qualifier: subtotal
      total:
        currency: USD
        daily: 21.43
        weekly: 150.01
        monthly: 652.28
      basis: 'Per-model daily maximum: 200 API-Inference calls per model per day saturated on the costliest documented standard-quota example model, Qwen/Qwen3.5-35B-A3B, with per-request ceilings taken from the served model''s documented specifications (262,144 native context; 81,920 max completion tokens per the Qwen model card and the exact-model OpenRouter listing), priced at the cheapest current OpenRouter provider listing for the exact model ($0.14 input / $1.00 output per 1M) because ModelScope''s API-Inference has no paid tier and the exact model is absent from Alibaba Cloud Model Studio''s international catalog. Subtotal because the shared 2,000-call/day account ceiling could add concurrent per-model quotas from the large badge-gated catalog that is not enumerated here; restricted 100-call/day models (DeepSeek rows) value lower and shared account calls are counted once, not summed across models. Assumes uninterrupted saturation; access is noncommercial development-only with dynamic model-specific concurrency.'
      model_values:
      - id: Qwen/Qwen3.5-35B-A3B
        allowance: 200 calls/model/day within the shared 2,000 calls/day account cap; 262,144-token context and 81,920 max output tokens per request
        paid_rate: $0.14 input / $1.00 output per 1M tokens (cheapest current OpenRouter provider listing for the exact model; range $0.14-0.3125 / $1.00-1.80)
        currency: USD
        daily: 21.43
        weekly: 150.01
        monthly: 652.28
      - id: DeepSeek-R1-0528
        allowance: Restricted 100 calls/day; 163,840-token context and 32,768 max output tokens per request
        paid_rate: $0.50 input / $2.15 output per 1M tokens (cheapest current OpenRouter provider listing for deepseek/deepseek-r1-0528; DeepSeek's first-party API no longer serves this exact model)
        basis: Per-model maximum for a restricted expensive model; drawn from the same shared account cap, not additive with the total.
        currency: USD
        daily: 13.6
        weekly: 95.19
        monthly: 413.91
      - id: DeepSeek-V3.2-Exp
        allowance: Restricted 100 calls/day; 163,840-token context and 65,536 max output tokens per request
        paid_rate: $0.27 input / $0.41 output per 1M tokens (current OpenRouter provider listing for deepseek/deepseek-v3.2-exp; DeepSeek's first-party API no longer serves this exact model)
        basis: Per-model maximum for a restricted expensive model; drawn from the same shared account cap, not additive with the total.
        currency: USD
        daily: 5.34
        weekly: 37.39
        monthly: 162.57
      calculation:
        method: best_case_rate_limit_envelope
        model_id: Qwen/Qwen3.5-35B-A3B
        currency: USD
        pricing_snapshot: '2026-08-22'
        limits:
          requests_per_day: 200
          max_output_tokens_per_request: 81920
          max_total_tokens_per_request: 262144
        limits_notes: '200 is the documented per-model daily call cap and binds before the 2,000-call/day account cap for a single model. Per-request ceilings are model-spec tier: the Qwen3.5 model card documents 262,144 native context and max_tokens=81920 as the top recommended generation length, and the exact-model OpenRouter listing states the same 81,920 completion-token maximum; ModelScope''s own limits page documents calls and concurrency only. ModelScope''s per-day windows are treated as daily periods.'
        prices:
          input_per_million: 0.14
          output_per_million: 1.0
        result:
          daily:
            input_tokens: 36044800
            output_tokens: 16384000
            total_tokens: 52428800
            value: 21.4303
      sources:
      - type: official_docs
        title: ModelScope API-Inference limits
        url: https://modelscope.cn/docs/model-service/API-Inference/limits
      - type: official_product
        title: ModelScope free API-Inference resources
        url: https://www.modelscope.cn/learn/1409
      - type: model_card
        title: Qwen3.5-35B-A3B model card (context and generation length)
        url: https://huggingface.co/Qwen/Qwen3.5-35B-A3B
      - type: router_pricing
        title: OpenRouter qwen/qwen3.5-35b-a3b
        url: https://openrouter.ai/qwen/qwen3.5-35b-a3b
      - type: router_pricing
        title: OpenRouter deepseek/deepseek-r1-0528
        url: https://openrouter.ai/deepseek/deepseek-r1-0528
      - type: router_pricing
        title: OpenRouter deepseek/deepseek-v3.2-exp
        url: https://openrouter.ai/deepseek/deepseek-v3.2-exp
    awanllm:
      status: quantified
      cadence: recurring
      qualifier: subtotal
      total:
        currency: USD
        daily: 2.62144
        weekly: 18.35008
        monthly: 79.79008
      basis: The free plan documents independent per-size-class daily request quotas (Small 200/day, Medium 10/day, Large 10/day at a shared 20 requests/minute) with "Unlimited Tokens!" per request, so each request is bounded only by the model's context length as published on AwanLLM's own models page (131,072 tokens for Meta-Llama-3.1-8B-Instruct and Meta-Llama-3.1-70B-Instruct). Each class is valued at its most expensive eligible exact model using current OpenRouter per-token prices for the same models (snapshot 2026-08-22; Meta publishes no first-party API rate and AwanLLM's own paid plans are flat subscriptions without per-token prices). The Medium quota is excluded because the current catalog labels every model Small or Large only, making this a subtotal. Class quotas are independent and summed; models inside a class share that class quota and are not additive. Assumes uninterrupted saturation with the full context consumed by every request.
      model_values:
      - id: Meta-Llama-3.1-8B-Instruct
        allowance: 200 Small-class requests/day, each up to the 131,072-token published context
        paid_rate: $0.05 input + $0.08 output per 1M tokens (OpenRouter meta-llama/llama-3.1-8b-instruct)
        basis: Most expensive Small-class model; envelope assigned to costlier output
        currency: USD
        daily: 2.097152
        weekly: 14.680064
        monthly: 63.832064
      - id: Meta-Llama-3.1-70B-Instruct
        allowance: 10 Large-class requests/day, each up to the 131,072-token published context
        paid_rate: $0.40 input + $0.40 output per 1M tokens (OpenRouter meta-llama/llama-3.1-70b-instruct)
        basis: Most expensive Large-class model; input and output are priced identically
        currency: USD
        daily: 0.524288
        weekly: 3.670016
        monthly: 15.958016
      calculation:
        method: best_case_rate_limit_envelope
        model_id: Meta-Llama-3.1-8B-Instruct
        currency: USD
        limits:
          requests_per_minute: 20
          requests_per_day: 200
          max_input_tokens_per_request: 131072
          max_output_tokens_per_request: 131072
          max_total_tokens_per_request: 131072
        prices:
          input_per_million: 0.05
          output_per_million: 0.08
        result:
          daily:
            input_tokens: 0
            output_tokens: 26214400
            total_tokens: 26214400
            value: 2.097152
          weekly:
            input_tokens: 0
            output_tokens: 183500800
            total_tokens: 183500800
            value: 14.680064
          monthly:
            input_tokens: 0
            output_tokens: 797900800
            total_tokens: 797900800
            value: 63.832064
      sources:
      - type: official_pricing
        title: AwanLLM pricing (free-plan class quotas and unlimited tokens)
        url: https://www.awanllm.com/pricing
      - type: official_catalog
        title: AwanLLM models (size classes and published context lengths)
        url: https://www.awanllm.com/models
      - type: live_catalog
        title: OpenRouter models endpoint (exact-model paid prices)
        url: https://openrouter.ai/api/v1/models
    arliai:
      status: quantified
      cadence: recurring
      qualifier: subtotal
      total:
        currency: USD
        daily: 0.0765
        weekly: 0.5355
        monthly: 2.328468
      basis: Free accounts may use each model for 5 requests every 2 days (an independent per-model quota per first-party docs), capped at 12K context tokens and 1 concurrent request, so per-model best case is 2.5 requests/day with the full 12,000-token per-request envelope assigned to costlier output. This subtotal saturates that quota only for the four catalog entries that are unmodified release models with exact-model paid prices on OpenRouter (GLM-4.7, MiMo-V2.5, DeepSeek-V4-Flash-0731, Gemma-4-31B-it; snapshot 2026-08-22); the remaining 88 finetune/merge/derestricted catalog entries carry equal independent quotas but have no exact-model paid counterpart, so their value is real but unpriced. Arli AI's own paid offering is an unlimited-usage subscription with no per-token rate, so exact-model router prices are the comparison. Per-model rows are additive here because the quota is documented as independent per model. Assumes uninterrupted saturation across the two-day reset cycle.
      model_values:
      - id: GLM-4.7
        allowance: 5 requests per 2 days for this model, 12,000-token context per request
        paid_rate: $0.40 input + $1.75 output per 1M tokens (OpenRouter z-ai/glm-4.7)
        currency: USD
        daily: 0.0525
        weekly: 0.3675
        monthly: 1.597969
      - id: Gemma-4-31B-it
        allowance: 5 requests per 2 days for this model, 12,000-token context per request
        paid_rate: $0.10 input + $0.34 output per 1M tokens (OpenRouter google/gemma-4-31b-it)
        currency: USD
        daily: 0.0102
        weekly: 0.0714
        monthly: 0.310462
      - id: MiMo-V2.5
        allowance: 5 requests per 2 days for this model, 12,000-token context per request
        paid_rate: $0.14 input + $0.28 output per 1M tokens (OpenRouter xiaomi/mimo-v2.5)
        currency: USD
        daily: 0.0084
        weekly: 0.0588
        monthly: 0.255675
      - id: DeepSeek-V4-Flash-0731
        allowance: 5 requests per 2 days for this model, 12,000-token context per request
        paid_rate: $0.08 input + $0.18 output per 1M tokens (OpenRouter deepseek/deepseek-v4-flash-0731)
        currency: USD
        daily: 0.0054
        weekly: 0.0378
        monthly: 0.164362
      calculation:
        method: best_case_rate_limit_envelope
        model_id: GLM-4.7
        currency: USD
        limits:
          requests_per_day: 2.5
          max_input_tokens_per_request: 12000
          max_output_tokens_per_request: 12000
          max_total_tokens_per_request: 12000
        prices:
          input_per_million: 0.4
          output_per_million: 1.75
        result:
          daily:
            input_tokens: 0
            output_tokens: 30000
            total_tokens: 30000
            value: 0.0525
          weekly:
            input_tokens: 0
            output_tokens: 210000
            total_tokens: 210000
            value: 0.3675
          monthly:
            input_tokens: 0
            output_tokens: 913125
            total_tokens: 913125
            value: 1.597969
      sources:
      - type: official_pricing
        title: "Arli AI pricing (free plan quota, 12K context, concurrency)"
        url: https://www.arliai.com/pricing
      - type: official_docs
        title: Text generation limits (5 requests per model per 2 days)
        url: https://www.arliai.com/docs/textgen
      - type: live_catalog
        title: "Arli AI public model catalog (92 models, release-model identification)"
        url: https://api.arliai.com/model/all
      - type: live_catalog
        title: OpenRouter models endpoint (exact-model paid prices)
        url: https://openrouter.ai/api/v1/models
    freeinference_org: {status: not_quantifiable, reason: "The provider describes a generous quota but publishes no numeric allowance."}
    fastrouter: {status: not_quantifiable, reason: "Model-specific free routes have no published finite quota or like-for-like paid price."}
    kilo_ai_gateway: {status: not_quantifiable, reason: "The request-rate ceiling has no bounded request size and the rotating aliases have no stable paid counterpart."}
    scaleway_generative_apis:
      status: quantified
      cadence: one_time
      qualifier: up_to
      total: {currency: EUR, one_time: 5.68}
      basis: The 1M shared text-token trial is assigned to the highest-priced captured eligible model, GLM 5.2, and its costlier output rate for a deterministic best-case upper bound, then the separately documented 60 transcription minutes are added. Text-model rows show alternative maxima for the same shared pool and are not additive.
      model_values:
        - {id: glm-5.2, allowance: 1M shared total tokens, paid_rate: "€1.80 input + €5.50 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: EUR, one_time: 5.50}
        - {id: deepseek-v4-flash-0731, allowance: 1M shared total tokens, paid_rate: "€0.40 input + €0.80 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: EUR, one_time: 0.80}
        - {id: qwen3.6-35b-a3b, allowance: 1M shared total tokens, paid_rate: "€0.25 input + €1.50 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: EUR, one_time: 1.50}
        - {id: pixtral-12b-2409, allowance: 1M shared total tokens, paid_rate: €0.20 per 1M input or output tokens, currency: EUR, one_time: 0.20}
        - {id: mistral-small-3.2-24b-instruct-2506, allowance: 1M shared total tokens, paid_rate: "€0.15 input + €0.35 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: EUR, one_time: 0.35}
        - {id: gpt-oss-120b, allowance: 1M shared total tokens, paid_rate: "€0.15 input + €0.60 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: EUR, one_time: 0.60}
        - {id: whisper-large-v3, allowance: 60 separate transcription minutes, paid_rate: €0.003 per audio minute, currency: EUR, one_time: 0.18}
      sources:
        - {type: official_pricing, title: Scaleway Model as a Service pricing, url: https://www.scaleway.com/en/pricing/model-as-a-service/}
    qwen_cloud: {status: not_quantifiable, reason: "Per-model trial quotas and prices are dynamic and are not captured as an exact auditable model snapshot in this record."}
    sea_lion_api: {status: not_quantifiable, reason: "Only a request-rate limit is public; total tokens, duration, and a paid comparison are not documented."}
    ndif: {status: not_quantifiable, reason: "Research access has no public numeric user allocation or commercial paid counterpart."}
    ai_horde: {status: not_quantifiable, reason: "Capacity is donated and prioritized with non-purchasable Kudos rather than a fixed monetary or token allowance."}
    pollinations: {status: not_quantifiable, reason: "The recurring Pollen refill amount and community-route capacity are not fixed publicly."}
    puter_js: {status: not_quantifiable, reason: "The end user's monthly allowance is not documented and model prices are paid by that user rather than the developer."}
    public_ai: {status: not_quantifiable, reason: "The starter-credit amount is not public."}
    lightning_ai_model_apis:
      status: partial
      cadence: recurring
      qualifier: at_least
      total: {currency: USD, daily: 0.4928, weekly: 3.45, monthly: 15}
      basis: Conservative face value of the separately documented recurring platform balance. The advertised 30M Model API tokens are excluded because public pages do not reconcile them with token-priced litAI calls.
      reason: The signed-in billing page is authoritative; do not add the 30M-token claim to the $15 balance without confirming whether they are independent.
      model_values:
        - {id: Eligible Model API or user-deployed model, allowance: At least $15 shared credit/month, paid_rate: Current Lightning catalog or compute rate, currency: USD, daily: 0.4928, weekly: 3.45, monthly: 15}
      sources:
        - {type: official_pricing, title: Lightning pricing, url: https://lightning.ai/pricing}
        - {type: official_docs, title: Model APIs, url: https://lightning.ai/docs/overview/model-apis}
    modal:
      status: quantified
      cadence: recurring
      qualifier: exact
      total: {currency: USD, daily: 0.9856, weekly: 6.90, monthly: 30}
      basis: Face value of the Starter plan's recurring compute credit. Any one deployed model can consume the shared balance; model rows are not additive.
      model_values:
        - {id: Any user-deployed model, allowance: $30 shared compute credit/month, paid_rate: Selected CPU/GPU compute rate, currency: USD, daily: 0.9856, weekly: 6.90, monthly: 30}
      sources:
        - {type: official_pricing, title: Modal pricing, url: https://modal.com/pricing}
    beam_cloud:
      status: quantified
      cadence: recurring
      qualifier: exact
      total: {currency: USD, daily: 0.9856, weekly: 6.90, monthly: 30}
      basis: Face value of Beam's recurring shared compute credit. The value is provider spend, not a separate $30 allocation for every deployed model.
      model_values:
        - {id: Any user-deployed model, allowance: $30 shared compute credit/month, paid_rate: Selected CPU/GPU compute rate, currency: USD, daily: 0.9856, weekly: 6.90, monthly: 30}
      sources:
        - {type: official_pricing, title: Beam pricing, url: https://www.beam.cloud/pricing}
    sail_research:
      status: quantified
      cadence: recurring
      qualifier: exact
      total: {currency: USD, daily: 0.1643, weekly: 1.15, monthly: 5}
      basis: Face value of Sail's recurring shared model credit; per-model values are maxima against the same balance and are not additive.
      model_values:
        - {id: Any eligible Sail model, allowance: $5 shared credit/month, paid_rate: Current Sail model rate, currency: USD, daily: 0.1643, weekly: 1.15, monthly: 5}
      sources:
        - {type: official_pricing, title: Sail Research pricing, url: https://www.sailresearch.com/}
    cartesia:
      status: quantified
      cadence: recurring
      qualifier: up_to
      total: {currency: USD, daily: 0.0329, weekly: 0.23, monthly: 1}
      basis: The free plan's 20K shared credits are valued at the Pro plan's $5 per 100K-credit ratio. Sonic and Ink examples are alternative uses of the same balance, not additive allocations.
      model_values:
        - {id: Sonic-3.5, allowance: About 27 TTS minutes/month, paid_rate: About $5 per 133 minutes on Pro, currency: USD, daily: 0.0329, weekly: 0.23, monthly: 1}
        - {id: Ink-2, allowance: About 1.85 STT hours/month, paid_rate: About $5 per 9.27 hours on Pro, currency: USD, daily: 0.0329, weekly: 0.23, monthly: 1}
      sources:
        - {type: official_pricing, title: Cartesia pricing, url: https://www.cartesia.ai/pricing}
    elevenlabs_api:
      status: quantified
      cadence: recurring
      qualifier: up_to
      total: {currency: USD, daily: 0.0329, weekly: 0.23, monthly: 1}
      basis: Current pay-as-you-go rates value either 10K multilingual/v3 characters at $0.10 per 1K or 20K Flash/Turbo characters at $0.05 per 1K. These are alternative uses of shared free credits.
      model_values:
        - {id: eleven_v3 or v2 Multilingual, allowance: 10K characters/month, paid_rate: $0.10 per 1K characters, currency: USD, daily: 0.0329, weekly: 0.23, monthly: 1}
        - {id: eleven_flash_v2_5 or Turbo, allowance: 20K characters/month, paid_rate: $0.05 per 1K characters, currency: USD, daily: 0.0329, weekly: 0.23, monthly: 1}
      sources:
        - {type: official_pricing, title: ElevenAPI pricing, url: https://elevenlabs.io/pricing/api}
    ovhcloud_ai_endpoints:
      status: partial
      cadence: one_time
      qualifier: at_least
      total: {currency: USD, daily: 6.5708, weekly: 45.9956, monthly: 200, one_time: 200}
      basis: Face value of the separate one-month Public Cloud trial, normalized across its stated duration. The ongoing zero-price endpoints add unquantified value because they publish request-rate ceilings rather than a finite usage allowance.
      reason: A payment method is required for the trial; the anonymous and authenticated zero-price endpoints remain usable separately under their documented limits.
      model_values:
        - {id: Any Public Cloud trial-eligible AI endpoint, allowance: $200 shared credit for one month, paid_rate: Current OVHcloud endpoint rate, currency: USD, daily: 6.5708, weekly: 45.9956, monthly: 200, one_time: 200}
      sources:
        - {type: official_trial, title: OVHcloud Public Cloud free trial, url: https://www.ovhcloud.com/en/public-cloud/free-trial/}
        - {type: official_catalog, title: AI Endpoints catalog, url: https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/}
    requesty:
      status: quantified
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 47.1859
        weekly: 330.3014
        monthly: 1436.2214
      basis: 'Documented Free-plan envelope of 200 requests/day on zero-priced models multiplied by the 65,536-token maximum completion that Requesty''s own live catalog documents for nvidia/nemotron-3-ultra-550b-a55b, the most expensive eligible model. The same models are zero-priced on Requesty''s PAYG catalog, so no same-provider paid rate exists; the paid reference is OpenRouter''s current paid route for the identical model ID ($0.60 input / $3.60 output per 1M tokens, snapshot 2026-08-22). Assumes uninterrupted saturation with no latency, concurrency, availability, or fair-use loss, and counts completion tokens only because no per-request input ceiling binds below the context window. Rows are alternative maxima for the same shared 200-request pool and are not additive. Excluded: poolside/laguna-xs.2, poolside/laguna-m.1, and nvidia/nemotron-3-nano-30b-a3b report max_output_tokens=0; nvidia/muse-glimmer-30b''s only external paid listing carries a conflicting meta/ vendor prefix; mistral/leanstral-1-5, novita/inclusionai/ling-3.0-tiny, and nvidia/nemotron-3.5-content-safety lack an exact-model paid rate. Zero-price catalog membership is volatile and runtime callability has not been verified with an authenticated call.'
      model_values:
      - id: nvidia/nemotron-3-ultra-550b-a55b
        allowance: 200 shared requests/day x 65536-token max completion
        paid_rate: $0.60 input + $3.60 output per 1M tokens (OpenRouter paid route, same model ID)
        currency: USD
        daily: 47.1859
        weekly: 330.3014
        monthly: 1436.2214
      - id: nvidia/nemotron-3-super-120b-a12b
        allowance: Same shared 200 requests/day; 65536-token max completion
        paid_rate: $0.085 input + $0.40 output per 1M tokens (OpenRouter paid route)
        currency: USD
        daily: 5.2429
      - id: nvidia/nemotron-3.5-lightning-30b-a3b
        allowance: Same shared 200 requests/day; 65536-token max completion
        paid_rate: $0.08 input + $0.20 output per 1M tokens (OpenRouter paid route nvidia/nemotron-3.5-lightning)
        currency: USD
        daily: 2.6214
      - id: google/gemma-4-31b-it
        allowance: Same shared 200 requests/day; 8192-token max completion
        paid_rate: $0.10 input + $0.34 output per 1M tokens (OpenRouter paid route)
        currency: USD
        daily: 0.5571
      calculation:
        method: best_case_rate_limit_envelope
        model_id: nvidia/nemotron-3-ultra-550b-a55b
        currency: USD
        limits:
          requests_per_day: 200
          max_output_tokens_per_request: 65536
        prices:
          input_per_million: 0.6
          output_per_million: 3.6
        result:
          daily:
            input_tokens: 0
            output_tokens: 13107200
            total_tokens: 13107200
            value: 47.18592
          weekly:
            input_tokens: 0
            output_tokens: 91750400
            total_tokens: 91750400
            value: 330.30144
          monthly:
            input_tokens: 0
            output_tokens: 398950400
            total_tokens: 398950400
            value: 1436.22144
      sources:
      - type: official_pricing
        title: Requesty pricing with the 200-requests/day Free plan on free models
        url: https://www.requesty.ai/pricing
      - type: live_catalog
        title: Requesty models API with zero prices and max output specs
        url: https://router.requesty.ai/v1/models
      - type: router_price_reference
        title: OpenRouter models API paid rates for the identical model IDs
        url: https://openrouter.ai/api/v1/models
    inception_platform:
      status: quantified
      cadence: one_time
      qualifier: up_to
      total: {currency: USD, one_time: 75}
      basis: The 100M-token signup allowance is valued at the current costlier $0.75 output rate for a deterministic best-case upper bound. Both models share the allowance, so values are not additive.
      model_values:
        - {id: mercury-2, allowance: 100M shared signup tokens, paid_rate: $0.25 input + $0.75 output per 1M tokens, basis: Best case assigns the shared envelope to output, currency: USD, one_time: 75}
        - {id: mercury-edit-2, allowance: 100M shared signup tokens, paid_rate: $0.25 input + $0.75 output per 1M tokens, basis: Best case assigns the shared envelope to output, currency: USD, one_time: 75}
      sources:
        - {type: official_pricing, title: Inception models and pricing, url: https://docs.inceptionlabs.ai/get-started/models}
    poolside_direct_api: {status: not_quantifiable, reason: "The promotion has no published numeric allowance, paid comparison, or expiry."}
    voyage_ai:
      status: quantified
      cadence: one_time
      qualifier: at_least
      total: {currency: USD, one_time: 114}
      basis: Each captured per-model token package is multiplied by Voyage's current overage price. The displayed provider figure is the largest defensible single package, not a sum, because cross-model quota independence and a current code-model naming conflict remain unresolved.
      model_values:
        - {id: voyage-4-large, allowance: 200M tokens once, paid_rate: $0.12 per 1M tokens, currency: USD, one_time: 24}
        - {id: voyage-4, allowance: 200M tokens once, paid_rate: $0.06 per 1M tokens, currency: USD, one_time: 12}
        - {id: voyage-4-lite, allowance: 200M tokens once, paid_rate: $0.02 per 1M tokens, currency: USD, one_time: 4}
        - {id: voyage-context-4, allowance: 200M tokens once, paid_rate: $0.12 per 1M tokens, currency: USD, one_time: 24}
        - {id: voyage-multilingual-2 / finance-2 / law-2 / code-2, allowance: 50M tokens per named model once, paid_rate: $0.12 per 1M tokens, currency: USD, one_time: 6}
        - {id: voyage-multimodal-3.5 or multimodal-3, allowance: 200M text tokens + 150B pixels once, paid_rate: $0.12 per 1M tokens + $0.60 per 1B pixels, currency: USD, one_time: 114}
        - {id: rerank-2.5 or rerank-2, allowance: 200M tokens once, paid_rate: $0.05 per 1M tokens, currency: USD, one_time: 10}
        - {id: rerank-2.5-lite or rerank-2-lite, allowance: 200M tokens once, paid_rate: $0.02 per 1M tokens, currency: USD, one_time: 4}
      sources:
        - {type: official_pricing, title: Voyage pricing and free tokens, url: https://docs.voyageai.com/docs/pricing}
    ai21_studio:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, daily: 0.1111, weekly: 0.7778, monthly: 3.33, one_time: 10}
      basis: The $10 signup balance is spread evenly over its three-month validity for period comparisons. It is a one-time credit, not recurring monthly value.
      model_values:
        - {id: Any eligible AI21 API model, allowance: $10 shared credit valid 3 months, paid_rate: Current AI21 catalog rate, currency: USD, daily: 0.1111, weekly: 0.7778, monthly: 3.33, one_time: 10}
      sources:
        - {type: official_pricing, title: AI21 usage and cost, url: https://docs.ai21.com/docs/usage-cost}
    deepgram:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, daily: 0.5476, weekly: 3.83, monthly: 16.67, one_time: 200}
      basis: The $200 signup balance is spread evenly across its one-year validity for period comparisons. It is not recurring.
      model_values:
        - {id: Any public Deepgram model endpoint, allowance: $200 shared credit valid 1 year, paid_rate: Current Deepgram endpoint rate, currency: USD, daily: 0.5476, weekly: 3.83, monthly: 16.67, one_time: 200}
      sources:
        - {type: official_pricing, title: Deepgram pricing, url: https://deepgram.com/pricing}
    jina_ai_search_foundation:
      status: quantified
      cadence: one_time
      qualifier: up_to
      total: {currency: USD, one_time: 0.50}
      basis: The 10M shared signup tokens are valued against the live first-party model catalog. Most captured models cost $0.05 per 1M input tokens, while the two nano models cost $0.02 per 1M; rows are alternative maxima for the same balance and are not additive. No period normalization is shown because Jina does not document the signup-token expiry.
      model_values:
        - {id: Current $0.05/M Jina models, allowance: 10M shared signup tokens, paid_rate: $0.05 per 1M input tokens, currency: USD, one_time: 0.50}
        - {id: jina-embeddings-v5-omni-nano / jina-embeddings-v5-text-nano, allowance: 10M shared signup tokens, paid_rate: $0.02 per 1M input tokens, currency: USD, one_time: 0.20}
      sources:
        - {type: live_catalog, title: Jina model catalog and prices, url: https://api.jina.ai/v1/models}
        - {type: official_product, title: Jina Reader and Search API limits, url: https://jina.ai/reader/}
    jina_ai_reader:
      status: partial
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 14.4
        weekly: 100.8
        monthly: 438.3
      basis: 'Modeled request envelope for the anonymous Reader tier: the documented keyless limit of 20 requests/minute per IP (28,800/day) is multiplied by a modeled 10,000 tokens per request and priced at Jina''s current $0.05 per 1M token API rate, the rate the live first-party catalog lists for ReaderLM-v2 and all standard models and the rate keyed Reader calls pay when Jina counts the tokens of the output response. The per-request size is modeled, not documented: r.jina.ai publishes no per-request or per-minute token cap at any tier, so the assumption borrows Jina''s only published per-request token quantity for web-content retrieval, the adjacent s.jina.ai charge of "a fixed number of tokens, starting from 10000 tokens" per request; individual Reader responses can be larger or smaller. Assumes uninterrupted single-IP saturation with no latency or availability loss. The anonymous Segmenter charges zero tokens even when keyed, so it adds no paid-equivalent value at Jina''s own rates, and s.jina.ai is blocked without a key; the separate 10M-token keyed signup balance is valued in the jina_ai_search_foundation record and is not double counted here.'
      model_values:
      - id: r.jina.ai anonymous Reader (ReaderLM-v2 pipeline)
        allowance: 20 requests/minute/IP with unmetered output tokens
        paid_rate: $0.05 per 1M tokens (Jina API token rate; keyed Reader bills output-response tokens)
        basis: Modeled at 10000 tokens/request
        currency: USD
        daily: 14.4
        weekly: 100.8
        monthly: 438.3
      calculation:
        method: modeled_request_envelope
        model_id: jina-ai/ReaderLM-v2
        currency: USD
        assumptions:
        - quantity: max_output_tokens_per_request
          value: 10000
          basis: Jina's own fixed per-request token charge for the adjacent s.jina.ai endpoint ('every request costs a fixed number of tokens, starting from 10000 tokens'); r.jina.ai bills actual output tokens with no documented per-request cap
          source_url: https://jina.ai/reader/
        - quantity: usd_per_million_tokens
          value: 0.05
          basis: Current per-token USD rate for jina-ai/ReaderLM-v2 and all standard models in the live first-party catalog
          source_url: https://api.jina.ai/v1/models
        limits:
          requests_per_minute: 20
          max_output_tokens_per_request: 10000
        prices:
          output_per_million: 0.05
        result:
          daily:
            input_tokens: 0
            output_tokens: 288000000
            total_tokens: 288000000
            value: 14.4
          weekly:
            input_tokens: 0
            output_tokens: 2016000000
            total_tokens: 2016000000
            value: 100.8
          monthly:
            input_tokens: 0
            output_tokens: 8766000000
            total_tokens: 8766000000
            value: 438.3
      sources:
      - type: official_product
        title: Reader API rate-limit table and token-billing rules
        url: https://jina.ai/reader/
      - type: live_catalog
        title: Jina model catalog with current per-token USD rates
        url: https://api.jina.ai/v1/models
    mancer_ai: {status: not_quantifiable, reason: "Free-model quota and paid comparison are not published numerically."}
    mara_inference_cloud:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, daily: 0.1667, weekly: 1.17, monthly: 5, one_time: 5}
      basis: Face value of the $5 signup balance spread over its 30-day validity. Model availability to trial accounts remains dashboard-dependent.
      model_values:
        - {id: Any trial-eligible MARA model, allowance: $5 shared credit valid 30 days, paid_rate: Current MARA catalog rate, currency: USD, daily: 0.1667, weekly: 1.17, monthly: 5, one_time: 5}
      sources:
        - {type: official_pricing, title: MARA plans, url: https://cloud.mara.com/plans}
    assemblyai:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, one_time: 50}
      basis: Face value of the non-expiring signup credit. No daily, weekly, or monthly figure is shown because there is no validity window or recurrence.
      model_values:
        - {id: Any eligible AssemblyAI endpoint except LLM Gateway, allowance: $50 non-expiring shared credit, paid_rate: Current endpoint rate, currency: USD, one_time: 50}
      sources:
        - {type: official_help, title: AssemblyAI free signup, url: https://support.assemblyai.com/articles/5370767329-can-i-sign-up-for-free}
    speechmatics:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, one_time: 100}
      basis: Face value of the shared signup credit. No period normalization is shown because the public record does not establish an expiry window.
      model_values:
        - {id: Batch or real-time STT and TTS, allowance: $100 shared signup credit, paid_rate: Current Speechmatics endpoint rate, currency: USD, one_time: 100}
      sources:
        - {type: official_pricing, title: Speech API pricing, url: https://www.speechmatics.com/pricing}
    aws_bedrock:
      status: quantified
      cadence: one_time
      qualifier: at_least
      total: {currency: USD, daily: 0.5476, weekly: 3.83, monthly: 16.67, one_time: 100}
      basis: The base $100 AWS signup credit is spread over the six-month Free Plan. The additional earnable $100 is conditional and excluded, so this is a conservative minimum.
      model_values:
        - {id: Any trial-credit-eligible Bedrock model, allowance: $100 base credit over 6-month Free Plan, paid_rate: Current regional Bedrock rate, currency: USD, daily: 0.5476, weekly: 3.83, monthly: 16.67, one_time: 100}
      sources:
        - {type: official_announcement, title: AWS Free Tier credits and six-month plan, url: https://aws.amazon.com/about-aws/whats-new/2025/07/aws-free-tier-credits-month-free-plan/}
    azure_ai_foundry:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, daily: 6.6667, weekly: 46.67, monthly: 200, one_time: 200}
      basis: The $200 general Azure signup credit is spread across its 30-day validity. Eligible Foundry inference competes with all other Azure trial spend.
      model_values:
        - {id: Any trial-credit-eligible Foundry model, allowance: $200 general cloud credit valid 30 days, paid_rate: Current regional Foundry rate, currency: USD, daily: 6.6667, weekly: 46.67, monthly: 200, one_time: 200}
      sources:
        - {type: official_trial, title: Azure free account, url: https://azure.microsoft.com/free/}
    google_vertex_ai:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, daily: 3.3333, weekly: 23.33, monthly: 100, one_time: 300}
      basis: The $300 general Google Cloud credit is spread across its 90-day validity. Eligible Vertex services share it with other cloud usage.
      model_values:
        - {id: Any trial-credit-eligible Google-managed Vertex model, allowance: $300 general cloud credit valid 90 days, paid_rate: Current regional Vertex rate, currency: USD, daily: 3.3333, weekly: 23.33, monthly: 100, one_time: 300}
      sources:
        - {type: official_trial, title: Google Cloud free program, url: https://docs.cloud.google.com/free/docs/free-cloud-features}
    oracle_oci_generative_ai:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, daily: 10, weekly: 70, monthly: 300, one_time: 300}
      basis: The $300 general OCI signup credit is spread across its 30-day validity. Generative AI shares it with other OCI services.
      model_values:
        - {id: Any trial-credit-eligible OCI Generative AI model, allowance: $300 general cloud credit valid 30 days, paid_rate: Current regional OCI rate, currency: USD, daily: 10, weekly: 70, monthly: 300, one_time: 300}
      sources:
        - {type: official_trial, title: OCI Free Tier, url: https://docs.oracle.com/en-us/iaas/Content/FreeTier/freetier.htm}
    replicate:
      status: not_quantifiable
      reason: 'The free allowance is a deliberately undisclosed number of runs on the rotating try-for-free collection of media models: Replicate documents the granted-credit rate limit (1 request/second, maximum 6 requests/minute without a payment method) and per-model paid prices, but publishes no credit face value, run count, duration, or reset anywhere in its billing, prepaid-credit, or rate-limit documentation, and its own "free for a limited number of runs" and "after a bit you''ll be asked to set up billing" language contradicts any assumed saturation window, so no finite envelope survives the model-spec or modeled tiers.'
      sources:
      - type: official_docs
        title: Billing (select models free until billing is required)
        url: https://replicate.com/docs/topics/billing
      - type: official_docs
        title: Prediction rate limits including the granted-credit limit
        url: https://replicate.com/docs/topics/predictions/rate-limits
      - type: official_catalog
        title: "Try-for-free collection (limited number of runs, count undisclosed)"
        url: https://replicate.com/collections/try-for-free
    cerebras_inference:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, daily: 0.1667, weekly: 1.17, monthly: 5, one_time: 5}
      basis: Face value of the $5 signup balance spread across its 30-day validity. The two published model rate limits do not create separate $5 balances.
      model_values:
        - {id: gpt-oss-120b or gemma-4-31b, allowance: $5 shared credit valid 30 days, paid_rate: Current Cerebras model rate, currency: USD, daily: 0.1667, weekly: 1.17, monthly: 5, one_time: 5}
      sources:
        - {type: official_docs, title: Cerebras trial and rate limits, url: https://inference-docs.cerebras.ai/support/rate-limits}
    clarifai:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, daily: 0.1667, weekly: 1.17, monthly: 5, one_time: 5}
      basis: Face value of one welcome bonus spread across its 30-day validity. A possible second bonus is excluded because eligibility is conditional.
      model_values:
        - {id: Any welcome-credit-eligible Clarifai model, allowance: $5 shared credit valid 30 days, paid_rate: Current Clarifai compute rate, currency: USD, daily: 0.1667, weekly: 1.17, monthly: 5, one_time: 5}
      sources:
        - {type: official_docs, title: Clarifai account billing, url: https://docs.clarifai.com/control/account-billing/}
    fireworks_ai:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, one_time: 1}
      basis: Face value of the one-time signup credit. No period normalization is shown because the public offer does not state an expiry.
      model_values:
        - {id: Any signup-credit-eligible Fireworks model, allowance: $1 shared signup credit, paid_rate: Current Fireworks model rate, currency: USD, one_time: 1}
      sources:
        - {type: official_pricing, title: Fireworks pricing, url: https://fireworks.ai/pricing}
    nebius_token_factory:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, one_time: 1}
      basis: Face value of the one-time signup credit. No period normalization is shown because the public offer does not state an expiry.
      model_values:
        - {id: Any signup-credit-eligible Token Factory model, allowance: $1 shared signup credit, paid_rate: Current Token Factory rate, currency: USD, one_time: 1}
      sources:
        - {type: official_pricing, title: Token Factory prices, url: https://nebius.com/token-factory/prices}
    novita_ai:
      status: not_quantifiable
      reason: 'The decisive terms remain authenticated-only or unbounded: the new-user voucher''s amount and expiry are disclosed only in the signed-in dashboard (third-party $0.50 reports are not first-party evidence), and while the live catalog now lists genuinely zero-priced LLM routes with documented per-request output ceilings (inclusionai/ling-3.0-tiny at 32,768 and nex-agi/nex-n2-pro at 262,144 max output tokens; the latter priced $0.25/$1.00 per 1M on OpenRouter), Novita''s public rate-limit documentation covers only image and video models and publishes no LLM requests-per-minute or tokens-per-minute ceiling, so no finite period envelope can be built for the zero-priced routes without inventing a request rate.'
      sources:
      - type: official_quickstart
        title: Quickstart confirming an unquantified new-user voucher
        url: https://novita.ai/docs/guides/quickstart
      - type: live_catalog
        title: Novita LLM models API with zero-priced routes and output caps
        url: https://api.novita.ai/v3/openai/models
      - type: official_docs
        title: Rate limits page documenting image/video defaults only
        url: https://novita.ai/docs/guides/model-apis-rate-limits
    hyperbolic:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, one_time: 1}
      basis: Face value of the one-time signup credit. No period normalization is shown because no expiry is captured.
      model_values:
        - {id: Any signup-credit-eligible Hyperbolic model, allowance: $1 shared signup credit, paid_rate: Current Hyperbolic model rate, currency: USD, one_time: 1}
      sources:
        - {type: official_billing_docs, title: Hyperbolic billing and payments, url: https://www.hyperbolic.ai/docs/general/billing-payments}
    waterfall: {status: not_quantifiable, reason: "The community quota and model prices are not published numerically."}
    logfare: {status: not_quantifiable, reason: "Fair-use access has no numeric allowance or like-for-like paid price."}
    bazaarlink: {status: not_quantifiable, reason: "Requests per day are known, but token size and paid model rates are not documented for the free router."}
    dreamprompting:
      status: quantified
      cadence: recurring
      qualifier: up_to
      total:
        currency: USD
        daily: 5.0
        weekly: 35.0
        monthly: 152.19
      basis: The documented rolling 24-hour per-key pool of 500,000 tokens (treated as shared input+output) assigned entirely to output of the costliest currently listed caller-selectable model, cohere/command-a-03-2025, at Cohere's current first-party rate of $2.50 input / $10.00 output per 1M tokens. Request ceilings (5,000 requests/24h, 100 requests/minute per IP, 32,000 max input and 8,192 max output tokens per request) never bind. The API documents explicit provider/model selection, so the prior unknown-router-mix objection no longer applies; cohere/command-r-plus-08-2024, also listed live, carries the identical $2.50/$10.00 first-party rate and yields the same maximum. Assumes uninterrupted saturation and upstream availability of this young aggregator's donated/free upstream capacity.
      model_values:
      - id: cohere/command-a-03-2025
        allowance: Shared 500k tokens per rolling 24h per key; 8,192 max output tokens per request
        paid_rate: $2.50 input / $10.00 output per 1M tokens (Cohere first-party rate, corroborated by the Cohere-served OpenRouter listing)
        currency: USD
        daily: 5.0
        weekly: 35.0
        monthly: 152.19
      - id: cohere/command-r-plus-08-2024
        allowance: Same shared pool; equal alternative maximum, not additive
        paid_rate: $2.50 input / $10.00 output per 1M tokens (Cohere pricing page)
        basis: Equal alternative maximum for the same shared pool; not additive.
        currency: USD
        daily: 5.0
        weekly: 35.0
        monthly: 152.19
      calculation:
        method: best_case_rate_limit_envelope
        model_id: cohere/command-a-03-2025
        currency: USD
        pricing_snapshot: '2026-08-22'
        limits:
          requests_per_minute: 100
          requests_per_day: 5000
          tokens_per_day: 500000
          max_input_tokens_per_request: 32000
          max_output_tokens_per_request: 8192
        limits_notes: The rolling 24-hour windows are treated as daily periods. The 500k-token pool is the tightest constraint (5,000 requests x 8,192 output tokens would allow 40.96M) and is allocated output-first.
        prices:
          input_per_million: 2.5
          output_per_million: 10.0
        result:
          daily:
            input_tokens: 0
            output_tokens: 500000
            total_tokens: 500000
            value: 5.0
      sources:
      - type: official_docs
        title: DreamPrompting API docs (limits and model selection)
        url: https://dreamprompting.com/api-docs
      - type: official_catalog
        title: DreamPrompting live model list
        url: https://dreamprompting.com/models
      - type: official_pricing
        title: Cohere pricing
        url: https://cohere.com/pricing
      - type: router_pricing
        title: OpenRouter cohere/command-a-03-2025 (Cohere-served)
        url: https://openrouter.ai/cohere/command-a-03-2025
    ch_at: {status: not_quantifiable, reason: "The service does not expose a stable model ID or a commercial paid counterpart."}
    opentyphoon: {status: not_quantifiable, reason: "Research rate limits do not define a finite allocation or commercial paid comparison."}
    alcf_inference_endpoints: {status: not_quantifiable, reason: "User allocation is project-specific and the facility has no like-for-like commercial unit price."}
    fikra_api:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, one_time: 0.15}
      basis: The 300K-token signup allowance is multiplied by Fikra's published pay-as-you-go rate of 2M tokens per $1. All named models share the balance.
      model_values:
        - {id: fikra-fast-8b / fikra-pro-20b / fikra-pro-120b, allowance: 300K shared signup tokens, paid_rate: $1 per 2M tokens, currency: USD, one_time: 0.15}
      sources:
        - {type: official_pricing, title: Fikra API pricing, url: https://fikraapi.co.ke/}
    sarvam_ai:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: INR, one_time: 100}
      basis: Face value of the non-expiring signup credit in its billed currency. It is not converted to USD because exchange rates are volatile and the credit itself is INR-denominated.
      model_values:
        - {id: Any credit-eligible Sarvam model, allowance: ₹100 shared non-expiring credit, paid_rate: Current Sarvam INR rate, currency: INR, one_time: 100}
      sources:
        - {type: official_pricing, title: Sarvam API pricing, url: https://web.sarvam.dev/api-pricing}
    byteplus_modelark:
      status: partial
      cadence: one_time
      qualifier: subtotal
      total: {currency: USD, one_time: 2.40}
      basis: Three captured representative models are each assigned the documented typical 500K-token package and valued at their costlier output rate for deterministic best-case upper bounds. The subtotal treats the per-model packages as independent, as described by the offer, but exact eligibility and expiry remain dynamic in the Model Activation and Billing Center.
      reason: Models without a captured matching rate are excluded, and the public offer describes 500K as typical rather than guaranteed for every model.
      model_values:
        - {id: seed-1.6, allowance: Typical 500K tokens for this eligible model, paid_rate: "$0.25 input + $2.00 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: USD, one_time: 1.00}
        - {id: seed-1.6-flash, allowance: Typical 500K tokens for this eligible model, paid_rate: "$0.075 input + $0.30 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: USD, one_time: 0.15}
        - {id: seed-2-1-turbo, allowance: Typical 500K tokens for this eligible model, paid_rate: "$0.50 input + $2.50 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: USD, one_time: 1.25}
      sources:
        - {type: official_docs, title: BytePlus free token package, url: https://docs.byteplus.com/en/docs/modelark/1399514}
        - {type: official_pricing, title: BytePlus ModelArk pricing, url: https://docs.byteplus.com/en/docs/ModelArk/1544106}
    tencent_hunyuan:
      status: quantified
      cadence: one_time
      qualifier: up_to
      total: {currency: CNY, daily: 0.0282, weekly: 0.1974, monthly: 0.8583, one_time: 10.30}
      basis: The one-year 1M generation-token pool is assigned to the highest-priced captured eligible generation model and its costlier output rate, then the separate 1M embedding-token package is added. Generation-model rows are deterministic best-case alternatives for the same shared pool and are not additive; period figures spread the packages across one year.
      model_values:
        - {id: Hunyuan-a13b, allowance: 1M shared generation tokens for one year, paid_rate: "¥0.50 input + ¥2.00 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: CNY, daily: 0.0055, weekly: 0.0383, monthly: 0.1667, one_time: 2}
        - {id: Hunyuan-role-latest, allowance: 1M shared generation tokens for one year, paid_rate: "¥2.40 input + ¥9.60 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: CNY, daily: 0.0263, weekly: 0.1840, monthly: 0.80, one_time: 9.60}
        - {id: Hunyuan-translation, allowance: 1M shared generation tokens for one year, paid_rate: "¥1.20 input + ¥3.60 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: CNY, daily: 0.0099, weekly: 0.0690, monthly: 0.30, one_time: 3.60}
        - {id: Hunyuan-translation-lite, allowance: 1M shared generation tokens for one year, paid_rate: "¥1.00 input + ¥3.00 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: CNY, daily: 0.0082, weekly: 0.0575, monthly: 0.25, one_time: 3}
        - {id: Tencent-HY-Vision-1.5-Instruct, allowance: 1M shared generation tokens for one year, paid_rate: "¥3.00 input + ¥9.00 output per 1M tokens", basis: Best case assigns the shared envelope to output, currency: CNY, daily: 0.0246, weekly: 0.1725, monthly: 0.75, one_time: 9}
        - {id: Hunyuan-embedding, allowance: 1M separate embedding tokens for one year, paid_rate: ¥0.70 per 1M tokens, currency: CNY, daily: 0.0019, weekly: 0.0134, monthly: 0.0583, one_time: 0.70}
      sources:
        - {type: official_docs, title: Tencent Hunyuan free resource package and pricing, url: https://cloud.tencent.com/document/product/1729/97731}
    upstage:
      status: partial
      cadence: one_time
      qualifier: at_least
      total: {currency: USD, one_time: 10}
      basis: Face value of the documented general signup credit. No period normalization is shown because its expiry is not public, and the separate approval-only institutional program is excluded.
      reason: Approved institutions can receive additional inference access for up to one year, but the program does not publish a monetary allowance.
      model_values:
        - {id: Any general signup-credit-eligible Upstage model, allowance: $10 shared signup credit, paid_rate: Current Upstage API rate, currency: USD, one_time: 10}
      sources:
        - {type: official_guide, title: Upstage Console and API guide, url: https://www.upstage.ai/blog/en/guide-1-upstage-console-api}
        - {type: official_pricing, title: Upstage API pricing, url: https://www.upstage.ai/pricing/api}
    maritaca_academic_credits:
      status: partial
      cadence: recurring
      qualifier: subtotal
      total:
        currency: BRL
        daily: 1209.6
        weekly: 8467.2
        monthly: 36817.2
      basis: Continuous saturation of the documented tier-0 per-model envelope (60 requests, 128k input tokens, and 10k output tokens per minute) for the costliest model that has both a documented tier-0 rate-limit row and a current price, sabia-4, at Maritaca's own BRL rates (R$5.00 input / R$20.00 output per 1M tokens). All figures are BRL; no currency conversion is applied. Partial because the academic credit's face amount and duration are not public and the granted credit balance, not this rate ceiling, ultimately bounds deliverable value, and because approved academic accounts are assumed to sit at the documented tier 0. Subtotal because the tier table lists per-model rows (sabiazinho-4 and sabia-3-family rows are identical) that are not summed since their independence is not explicit, and sabia-4-thinking / BR-SP / small variants are excluded for lacking a documented tier-0 row despite higher list prices (sabia-4-thinking output is R$40/1M). Assumes uninterrupted saturation with no latency or availability loss.
      model_values:
      - id: sabia-4
        allowance: Tier-0 per-model 60 RPM, 128k input TPM, 10k output TPM
        paid_rate: R$5.00 input / R$20.00 output per 1M tokens (Maritaca's own current pricing)
        currency: BRL
        daily: 1209.6
        weekly: 8467.2
        monthly: 36817.2
      - id: sabiazinho-4
        allowance: Own identical tier-0 row (60 RPM, 128k input TPM, 10k output TPM); alternative per-model maximum, not additive
        paid_rate: R$1.00 input / R$4.00 output per 1M tokens (Maritaca's own current pricing)
        basis: Alternative per-model maximum; not added to the total because per-model quota independence is not explicitly documented.
        currency: BRL
        daily: 241.92
        weekly: 1693.44
        monthly: 7363.44
      calculation:
        method: best_case_rate_limit_envelope
        model_id: sabia-4
        currency: BRL
        pricing_snapshot: '2026-08-22'
        limits:
          requests_per_minute: 60
          input_tokens_per_minute: 128000
          output_tokens_per_minute: 10000
        limits_notes: Documented tier-0 (R$0 spend) row for sabia-4; both token directions are separately documented, so each is priced directly and no shared-envelope allocation is needed. Request count never binds. The tier-0 table has no rows for sabia-4-thinking, sabia-4-small, sabiazim-4, or BR-SP variants.
        prices:
          input_per_million: 5.0
          output_per_million: 20.0
        result:
          daily:
            input_tokens: 184320000
            output_tokens: 14400000
            total_tokens: 198720000
            value: 1209.6
      sources:
      - type: official_docs
        title: Maritaca rate limits (tier 0)
        url: https://docs.maritaca.ai/pt/rate-limits
      - type: official_pricing
        title: Maritaca pricing (BRL)
        url: https://docs.maritaca.ai/pt/precos
      - type: official_program
        title: Maritaca academic credits
        url: https://www.maritaca.ai/research
    wavespeedai:
      status: quantified
      cadence: one_time
      qualifier: exact
      total: {currency: USD, one_time: 1}
      basis: Face value of the one-time signup credit. No period normalization is shown because the public offer does not state an expiry.
      model_values:
        - {id: Any trial-credit-eligible WaveSpeedAI model, allowance: $1 shared signup credit, paid_rate: Current WaveSpeedAI model rate, currency: USD, one_time: 1}
      sources:
        - {type: official_pricing, title: WaveSpeedAI pricing, url: https://wavespeed.ai/pricing}

data_governance_audit:
  as_of: "2026-08-22"
  checked_at: "2026-08-22T02:00:00-05:00"
  timezone: America/Chicago
  scope: Governing terms and first-party disclosures for prompt and response retention, model training, product improvement, human or operator access, subprocessors and routing, deletion controls, and plan-specific differences.
  interpretation: A reviewed status means the cited documents were inspected; it is not a favorable privacy rating. Missing or silent terms are recorded conservatively and never treated as a no-training or zero-retention promise.
  records:
    openrouter:
      agreements:
        terms_of_service: https://openrouter.ai/terms/
        privacy_policy: https://openrouter.ai/privacy/
      data_governance:
        review_status: reviewed
        plan_scope: OpenRouter API and chat; downstream handling varies by selected model endpoint and account privacy controls.
        prompt_retention: OpenRouter does not retain prompts or responses by default. Optional private logging retains content for at least three months and possibly longer until deletion is requested; some processing and cache features have separate retention.
        response_retention: Same policy as prompts.
        ordinary_logging: Request metadata such as model, token counts, timestamps, cost, and latency is retained without prompt or response content.
        model_training: OpenRouter does not train on inputs or outputs by default. A separate opt-in permits product improvement; downstream model providers may retain or train unless endpoint policy controls exclude them.
        product_improvement: Optional input/output sharing is off by default and provides a usage discount when enabled. Anonymous prompt categorization may occur under a zero-data-retention arrangement.
        human_or_operator_access: Privately logged content is visible to the account owner or organization admins; OpenRouter says it does not access that content except where feature terms, operations, security, troubleshooting, or law require.
        subprocessors_and_routing: Requests are sent to the selected or automatically routed model provider, whose own endpoint-specific policy applies. Zero-data-retention routing can be enforced.
        deletion_controls: Account owners may request deletion of privately logged prompt and response content; retention for optional features is documented separately.
        caveat: Privacy depends on both OpenRouter settings and the final model endpoint. OpenRouter conservatively treats an unknown provider policy as retaining and training on data.
      sources:
        - {type: official_terms, title: Terms of Service, url: https://openrouter.ai/terms/}
        - {type: official_privacy, title: Privacy Policy, url: https://openrouter.ai/privacy/}
        - {type: official_docs, title: Data Collection, url: https://openrouter.ai/docs/guides/privacy/data-collection}
        - {type: official_docs, title: Zero Data Retention, url: https://openrouter.ai/docs/guides/features/zdr}
        - {type: official_docs, title: Input and Output Logging, url: https://openrouter.ai/docs/guides/features/input-output-logging}

    nous_portal:
      data_governance:
        review_status: not_found
        plan_scope: Nous Portal hosted inference API.
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: No service-specific terms, privacy policy, or data-handling documentation was found on the public Nous Research, Portal, or inference API surfaces during this audit. Do not send sensitive data based on silence.
      sources:
        - {type: official_product, title: Nous Research, url: https://nousresearch.com/}
        - {type: official_portal, title: Nous Portal, url: https://portal.nousresearch.com/}

    orcarouter:
      agreements:
        terms_of_service: https://www.orcarouter.ai/terms.html
        privacy_policy: https://www.orcarouter.ai/privacy.html
      data_governance:
        review_status: reviewed
        plan_scope: OrcaRouter gateway; upstream provider policies separately govern routed processing.
        prompt_retention: OrcaRouter states prompt and output content is processed in transit and not logged, stored, or retained by OrcaRouter.
        response_retention: Not retained by OrcaRouter.
        ordinary_logging: Request metadata needed for operation, security, and billing is retained without prompt or output content.
        model_training: OrcaRouter states it does not train models on user content.
        product_improvement: Performance and reliability improvement uses metadata rather than prompt or output content.
        human_or_operator_access: OrcaRouter states it keeps no content copy; the selected upstream provider necessarily processes the request under its own controls.
        subprocessors_and_routing: Prompts and responses are disclosed to upstream LLM providers such as OpenAI, Anthropic, Google, Together AI, and Groq; their terms and privacy policies apply.
        deletion_controls: Users may delete accounts and exercise privacy rights; metadata follows the published retention schedule and legal exceptions.
        caveat: OrcaRouter's no-retention promise covers its gateway, not the independently governed upstream provider.
      sources:
        - {type: official_terms, title: Terms of Service, url: https://www.orcarouter.ai/terms.html}
        - {type: official_privacy, title: Privacy Policy, url: https://www.orcarouter.ai/privacy.html}

    groqcloud:
      agreements:
        services_agreement: https://console.groq.com/docs/legal/services-agreement
        privacy_policy: https://groq.com/privacy-policy
        acceptable_use_policy: https://console.groq.com/docs/legal/ai-policy
        data_processing_addendum: https://console.groq.com/docs/legal/customer-data-processing-addendum
      data_governance:
        review_status: reviewed
        plan_scope: GroqCloud inference and related API features; offline negotiated terms may supersede the online agreement.
        prompt_retention: Inference inputs and outputs are not retained by default. Reliability or abuse logs may retain them up to 30 days; batch files are retained up to 30 days and fine-tuning data until deleted.
        response_retention: Same policy as prompts and feature state.
        ordinary_logging: Usage metadata is always retained and excludes customer inputs and outputs.
        model_training: Groq is not permitted to train or fine-tune models on inputs or outputs without explicit customer permission or instruction.
        product_improvement: Customer data use is limited to providing the service, customer instructions, law, reliable operation, and acceptable-use enforcement under the current agreement.
        human_or_operator_access: Access may occur as needed for reliable operation, abuse investigation, legal compliance, or customer-instructed features; zero-data-retention controls restrict reliability and abuse access.
        subprocessors_and_routing: Groq affiliates, subprocessors, and contractors receive limited rights needed to deliver the service; third-party model terms may also apply.
        deletion_controls: All customers may enable zero data retention. The agreement calls for deletion of customer data within 30 days after termination, subject to documented or legal exceptions.
        caveat: Zero data retention disables features that require stored application state; regional and offline agreements may differ.
      sources:
        - {type: official_terms, title: Groq Services Agreement, url: https://console.groq.com/docs/legal/services-agreement, updated_at: "2026-06-22"}
        - {type: official_privacy, title: Privacy Policy, url: https://groq.com/privacy-policy}
        - {type: official_aup, title: Acceptable Use and Responsible AI Policy, url: https://console.groq.com/docs/legal/ai-policy}
        - {type: official_dpa, title: Customer Data Processing Addendum, url: https://console.groq.com/docs/legal/customer-data-processing-addendum}
        - {type: official_docs, title: Your Data in GroqCloud, url: https://console.groq.com/docs/your-data}

    google_gemini_api:
      agreements:
        service_specific_terms: https://ai.google.dev/gemini-api/terms
        privacy_policy: https://policies.google.com/privacy
        prohibited_use_policy: https://policies.google.com/terms/generative-ai/use-policy
      data_governance:
        review_status: reviewed
        plan_scope: Gemini Developer API and Google AI Studio unpaid quota; paid services and EEA, Switzerland, or UK treatment differ.
        prompt_retention: Unpaid-service content may be retained and used for product and model improvement. Abuse-monitoring data is retained for 55 days. Feature-specific storage includes grounding data for 30 days and optional or stateful logs with configurable retention.
        response_retention: Generated responses follow the same unpaid-service, abuse-monitoring, and feature-specific rules.
        ordinary_logging: Google may log prompts, context, and outputs for abuse monitoring. Paid-project user logs are optional except stateful features and default to a maximum 55-day retention.
        model_training: Unpaid-service inputs and outputs may be used to improve and train Google models. Paid-service content is not used for product improvement without permission; EEA, Swiss, and UK unpaid use receives the paid-service data treatment.
        product_improvement: Allowed for unpaid services outside the stated regional exception. Paid logging datasets can be voluntarily shared for improvement and training.
        human_or_operator_access: Authorized human reviewers may read, annotate, and process unpaid-service content and content flagged for abuse review; Google says identifiers are disconnected before improvement review.
        subprocessors_and_routing: Google and its service infrastructure process content; grounding and other integrations introduce feature-specific handling under the additional terms.
        deletion_controls: Paid-project logs can use 7, 14, 28, or 55-day retention; datasets persist without a fixed period, files remain until deletion or expiry, and stateful features require explicit configuration for zero-retention behavior.
        caveat: The free tier has materially different data-use terms from the paid tier, and users are told not to submit sensitive, confidential, or personal information to unpaid services.
      sources:
        - {type: official_terms, title: Gemini API Additional Terms of Service, url: https://ai.google.dev/gemini-api/terms}
        - {type: official_privacy, title: Google Privacy Policy, url: https://policies.google.com/privacy}
        - {type: official_aup, title: Generative AI Prohibited Use Policy, url: https://policies.google.com/terms/generative-ai/use-policy}
        - {type: official_docs, title: Data logging and sharing, url: https://ai.google.dev/gemini-api/docs/logs-policy, updated_at: "2026-08-18"}
        - {type: official_docs, title: Zero data retention, url: https://ai.google.dev/gemini-api/docs/zdr, updated_at: "2026-05-28"}

    cloudflare_workers_ai:
      agreements:
        self_serve_subscription_agreement: https://www.cloudflare.com/terms/
        privacy_policy: https://www.cloudflare.com/privacypolicy/
        data_processing_addendum: https://www.cloudflare.com/cloudflare-customer-dpa/
      data_governance:
        review_status: reviewed
        plan_scope: Workers AI under Cloudflare self-serve or enterprise terms; optional Cloudflare storage products have their own retention configuration.
        prompt_retention: Inputs, outputs, embeddings, and training data are treated as Customer Content and are not stored by Workers AI unless the customer deliberately combines the service with storage products.
        response_retention: Same policy as other Customer Content.
        ordinary_logging: Operational account and service metadata is processed under the applicable Cloudflare agreement; the Workers AI data-use page does not claim that all metadata is zero retention.
        model_training: Cloudflare states it does not use Workers AI Customer Content to train models without explicit consent.
        product_improvement: Cloudflare states it does not use Workers AI Customer Content to improve Cloudflare or third-party services without explicit consent.
        human_or_operator_access: Access is governed by the applicable subscription agreement and DPA; no public Workers AI-specific human-review workflow is documented.
        subprocessors_and_routing: Cloudflare hosts inference for third-party model weights; model licenses may separately govern permitted use, while Cloudflare remains the service processor.
        deletion_controls: Customer-controlled storage services such as R2, KV, Durable Objects, or Vectorize determine storage and deletion when used with Workers AI.
        caveat: The no-training and no-improvement commitments apply to Customer Content in Workers AI; separately enabled storage and other Cloudflare products can persist that content.
      sources:
        - {type: official_terms, title: Self-Serve Subscription Agreement, url: https://www.cloudflare.com/terms/}
        - {type: official_privacy, title: Privacy Policy, url: https://www.cloudflare.com/privacypolicy/}
        - {type: official_dpa, title: Customer Data Processing Addendum, url: https://www.cloudflare.com/cloudflare-customer-dpa/}
        - {type: official_docs, title: Your Data and Workers AI, url: https://developers.cloudflare.com/workers-ai/platform/data-usage/, updated_at: "2026-04-21"}

    nvidia_build:
      agreements:
        technology_access_terms: https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA_Technology_Access_TOU.pdf
        privacy_policy: https://www.nvidia.com/en-us/about-nvidia/privacy-policy/
      data_governance:
        review_status: reviewed
        plan_scope: NVIDIA API Catalog and NVIDIA-hosted NIM trial access under the Technology Access Terms; a model or product-specific agreement can supersede those terms.
        prompt_retention: No fixed prompt-retention period is promised. The terms permit NVIDIA, affiliates, and service providers to host and store User Content to provide and support the service, for security, and to improve products or underlying technology.
        response_retention: No inference-output-specific retention commitment was found in the governing trial terms.
        ordinary_logging: NVIDIA may monitor, scan, or review communications and User Content transmitted through its servers for safety, security, moderation, or legal requests.
        model_training: The terms do not make a narrow no-training commitment. Their User Content license permits modification and improvement of NVIDIA products, services, and underlying technology, so users should treat training or equivalent improvement use as permitted unless a product-specific agreement says otherwise.
        product_improvement: Expressly permitted under the User Content license.
        human_or_operator_access: Monitoring, scanning, and review are permitted for security, moderation, and legal purposes; service providers may process content under the license.
        subprocessors_and_routing: NVIDIA affiliates and service providers may process User Content, and separately identified product agreements can apply to individual models or services.
        deletion_controls: The terms do not guarantee permanent access or recovery if data is deleted or lost; public User Content may remain after reposting, and NVIDIA accepts no general duty to remove it.
        caveat: Unless a separate product agreement expressly permits it, User Content must not contain confidential information, personal data, protected health information, payment-card data, or sensitive human-subject research. This makes the free trial unsuitable for sensitive prompts.
      sources:
        - {type: official_terms, title: NVIDIA Technology Access Terms of Use, url: https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA_Technology_Access_TOU.pdf}
        - {type: official_privacy, title: NVIDIA Privacy Policy, url: https://www.nvidia.com/en-us/about-nvidia/privacy-policy/}

    mistral:
      agreements:
        terms_of_service: https://legal.mistral.ai/terms
        privacy_policy: https://legal.mistral.ai/terms/privacy-policy?language=en-US
        data_processing_addendum: https://legal.mistral.ai/terms/data-processing-addendum
      data_governance:
        review_status: reviewed
        plan_scope: Mistral Studio and API. Free, pay-as-you-go, enterprise, Labs, and stateful features have materially different data-use and retention rules.
        prompt_retention: Free-mode inputs and outputs may be retained for model improvement unless the user opts out. Paid stateless API traffic can qualify for zero data retention when approved; stateful features necessarily store application data.
        response_retention: Follows the same plan, endpoint, and feature-specific rules as prompts.
        ordinary_logging: Mistral processes technical and usage logs for service delivery, security, abuse prevention, and legal compliance; zero-data-retention eligibility does not eliminate all account or operational metadata.
        model_training: Free Mistral Studio and API usage may be used to train and improve models by default, with an opt-out. Pay-as-you-go API usage is opted out by default. Labs models may use data for training regardless of the general setting.
        product_improvement: Depends on plan and opt-out status. Free usage is eligible by default; paid API traffic is excluded by default, subject to feature and model exceptions.
        human_or_operator_access: Content retained for improvement, safety, support, or legal reasons may be accessed by authorized personnel under Mistral's controls; no universal operator-blind commitment applies to free mode.
        subprocessors_and_routing: Mistral and its listed subprocessors process service data under the privacy policy and DPA; third-party integrations can add separate terms.
        deletion_controls: Users can opt out of training in privacy controls. Zero data retention is available only for eligible paid stateless endpoints and requires approval; stored conversations, files, fine-tunes, and other stateful resources must be deleted through their feature controls.
        caveat: Mistral's general API privacy documentation and its free-mode documentation are easy to read as conflicting. The narrower free-mode disclosure controls this audit's free-tier finding, and Labs models remain an explicit exception.
      sources:
        - {type: official_terms, title: Mistral Terms of Service, url: https://legal.mistral.ai/terms}
        - {type: official_privacy, title: Mistral Privacy Policy, url: "https://legal.mistral.ai/terms/privacy-policy?language=en-US"}
        - {type: official_dpa, title: Mistral Data Processing Addendum, url: https://legal.mistral.ai/terms/data-processing-addendum}
        - {type: official_docs, title: Zero Data Retention, url: https://docs.mistral.ai/admin/monitor-comply/zero-data-retention}
        - {type: official_docs, title: Privacy and data controls, url: https://docs.mistral.ai/admin/monitor-comply/privacy-data-controls}

    cohere:
      agreements:
        trial_terms: https://cohere.com/terms-of-use
        saas_agreement: https://cohere.com/saas-agreement
        privacy_policy: https://cohere.com/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Cohere trial API and enterprise SaaS; trial, enterprise, training opt-in, and approved zero-data-retention configurations differ.
        prompt_retention: Cohere says logged prompts and generations are automatically deleted after 30 days, except for legal or contractual requirements, flagged abuse, or content separately allowed for training. Approved zero-data-retention accounts do not log prompts or generations.
        response_retention: Same 30-day default and exceptions as prompts.
        ordinary_logging: Cohere logs and monitors platform use for agreement enforcement and security, and collects non-customer-identifying usage measures such as frequency, duration, features, preferences, and aggregate input-token counts.
        model_training: Training use is opt-in. Cohere says common personal information is filtered from opted-in prompts and generations before possible model training; separately submitted fine-tuning data is governed by its feature and agreement terms.
        product_improvement: Aggregate usage data may be used to understand use and improve performance. Prompt or generation training use requires opt-in, while de-identified flagged content may be aggregated to evaluate safety detection and policy enforcement.
        human_or_operator_access: Safety and security teams may review prompts, generations, and logs flagged as possible misuse.
        subprocessors_and_routing: Cohere uses published subprocessors, including Google Cloud infrastructure and monitoring, delivery, support, and analytics vendors; third-party platforms integrating Cohere can impose their own handling.
        deletion_controls: Trial users can delete the platform account and request deletion of inadvertently submitted personal information. Enterprise data is normally deleted after 30 days, while approved ZDR purges content after processing and agreement-specific exceptions can apply.
        caveat: The free trial is not intended to process personal information. Zero data retention is not the default and requires Cohere approval.
      sources:
        - {type: official_terms, title: Cohere Terms of Use, url: https://cohere.com/terms-of-use}
        - {type: official_terms, title: Cohere SaaS Agreement, url: https://cohere.com/saas-agreement}
        - {type: official_privacy, title: Cohere Privacy Policy, url: https://cohere.com/privacy, updated_at: "2026-05-01"}
        - {type: official_docs, title: Enterprise Data Commitments, url: https://cohere.com/enterprise-data-commitments}
        - {type: official_trust, title: Cohere Trust Center, url: https://trustcenter.cohere.com/}

    ibm_watsonx_ai_runtime:
      agreements:
        cloud_service_terms: https://www.ibm.com/support/customer/csol/terms/
        privacy_statement: https://www.ibm.com/privacy
        data_processing_addendum: https://www.ibm.com/support/customer/csol/terms/?id=i126-7875
      data_governance:
        review_status: reviewed
        plan_scope: IBM watsonx.ai SaaS foundation-model inference; saved assets, prompt sessions, tuning data, and third-party models have additional state and terms.
        prompt_retention: IBM states it does not access, log, or store unsaved API prompts or outputs. Content deliberately saved as project assets is stored until the customer deletes it; temporary prompt-session assets can persist for 30 days.
        response_retention: Unsaved API outputs receive the same no-log and no-store treatment; saved results follow project and feature lifecycle controls.
        ordinary_logging: IBM retains account and service-usage metadata for operations, security, metering, support, and contractual obligations without treating unsaved prompt content as ordinary logs.
        model_training: IBM states customer prompts and outputs are not used to train foundation models or improve the service without permission.
        product_improvement: Unsaved prompts and outputs are excluded from service or model improvement; separately supplied feedback, tuning data, or expressly authorized data can be governed by their feature terms.
        human_or_operator_access: IBM says unsaved API prompts and outputs are not accessed. Saved assets and support cases can be accessed as needed under IBM Cloud security and contractual controls.
        subprocessors_and_routing: IBM Cloud subprocessors and, where selected, third-party model providers may process data under the service terms and DPA; model-specific terms can add restrictions.
        deletion_controls: Customers can delete saved project assets; prompt-session assets expire after 30 days. Contract termination and personal-data rights follow IBM Cloud terms and privacy procedures.
        caveat: The strongest no-store statement applies to unsaved API requests. Prompt Lab history, projects, files, deployments, tuning data, support material, and third-party models can create separately retained state.
      sources:
        - {type: official_terms, title: IBM Cloud Service terms, url: https://www.ibm.com/support/customer/csol/terms/}
        - {type: official_privacy, title: IBM Privacy Statement, url: https://www.ibm.com/privacy}
        - {type: official_dpa, title: IBM Data Processing Addendum, url: "https://www.ibm.com/support/customer/csol/terms/?id=i126-7875"}
        - {type: official_docs, title: Security of foundation models in watsonx.ai, url: "https://www.ibm.com/docs/en/watsonx/saas?topic=watsonx-security-foundation-models"}

    llmapi_ai:
      agreements:
        terms_of_use: https://llmapi.ai/terms/
        data_processing_agreement: https://llmapi.ai/dpa/
      data_governance:
        review_status: reviewed
        plan_scope: LLM.API gateway. The gateway's defaults and optional All Data Mode are separate from each selected AI provider's independent terms.
        prompt_retention: Gateway default is zero content retention beyond transaction time. Optional All Data Mode retains prompts and outputs for up to 90 days for analytics, semantic caching, and debugging.
        response_retention: Same gateway policy as prompts; cached responses and All Data Mode create retained content.
        ordinary_logging: Request metadata is maintained for billing and analytics. Account, billing, fraud, security, and service-improvement data may be processed independently of customer instructions.
        model_training: LLM.API says it does not train, fine-tune, evaluate, benchmark, or improve models with content processed under its DPA. Upstream AI providers may use inputs or outputs for improvement or training under their own terms.
        product_improvement: The gateway may use controller-side operational data for service improvement, but its DPA prohibits model improvement with processor content. All Data Mode enables product features using retained content.
        human_or_operator_access: LLM.API says it does not inspect routed content except where All Data Mode is enabled; authorized personnel processing personal data must be under confidentiality obligations.
        subprocessors_and_routing: Requests go to an independently governed AI provider that is expressly not treated as LLM.API's subprocessor. Customers must assess and contract with that provider themselves.
        deletion_controls: All Data Mode can be disabled and its retained content deleted from the dashboard at any time. Upstream deletion and data-subject requests remain provider-specific.
        caveat: The gateway's zero-content-retention and no-training promises do not bind the final AI provider. The terms prohibit submitting personal, confidential, or third-party material without all necessary rights and authorizations.
      sources:
        - {type: official_terms, title: LLM.API Terms of Use, url: https://llmapi.ai/terms/, updated_at: "2026-06-12"}
        - {type: official_dpa, title: LLM.API Data Processing Agreement, url: https://llmapi.ai/dpa/, updated_at: "2026-07"}

    api_airforce:
      agreements:
        terms_of_service: https://api.airforce/terms/
        privacy_policy: https://api.airforce/privacy/
      data_governance:
        review_status: partial
        plan_scope: Api.Airforce website, dashboard, and inference gateway; downstream model-provider policies also apply.
        prompt_retention: Submitted content is retained as needed for abuse, fraud, illegal-activity detection, and service safety, but no fixed content-retention period is published.
        response_retention: Responses may be processed by third-party model providers; a gateway-specific output-retention period is not documented.
        ordinary_logging: Request counts, IP addresses, model usage, request logs, subscription data, and authentication information are collected for operations, analytics, security, rate limits, abuse prevention, and improvement.
        model_training: not_documented
        product_improvement: Analytics and request logs may be used to improve and maintain the platform; whether prompt content itself is used for model or product improvement is not clearly stated.
        human_or_operator_access: Content can be processed for abuse, fraud, illegal-activity detection, and safety; the policy does not define operator roles or access limitations.
        subprocessors_and_routing: Requests and outputs may be handled by changing third-party model providers under their own terms and policies.
        deletion_controls: Closing an account disables access and releases identifiers but is described as a reversible soft deletion. EU/EEA and equivalent-region users may request erasure, subject to legal-retention exceptions.
        caveat: The policy acknowledges content retention without a duration and does not publish a no-training commitment. Do not infer privacy from the service's proxy role.
      sources:
        - {type: official_terms, title: Api.Airforce Terms of Service, url: https://api.airforce/terms/, updated_at: "2025-12-12"}
        - {type: official_privacy, title: Api.Airforce Privacy Policy, url: https://api.airforce/privacy/, updated_at: "2025-12-12"}

    awanllm:
      agreements:
        terms_and_conditions: https://www.awanllm.com/terms
        privacy_policy: https://www.awanllm.com/privacy
      data_governance:
        review_status: reviewed
        plan_scope: AwanLLM hosted text-generation API under its public terms and privacy policy.
        prompt_retention: AwanLLM states it does not log user prompts or generations.
        response_retention: AwanLLM states generations are not logged.
        ordinary_logging: Request count and request rate are logged for rate limiting and usage tracking; account email or wallet address and session information are stored.
        model_training: No prompt or generation content is available from ordinary API logging for training; the policy does not make a broader contractual statement about independently submitted feedback or fine-tuning data.
        product_improvement: No content-based improvement use is disclosed. The published privacy policy limits tracked API data to request count and rate.
        human_or_operator_access: The policy states prompts and generations are not logged, so no stored content-review workflow is disclosed.
        subprocessors_and_routing: The privacy policy says personal information is not shared with third parties; infrastructure subprocessors and model-hosting architecture are not described in detail.
        deletion_controls: not_documented
        caveat: The policy is short and does not publish retention periods for account or metadata, deletion procedures, a DPA, or a subprocessor list.
      sources:
        - {type: official_terms, title: AwanLLM Terms and Conditions, url: https://www.awanllm.com/terms}
        - {type: official_privacy, title: AwanLLM Privacy Policy, url: https://www.awanllm.com/privacy}

    arliai:
      agreements:
        terms_and_conditions: https://www.arliai.com/terms
        privacy_policy: https://www.arliai.com/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Arli AI hosted inference API.
        prompt_retention: Arli AI states prompts and inputs are used only transiently to return a response and are not logged, stored, retained, or accessed afterward.
        response_retention: Generated outputs receive the same zero-log treatment.
        ordinary_logging: Usage metadata such as request count, model, and request parameters is logged for rate limiting and usage tracking without prompt or output content.
        model_training: Arli AI's transient-use license does not authorize training on prompts or outputs, and its zero-log design leaves no ordinary stored content for training.
        product_improvement: Aggregated or non-content usage data may support operations; the terms do not authorize prompt or output use for product improvement.
        human_or_operator_access: Arli AI states it does not store or have access to prompt and output content beyond transient request processing.
        subprocessors_and_routing: Open-weight or third-party models may carry model licenses and use restrictions, but the published terms describe Arli AI as the inference host rather than a changing external gateway.
        deletion_controls: There is no prompt history to delete under the stated zero-log design. Account and personal-data rights are governed by the privacy policy.
        caveat: Zero-log claims are first-party contractual statements, not an independently verified technical guarantee; metadata is still retained.
      sources:
        - {type: official_terms, title: Arli AI Terms and Conditions, url: https://www.arliai.com/terms}
        - {type: official_privacy, title: Arli AI Privacy Policy, url: https://www.arliai.com/privacy}

    freeinference_org:
      agreements:
        terms_of_service: https://freeinference.org/terms
      data_governance:
        review_status: reviewed
        plan_scope: FreeInference.org service, including local inference servers and requests routed to remote model providers.
        prompt_retention: All prompts and responses may be logged, stored, hashed, redacted, or otherwise processed according to operator configuration and service needs; no fixed maximum retention period is stated.
        response_retention: Same broad logging and retention policy as prompts.
        ordinary_logging: Requests are analyzed for operations, security, debugging, research, service improvement, usage statistics, and routing metrics.
        model_training: The terms authorize research and derived-data use but do not clearly state whether retained prompt or response content may train a model. Users should not treat this as a no-training commitment.
        product_improvement: Logs and derived data may be used to improve and analyze the service. Sanitized or anonymized prompts, responses, metrics, and other derived datasets may be published or open-sourced for reproducible research.
        human_or_operator_access: Operators may process retained content for research, security, debugging, improvement, and analysis; sanitization is performed where feasible but is not guaranteed to remove sensitive information.
        subprocessors_and_routing: Requests may be sent to remote model providers, which can process prompts, responses, metadata, and usage information under their own policies.
        deletion_controls: not_documented
        caveat: The terms explicitly warn users not to submit sensitive, confidential, regulated, secret, credential, or unauthorized data. Sanitization is not a privacy guarantee.
      sources:
        - {type: official_terms, title: FreeInference.org Terms of Service, url: https://freeinference.org/terms}

    fastrouter:
      agreements:
        terms_of_service: https://fastrouter.ai/terms
        privacy_policy: https://fastrouter.ai/privacy
        model_provider_terms: https://fastrouter.ai/llm-model-terms
      data_governance:
        review_status: reviewed
        plan_scope: FastRouter gateway and chat services; the selected model provider's separate terms govern model-side input and output use.
        prompt_retention: FastRouter's terms grant it a perpetual license to host, store, transfer, reproduce, adapt, and otherwise process inputs and outputs for service delivery, debugging, optimization, improvement, and compliance; no fixed deletion deadline is stated.
        response_retention: The same license and undefined retention apply to outputs.
        ordinary_logging: Account, service, API, technical, and usage information is processed under FastRouter's privacy policy, while downstream provider telemetry is independently governed.
        model_training: The terms do not provide a gateway-wide no-training promise. Downstream providers can impose their own training rules, and FastRouter's own broad improvement license is not limited to metadata.
        product_improvement: Inputs and outputs may be used to optimize and improve the service and API keys under an express perpetual license.
        human_or_operator_access: Content may be processed for chat, debugging, optimization, improvement, and legal compliance; no operator-blind or zero-access commitment is published.
        subprocessors_and_routing: Requests go to model providers including Amazon, Anthropic, Azure, DeepInfra, DeepSeek, OpenAI, Google, Groq, Perplexity, xAI, Meta, and fal.ai, each under separate terms.
        deletion_controls: not_documented
        caveat: The broad perpetual content license and provider-specific rules make this unsuitable for sensitive prompts absent a separate written agreement.
      sources:
        - {type: official_terms, title: FastRouter Terms of Service, url: https://fastrouter.ai/terms, updated_at: "2025-08-11"}
        - {type: official_privacy, title: FastRouter Privacy Policy, url: https://fastrouter.ai/privacy}
        - {type: official_terms, title: FastRouter model-provider terms directory, url: https://fastrouter.ai/llm-model-terms}

    kilo_ai_gateway:
      agreements:
        terms_of_service: https://kilo.ai/terms
        privacy_policy: https://kilo.ai/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Kilo Gateway and Kilo coding products; free and paid plans and the selected OpenRouter or downstream model route differ.
        prompt_retention: Kilo's terms grant a perpetual, irrevocable, sublicensable license to use Customer Data for service delivery and improvement. Paid plans advertise no-retention support, but that is not a default commitment for the free plan.
        response_retention: Outputs are Customer Data under the same broad terms; paid plans can support no-retention routing.
        ordinary_logging: Kilo collects account and technical data and states analytics telemetry excludes prompt or conversation content. Audit and usage logs may be available for governed plans.
        model_training: A user can decline to license data to an AI model for training, but then some models may be unavailable. Downstream model rules remain route-specific.
        product_improvement: The Customer Data license expressly permits Kilo to improve its service and other products, including development, diagnostics, and corrective work.
        human_or_operator_access: The terms authorize Kilo processing for development and diagnostics. OpenRouter and the selected downstream provider process prompt and conversation content for inference.
        subprocessors_and_routing: Prompts are routed through OpenRouter to the selected provider, including Anthropic, OpenAI, Google, xAI, Mistral, Meta, and others. Their terms and route controls also apply.
        deletion_controls: Users may delete the account from their profile or request deletion. Paid-plan no-retention support and provider-routing controls require configuration and do not erase previously retained data automatically.
        caveat: The free plan does not receive the paid no-retention statement, and Kilo's terms contain a broad perpetual data license. Model access may depend on accepting downstream training rights.
      sources:
        - {type: official_terms, title: Kilo Terms of Service, url: https://kilo.ai/terms, updated_at: "2026-01-29"}
        - {type: official_privacy, title: Kilo Privacy Policy, url: https://kilo.ai/privacy, updated_at: "2026-05-29"}
        - {type: official_security, title: Kilo Security and Compliance, url: https://kilo.ai/security-and-compliance}

    hetzner_experiments:
      agreements:
        terms_and_conditions: https://www.hetzner.com/legal/terms-and-conditions
        privacy_policy: https://www.hetzner.com/legal/privacy-policy
        data_processing_agreement: https://www.hetzner.com/AV/DPA_en.pdf
      data_governance:
        review_status: reviewed
        plan_scope: Hetzner Experiments Inference API, a best-effort experimental service for existing Hetzner customers.
        prompt_retention: Hetzner states it does not retain prompt or response content after processing unless legally compelled.
        response_retention: Same non-retention statement as prompts.
        ordinary_logging: Usage metadata is retained for operation and enforcement without prompt or response content.
        model_training: No retained inference content is available for ordinary model training, and no training right is disclosed for the experimental API.
        product_improvement: Service telemetry may support operation of the experiment; the product documentation does not authorize content-based improvement.
        human_or_operator_access: Transient access necessary to process a request is possible; retained content access is limited to a legal-compulsion exception.
        subprocessors_and_routing: Hetzner provides the hosted inference infrastructure in the EU under its terms and DPA; model licenses can separately restrict use of outputs.
        deletion_controls: Prompt and response content is not ordinarily retained. Account and personal-data rights are available through Hetzner's privacy and DPA procedures.
        caveat: This is a non-production experiment without an availability commitment. The legal-compulsion exception and retained usage metadata mean it is not a universal zero-data promise.
      sources:
        - {type: official_terms, title: Hetzner Terms and Conditions, url: https://www.hetzner.com/legal/terms-and-conditions}
        - {type: official_privacy, title: Hetzner Privacy Policy, url: https://www.hetzner.com/legal/privacy-policy}
        - {type: official_dpa, title: Hetzner Data Processing Agreement, url: https://www.hetzner.com/AV/DPA_en.pdf}
        - {type: official_docs, title: Hetzner Experiments Inference API, url: https://docs.hetzner.com/general/company-and-policy/experiments/inference/}
        - {type: official_docs, title: Hetzner Experiments Platform, url: https://docs.hetzner.com/general/company-and-policy/experiments/experiments-platform/}

    scaleway_generative_apis:
      agreements:
        ai_service_terms: https://www-uploads.scaleway.com/Conditions_Particulieres_Services_IA_61a1a5f301.pdf
        privacy_policy: https://www.scaleway.com/en/privacy-policy/
        data_processing_agreement: https://www-uploads.scaleway.com/DPA_2024_ENG_b0abb5cc26.pdf
      data_governance:
        review_status: reviewed
        plan_scope: Scaleway Generative APIs hosted in Europe; batch processing and abuse investigations have explicit exceptions to the default zero-data-retention policy.
        prompt_retention: Prompts are not ordinarily collected, read, reused, or analyzed. For harmful misuse or failures, full HTTP request content may be stored and accessed for up to two weeks. Batch inputs are stored during processing for up to 24 hours.
        response_retention: Outputs are not ordinarily collected, read, reused, or analyzed; incident-related request content and batch-state exceptions apply.
        ordinary_logging: Anonymized request metadata, status codes, timestamps, input and output token counts, and non-content parameters are retained up to six months for performance and improvement.
        model_training: Scaleway states customer data is not used to train, retrain, or improve base models.
        product_improvement: Aggregated and anonymized metadata is used to monitor and improve the API; prompt and output content is excluded absent the documented incident exception.
        human_or_operator_access: Full request content may be accessed temporarily to reproduce and fix harmful misuse, unexpected errors, or security vulnerabilities.
        subprocessors_and_routing: Scaleway says it hosts the models on its own European infrastructure without interaction with third-party model services; ordinary support and payment subprocessors remain possible.
        deletion_controls: Incident content expires within two weeks, batch input within 24 hours, and anonymized usage data within six months; account-data rights follow the general privacy policy and DPA.
        caveat: The service calls its default policy zero data retention, but explicit incident, batch, and metadata retention remain.
      sources:
        - {type: official_terms, title: Scaleway AI Services Special Conditions, url: https://www-uploads.scaleway.com/Conditions_Particulieres_Services_IA_61a1a5f301.pdf}
        - {type: official_privacy, title: Scaleway Privacy Policy, url: https://www.scaleway.com/en/privacy-policy/}
        - {type: official_dpa, title: Scaleway Data Processing Agreement, url: https://www-uploads.scaleway.com/DPA_2024_ENG_b0abb5cc26.pdf}
        - {type: official_docs, title: Generative APIs Privacy Policy, url: https://www.scaleway.com/en/docs/generative-apis/reference-content/data-privacy/, updated_at: "2025-10-03"}

    qwen_cloud:
      agreements:
        product_terms: https://www.alibabacloud.com/help/en/legal/latest/alibaba-cloud-international-website-product-terms-of-service-v-3-8-0
        privacy_policy: https://www.alibabacloud.com/help/en/legal/latest/alibaba-cloud-international-website-privacy-policy
      data_governance:
        review_status: reviewed
        plan_scope: Alibaba Cloud Model Studio international service, including Qwen and third-party models; console history and stateful features differ from stateless inference.
        prompt_retention: Inference request data is stored in the selected access region while transient forwarding data is not persisted. Playground conversation history can remain without a stated time limit until manually deleted.
        response_retention: Model Studio outputs are Member Content and can be stored by selected features; stateless inference transmission is described as transient.
        ordinary_logging: Alibaba Cloud may monitor service use for compliance and can apply automated content-detection mechanisms; billing and token-usage statistics are retained.
        model_training: Alibaba Cloud states it does not use Member Content or customer business data to develop or improve Model Studio models without separate explicit consent.
        product_improvement: Content-based model development requires consent; ordinary cloud telemetry and automated security processing remain permitted.
        human_or_operator_access: The product terms allow security monitoring and processing for service delivery and compliance but do not publish a universal operator-blind commitment.
        subprocessors_and_routing: Content may be transferred to and processed where Alibaba Cloud, affiliates, or subcontractors operate Model Studio. Third-party models have separate provider terms and risks.
        deletion_controls: Playground histories can be manually deleted. Alibaba Cloud can remove stored content on suspension or termination, and customers must maintain their own backups.
        caveat: Storage location and inference scope are region-dependent and may involve cross-border transfer. Selecting a third-party model adds independent terms not covered by Alibaba's no-training statement.
      sources:
        - {type: official_terms, title: Alibaba Cloud Product Terms including Model Studio, url: https://www.alibabacloud.com/help/en/legal/latest/alibaba-cloud-international-website-product-terms-of-service-v-3-8-0, updated_at: "2026-05-29"}
        - {type: official_privacy, title: Alibaba Cloud Privacy Policy, url: https://www.alibabacloud.com/help/en/legal/latest/alibaba-cloud-international-website-privacy-policy}
        - {type: official_privacy, title: Model Studio security certifications and privacy, url: https://www.alibabacloud.com/help/en/model-studio/privacy-notice, updated_at: "2026-05-15"}
        - {type: official_docs, title: Model Studio FAQ, url: https://www.alibabacloud.com/help/en/model-studio/faq-about-alibaba-cloud-model-studio, updated_at: "2026-06-25"}
        - {type: official_docs, title: Region and service deployment scope, url: https://www.alibabacloud.com/help/en/model-studio/regions/, updated_at: "2026-06-30"}
        - {type: official_docs, title: Qwen and Wan training-data governance, url: https://www.alibabacloud.com/help/en/model-studio/qwen-and-wan-training-data-disclosure, updated_at: "2026-03-15"}

    sea_lion_api:
      agreements:
        terms_of_use: https://sea-lion.ai/terms-of-use/
        privacy_policy: https://sea-lion.ai/privacy-policy/
      data_governance:
        review_status: reviewed
        plan_scope: AI Singapore and NUS SEA-LION website, playground, API, and associated services under the published terms.
        prompt_retention: The terms authorize use of input and output content to provide, maintain, develop, and improve the service, enforce policies, comply with law, and keep the service safe; no fixed content-retention period is published.
        response_retention: Outputs are part of Content under the same broad use terms.
        ordinary_logging: IP address, device, pages, location, search terms, cookies, and other log data are collected for secure and reliable operation; service use may be monitored for quality, improvement, and compliance.
        model_training: The terms do not expressly say whether service development and improvement includes model training, so no no-training commitment should be inferred.
        product_improvement: Input and output Content may expressly be used to develop and improve the services.
        human_or_operator_access: Monitoring is authorized for service quality, improvement, and terms compliance; the policy does not restrict review to automated processing.
        subprocessors_and_routing: AISG shares information with providers for hosting, maintenance, backup, storage, virtual infrastructure, analytics, and other operations, and third-party apps can receive selected content.
        deletion_controls: Users can request account deletion, including personal data, profile data, created content, and logs, subject to legal and compelling-interest exceptions; connected third parties must be contacted separately.
        caveat: The privacy policy specifically urges users to consider the sensitivity of information entered. Broad improvement rights and no fixed content-retention period make sensitive prompts inappropriate.
      sources:
        - {type: official_terms, title: SEA-LION Terms of Use, url: https://sea-lion.ai/terms-of-use/, updated_at: "2024-11-19"}
        - {type: official_privacy, title: SEA-LION Privacy Policy, url: https://sea-lion.ai/privacy-policy/, updated_at: "2024-11-19"}

    ai_horde:
      agreements:
        terms_of_service: https://aihorde.net/terms/
        privacy_policy: https://aihorde.net/privacy
      data_governance:
        review_status: reviewed
        plan_scope: AI Horde's central queue plus independently operated volunteer workers for text and image generation.
        prompt_retention: The central Horde says prompts and generations are held transiently in memory and deleted shortly after delivery or cancellation. A worker operator can technically modify open-source worker software to view and save every prompt.
        response_retention: Same central transient retention, with no enforceable guarantee that a volunteer worker did not save its generated result.
        ordinary_logging: The service stores account identifiers and operational queue, kudos, request, worker, and safety data; the public FAQ does not publish a single retention schedule for all metadata.
        model_training: The central service says it does not store prompt or generation details, but there is no contractual mechanism preventing a volunteer worker from saving or reusing received content, including for training.
        product_improvement: No central content-based improvement program is disclosed; safety filters inspect requests and workers can independently modify their software.
        human_or_operator_access: Each selected worker receives the full prompt and generates the output. Official documentation explicitly says technically capable worker operators can spy on and save both.
        subprocessors_and_routing: Requests are distributed to community-run computers outside AI Horde's direct physical or contractual control; workers do not receive the requester's ID or IP but do receive content.
        deletion_controls: Central prompt and generation state is deleted shortly after completion or cancellation. No central deletion control can retract data a worker operator may have copied.
        caveat: Treat every request as if posting to a public forum. The coordinator's transient storage does not make an untrusted volunteer-compute network suitable for confidential or personal data.
      sources:
        - {type: official_terms, title: AI Horde Terms of Service, url: https://aihorde.net/terms/}
        - {type: official_privacy, title: AI Horde Privacy Policy, url: https://aihorde.net/privacy}
        - {type: official_docs, title: AI Horde FAQ, url: https://github.com/Haidra-Org/AI-Horde/blob/main/FAQ.md}

    modal:
      agreements:
        terms_of_service: https://modal.com/legal/terms
        privacy_policy: https://modal.com/legal/privacy-policy
      data_governance:
        review_status: reviewed
        plan_scope: Modal serverless compute and Modal-hosted inference endpoints; Functions, endpoints, logs, volumes, and snapshots have different retention.
        prompt_retention: Server and inference endpoint request payloads are not stored and are proxied to the customer's container. Function arguments and return values can be retained encrypted for up to seven days.
        response_retention: Endpoint responses are not stored; Function return values can remain up to seven days. Customer-created volumes and images persist until deleted.
        ordinary_logging: App and container logs persist one day on Starter, 30 days on Team, and per contract on Enterprise; account and resource metadata persists for the account lifetime.
        model_training: Modal contractually says it will not train an AI model on Customer Data or export it into an LLM without prior written customer consent.
        product_improvement: Modal collects aggregate de-identified operational data, but Customer Data is licensed only as necessary to provide the service and is excluded from model training without consent.
        human_or_operator_access: Modal says it will not access code, function inputs or outputs, or stored customer data. Logs and metadata are accessed only with customer permission for troubleshooting, subject to service/security exceptions in the terms.
        subprocessors_and_routing: Modal pools compute across cloud providers and lists AI-tool subprocessors under its DPA; customers can select processing regions.
        deletion_controls: Inference endpoint payloads are never written to disk; function data expires within seven days; volumes and images persist until customer deletion; portability and erasure requests are supported.
        caveat: Modal is infrastructure, so code deployed by the customer can itself save or forward prompts. The endpoint ZDR statement covers Modal's proxy, not application-level logging inside the container.
      sources:
        - {type: official_terms, title: Modal Terms of Service and DPA, url: https://modal.com/legal/terms}
        - {type: official_privacy, title: Modal Privacy Policy, url: https://modal.com/legal/privacy-policy}
        - {type: official_docs, title: Security and privacy at Modal, url: https://modal.com/docs/guide/security}

    beam_cloud:
      agreements:
        terms_and_conditions: https://docs.beam.cloud/v2/security/terms-and-conditions
        privacy_policy: https://docs.beam.cloud/v2/security/privacy-policy
      data_governance:
        review_status: partial
        plan_scope: Beam cloud compute, APIs, and customer-deployed models under Smartshare's public terms.
        prompt_retention: Beam's terms say it may hold and store Your Data on the customer's behalf, but they do not publish an inference-payload retention period or a zero-retention default.
        response_retention: Outputs and other customer-created data can be stored under the same customer-data provisions; no fixed deletion schedule is published.
        ordinary_logging: Account, browser, IP, site-use, and service metadata are collected, and customer queries, submitted models, and usage metadata may be processed to measure and improve the service.
        model_training: not_documented
        product_improvement: The terms expressly permit Beam to use customer data, queries, submitted models, and usage metadata to measure and improve the service.
        human_or_operator_access: Stored data may be processed to provide, monitor, support, and improve the service; the public documents do not define operator-access restrictions for inference content.
        subprocessors_and_routing: Beam uses infrastructure and service subprocessors including Google and Sentry and offers a DPA on request for EU personal data.
        deletion_controls: Personal-data correction and deletion requests are available, but no service-content deletion deadline or self-service inference-history control is documented.
        caveat: The public agreement provides broad improvement permission and is silent on model training and inference retention. Do not submit sensitive data without a negotiated DPA and clearer controls.
      sources:
        - {type: official_terms, title: Beam Terms and Conditions, url: https://docs.beam.cloud/v2/security/terms-and-conditions, updated_at: "2025-01-23"}
        - {type: official_privacy, title: Beam Privacy Policy, url: https://docs.beam.cloud/v2/security/privacy-policy, updated_at: "2024-10-03"}
        - {type: official_docs, title: Beam Subprocessor List, url: https://docs.beam.cloud/v2/security/subprocessor-list}

    sail_research:
      agreements:
        terms_of_service: https://www.sailresearch.com/terms
        privacy_policy: https://www.sailresearch.com/privacy
        data_processing_agreement: https://docs.sailresearch.com/dpa
      data_governance:
        review_status: reviewed
        plan_scope: Sail Research inference jobs for all customers, including its distributed compute infrastructure and optional customer-owned storage.
        prompt_retention: Customer Content is persistently stored only in Sail's S3 buckets and automatically deleted shortly after processing, never later than 48 hours. Other processing is transient in memory.
        response_retention: Outputs are Customer Content under the same maximum 48-hour retention; customer-owned buckets follow customer configuration.
        ordinary_logging: Job identifiers, timestamps, status, routing, and state metadata are retained as needed to operate and track jobs without prompt content.
        model_training: Sail says it does not use Customer Content to train, fine-tune, or improve Sail or third-party machine-learning models.
        product_improvement: Customer Content is excluded from service improvement and marketing; ordinary operational metadata supports routing and job delivery.
        human_or_operator_access: Sail restricts customer-data access, and compute providers are not permitted to access Customer Content or store it unencrypted.
        subprocessors_and_routing: Multiple compute providers may execute jobs, while Amazon S3 stores encrypted content. Compute subprocessors are contractually restricted from accessing content.
        deletion_controls: Sail production buckets enforce a lifecycle policy with a 48-hour maximum; customers can instead use their own S3 bucket and retention policy.
        caveat: The 48-hour maximum permits longer-than-normal storage for retries, failures, and operational conditions, so this is not immediate zero retention.
      sources:
        - {type: official_terms, title: Sail Research Terms, url: https://www.sailresearch.com/terms}
        - {type: official_privacy, title: Sail Research Privacy Policy, url: https://www.sailresearch.com/privacy}
        - {type: official_dpa, title: Sail Research Data Processing Agreement, url: https://docs.sailresearch.com/dpa, updated_at: "2026-03-01"}

    cartesia:
      agreements:
        terms_of_service: https://www.cartesia.ai/legal/terms
        privacy_policy: https://www.cartesia.ai/legal/privacy
        acceptable_use_policy: https://www.cartesia.ai/legal/acceptable-use
      data_governance:
        review_status: reviewed
        plan_scope: Cartesia API and web services. Default consumer/developer terms differ from enterprise TTS and STT zero-data-retention arrangements.
        prompt_retention: Inputs and outputs may be hosted, cached, stored, reproduced, and used while stored to operate, improve, and promote the service. No default fixed retention period is published.
        response_retention: Outputs are covered by the same storage and use license. Enterprise ZDR can prevent retention for eligible TTS and STT payloads.
        ordinary_logging: Account, device, service activity, content, and safety or troubleshooting information is collected under the privacy policy.
        model_training: Unless otherwise agreed, Cartesia may use inputs, outputs, and interactions for labeling, classification, moderation, and model training under a perpetual license.
        product_improvement: Default terms expressly authorize service and model improvement and enhancement with Content.
        human_or_operator_access: Cartesia reserves the right to review or monitor inputs and outputs using automated and manual tools.
        subprocessors_and_routing: Contracted service providers can receive content rights needed to operate the service; third-party services and enterprise subprocessors may separately process data.
        deletion_controls: Users can opt out of future training use, but the opt-out does not reverse prior uses or model improvements. Enterprise ZDR is available for eligible TTS and STT APIs.
        caveat: Free-tier content is training-eligible by default. The opt-out is prospective, and enterprise ZDR is neither available by default nor applicable to every product.
      sources:
        - {type: official_terms, title: Cartesia Terms of Service, url: https://www.cartesia.ai/legal/terms, updated_at: "2024-06-14"}
        - {type: official_privacy, title: Cartesia Privacy Policy, url: https://www.cartesia.ai/legal/privacy}
        - {type: official_aup, title: Cartesia Acceptable Use Policy, url: https://www.cartesia.ai/legal/acceptable-use}
        - {type: official_docs, title: Cartesia Zero Data Retention, url: https://docs.cartesia.ai/enterprise/zero-data-retention}

    elevenlabs_api:
      agreements:
        terms_of_use: https://elevenlabs.io/terms-of-use
        privacy_policy: https://elevenlabs.io/privacy-policy
      data_governance:
        review_status: reviewed
        plan_scope: ElevenLabs APIs and web products; default, enterprise, product-specific, and zero-retention modes differ.
        prompt_retention: By default ElevenLabs retains API inputs and outputs for history, improvement, troubleshooting, and security. Users can delete generations, but debugging and moderation logs can remain; deleted database items remain in backups up to 30 days.
        response_retention: Same default retention as inputs. Agent conversations default to two years unless configured; some stateful products require storage.
        ordinary_logging: Request history, account, usage, security, debugging, and moderation information is retained. ZRM restricts content logging but does not cover support, account, or separately shared files.
        model_training: Non-enterprise customer data may be used to improve ElevenLabs audio models by default. Users can opt out for future submissions. Enterprise customer data is not trained on by default except as needed to provide the service.
        product_improvement: Enabled by default for eligible non-enterprise data and can be disabled in Data Use settings; previously completed improvements are not reversed.
        human_or_operator_access: Retained content can be accessed for troubleshooting, moderation, security, and service improvement. ZRM limits stored access for eligible API payloads.
        subprocessors_and_routing: ElevenLabs uses subprocessors and third-party LLM providers; it says its provider agreements prohibit training on customer content, while some image and video providers do not support ZRM.
        deletion_controls: Generations can be deleted by API, accounts can be deleted, and backups expire within 30 days. Select enterprise customers can enable ZRM for eligible API products; UI traffic is never covered.
        caveat: Free-plan data improvement is opt-out, not opt-in. ZRM is a select Enterprise control and excludes music, image/video, cloning, dubbing, Studio, support data, and UI traffic.
      sources:
        - {type: official_terms, title: ElevenLabs Terms of Use, url: https://elevenlabs.io/terms-of-use}
        - {type: official_privacy, title: ElevenLabs Privacy Policy, url: https://elevenlabs.io/privacy-policy}
        - {type: official_docs, title: ElevenLabs Zero Retention Mode, url: https://elevenlabs.io/docs/eleven-api/resources/zero-retention-mode}
        - {type: official_docs, title: Model-improvement data controls, url: https://elevenlabs.io/docs/help-center/legal/is-my-data-used-to-improve-eleven-labs-ai-models}
        - {type: official_docs, title: ElevenAgents retention, url: https://elevenlabs.io/docs/eleven-agents/customization/privacy/retention}

    aws_bedrock:
      agreements:
        customer_agreement: https://aws.amazon.com/agreement/
        service_terms: https://aws.amazon.com/service-terms/
        privacy_notice: https://aws.amazon.com/privacy/
        data_processing_addendum: https://d1.awsstatic.com/legal/aws-gdpr/AWS_GDPR_DPA.pdf
      data_governance:
        review_status: reviewed
        plan_scope: Amazon Bedrock model inference; model-specific abuse rules and customer-enabled stateful features can differ from the default.
        prompt_retention: Bedrock defaults to zero data retention and does not store model inputs or outputs. Current exceptions include up to 30-day retention for classifier-flagged OpenAI GPT-5.4/5.5/5.6 traffic and all Anthropic Claude Fable 5 traffic, plus legally required CSAM handling.
        response_retention: Same default and model-specific exceptions as inputs. Customer-created agents, knowledge bases, invocation logging, batch files, and other AWS storage persist under their configured lifecycles.
        ordinary_logging: Bedrock retains metering and operational metadata, and customers can deliberately enable model-invocation logging to CloudWatch or S3. Content is automatically screened for abuse.
        model_training: AWS says it does not use Bedrock inputs or outputs to train or improve base models and does not share them with third-party model providers for that purpose.
        product_improvement: Customer content is excluded from base-model improvement; aggregate service telemetry and customer-provided feedback are separately governed.
        human_or_operator_access: Bedrock uses a zero-operator-access model by default. Stored abuse exceptions can be reviewed only for the specified safety purpose; customer support or customer-configured logs create separate access paths.
        subprocessors_and_routing: AWS hosts access to Amazon and third-party foundation models without sharing prompts with the original model providers, but model licenses and use policies still apply.
        deletion_controls: Ordinary stateless payloads are not stored. Customers control CloudWatch, S3, agents, knowledge bases, batch inputs, and other feature state through AWS retention and deletion settings.
        caveat: Model-specific safety exceptions now prevent treating every Bedrock route as strict ZDR. Inspect the selected model and any enabled logging or stateful feature.
      sources:
        - {type: official_terms, title: AWS Customer Agreement, url: https://aws.amazon.com/agreement/}
        - {type: official_terms, title: AWS Service Terms, url: https://aws.amazon.com/service-terms/}
        - {type: official_privacy, title: AWS Privacy Notice, url: https://aws.amazon.com/privacy/}
        - {type: official_dpa, title: AWS GDPR Data Processing Addendum, url: https://d1.awsstatic.com/legal/aws-gdpr/AWS_GDPR_DPA.pdf}
        - {type: official_docs, title: Amazon Bedrock abuse detection and data handling, url: https://docs.aws.amazon.com/bedrock/latest/userguide/abuse-detection.html}

    azure_ai_foundry:
      agreements:
        customer_agreement: https://www.microsoft.com/licensing/docs/customeragreement
        product_terms: https://www.microsoft.com/licensing/terms/productoffering/MicrosoftAzure/all
        privacy_statement: https://privacy.microsoft.com/privacystatement
        data_protection_addendum: https://www.microsoft.com/licensing/docs/view/Microsoft-Products-and-Services-Data-Protection-Addendum-DPA
      data_governance:
        review_status: reviewed
        plan_scope: Microsoft Foundry Azure Direct Models, including Azure OpenAI and partner models; stateful APIs and modified abuse-monitoring customers differ.
        prompt_retention: Base inference is stateless, but prompts and completions selected for potential abuse can be stored for authorized human review. Assistants, Responses history, Stored Completions, Batch, and customer data sources intentionally persist content.
        response_retention: Same abuse-monitoring and feature-specific rules as prompts.
        ordinary_logging: Prompts and outputs are evaluated synchronously by safety systems. Usage, resource, security, and billing telemetry is retained; automated review alone does not store content.
        model_training: Microsoft says prompts, completions, embeddings, and fine-tuning data are not used to train, retrain, or improve base models without customer permission or instruction.
        product_improvement: Customer content is excluded from general model improvement absent permission; safety classification and service telemetry remain part of delivery.
        human_or_operator_access: Only content flagged for potential recurring or severe abuse enters the separated review store and can be accessed by authorized Microsoft employees through controlled, request-specific access. Approved modified-monitoring customers avoid this storage and human review.
        subprocessors_and_routing: Content stays within the Azure service boundary and selected deployment geography subject to Global or DataZone routing; partner-model licenses and Microsoft subprocessors apply.
        deletion_controls: Stateless model processing stores no model state. Customers manage Assistants, Responses, files, batches, Stored Completions, fine-tuning assets, and data sources; qualified managed customers can request modified abuse monitoring.
        caveat: “Not used for training” does not mean “never retained.” Free or unmanaged access should assume standard abuse monitoring and should avoid sensitive prompts unless the deployment's controls are confirmed.
      sources:
        - {type: official_terms, title: Microsoft Customer Agreement, url: https://www.microsoft.com/licensing/docs/customeragreement}
        - {type: official_terms, title: Microsoft Azure Product Terms, url: https://www.microsoft.com/licensing/terms/productoffering/MicrosoftAzure/all}
        - {type: official_privacy, title: Microsoft Privacy Statement, url: https://privacy.microsoft.com/privacystatement}
        - {type: official_dpa, title: Microsoft Products and Services DPA, url: https://www.microsoft.com/licensing/docs/view/Microsoft-Products-and-Services-Data-Protection-Addendum-DPA}
        - {type: official_docs, title: Data privacy and security for Azure Direct Models, url: https://learn.microsoft.com/en-us/azure/foundry/responsible-ai/openai/data-privacy}

    google_vertex_ai:
      agreements:
        cloud_terms: https://cloud.google.com/terms
        service_specific_terms: https://cloud.google.com/terms/service-terms
        cloud_data_processing_addendum: https://cloud.google.com/terms/data-processing-addendum
        privacy_notice: https://cloud.google.com/terms/cloud-privacy-notice
      data_governance:
        review_status: reviewed
        plan_scope: Generative AI on Vertex AI, including managed Google and partner models; grounding, caching, live sessions, stateful resources, and abuse-monitoring eligibility differ.
        prompt_retention: Stateless inference is not retained at rest by default, but Gemini uses project-isolated in-memory caching for up to 24 hours unless disabled. Grounding with Google Search or Maps stores prompts, context, and outputs for 30 days; session resumption caches content up to 24 hours.
        response_retention: Same feature-specific handling as prompts. Explicit context caches, batch files, tuned models, RAG stores, and other customer resources persist until their configured expiry or deletion.
        ordinary_logging: Some customers under Google Cloud Platform Terms are subject to prompt logging for abuse monitoring and can request an exception. Service, billing, and security metadata remains.
        model_training: Google contractually says it will not use customer data to train or fine-tune any AI/ML model without prior permission or instruction, including GA and pre-GA managed models.
        product_improvement: Customer content is excluded from general model improvement absent permission; grounding data may be used for debugging and testing the grounding systems during its 30-day retention.
        human_or_operator_access: Abuse and support access follows Google Cloud controls and Access Transparency where available; no universal operator-blind commitment applies to every free or feature path.
        subprocessors_and_routing: Google Cloud and its subprocessors process data in the configured region or multi-region, while partner models and grounding services can add separate terms and locations.
        deletion_controls: Customers can disable in-memory caching, request an abuse-monitoring exception where eligible, avoid stored grounding/session features, and delete explicit caches and other Vertex resources.
        caveat: Vertex AI's no-training promise is strong, but achieving zero retention requires disabling or avoiding every documented cache, grounding, session, abuse, and stateful-storage path.
      sources:
        - {type: official_terms, title: Google Cloud Terms, url: https://cloud.google.com/terms}
        - {type: official_terms, title: Google Cloud Service Specific Terms, url: https://cloud.google.com/terms/service-terms}
        - {type: official_dpa, title: Google Cloud Data Processing Addendum, url: https://cloud.google.com/terms/data-processing-addendum}
        - {type: official_privacy, title: Google Cloud Privacy Notice, url: https://cloud.google.com/terms/cloud-privacy-notice}
        - {type: official_docs, title: Vertex AI and zero data retention, url: https://docs.cloud.google.com/vertex-ai/generative-ai/docs/vertex-ai-zero-data-retention, updated_at: "2026-01-02"}

    oracle_oci_generative_ai:
      agreements:
        cloud_services_agreement: https://www.oracle.com/contracts/cloud-services/
        data_processing_agreement: https://www.oracle.com/contracts/cloud-services/
        privacy_policy: https://www.oracle.com/legal/privacy/privacy-policy.html
      data_governance:
        review_status: reviewed
        plan_scope: OCI Generative AI inference and fine-tuning; project Responses, Conversations, files, memory, agents, and other stateful resources have configurable retention.
        prompt_retention: Oracle states ordinary model-inference inputs and outputs are not stored inside OCI Generative AI. Project-based agent and OpenAI-compatible resources can deliberately retain responses and conversations under customer-selected settings.
        response_retention: Stateless outputs are not stored; project artifacts and agent sessions follow feature-specific retention.
        ordinary_logging: OCI retains service, audit, metering, security, and resource metadata under Oracle Cloud terms without treating stateless prompt content as ordinary logs.
        model_training: Oracle states customer prompts, responses, knowledge bases, and fine-tuning data are not used to improve general OCI models or services.
        product_improvement: Customer content is excluded from general service or model improvement; operational telemetry remains available for cloud operations.
        human_or_operator_access: OCI says stateless input and output are not retained or shared with third-party model providers. Saved project resources and support material remain accessible under customer IAM and Oracle Cloud controls.
        subprocessors_and_routing: OCI does not share prompts, responses, training data, or custom models with underlying third-party model providers such as Cohere, Meta, xAI, or Google Vertex AI.
        deletion_controls: Stateless content requires no deletion. Customers manage project response and conversation retention, files, memory, custom models, and Object Storage training data within their tenancy.
        caveat: The no-store statement covers ordinary inference, not newer stateful project and agent features. Verify retention settings whenever using Responses, conversations, files, containers, memory, or RAG.
      sources:
        - {type: official_terms, title: Oracle Cloud Services Agreements, url: https://www.oracle.com/contracts/cloud-services/}
        - {type: official_privacy, title: Oracle Privacy Policy, url: https://www.oracle.com/legal/privacy/privacy-policy.html}
        - {type: official_docs, title: OCI Generative AI data handling, url: https://docs.oracle.com/en-us/iaas/Content/generative-ai/data-handling.htm, updated_at: "2025-10-21"}
        - {type: official_docs, title: OCI Generative AI projects and retention, url: https://docs.oracle.com/en-us/iaas/Content/generative-ai/projects.htm, updated_at: "2026-04-17"}

    replicate:
      agreements:
        terms_of_service: https://replicate.com/terms/
        privacy_policy: https://replicate.com/privacy/
      data_governance:
        review_status: reviewed
        plan_scope: Replicate API, web predictions, marketplace models, custom deployments, and training; each model can also impose third-party terms.
        prompt_retention: API prediction inputs, outputs, files, and logs are removed after one hour by default. Web-interface prediction data is retained indefinitely until manually deleted.
        response_retention: Same API one-hour and web indefinite retention rules. Training artifacts, custom models, and customer-saved files have separate resource lifecycles.
        ordinary_logging: Prediction metadata survives removal of input and output data, and Replicate collects service-performance and resultant data for billing, operations, development, diagnostics, and improvement.
        model_training: Replicate's Customer Data license permits training customer-requested derivative models but does not expressly authorize training unrelated foundation models on ordinary inference content. Model authors or third-party offerings can add separate terms.
        product_improvement: Replicate may compile non-Customer-Data Resultant Data and use it to improve its services and products; the terms limit Customer Data use to service delivery, customer-requested derivative training, and resultant-data creation.
        human_or_operator_access: Content is stored during the stated prediction window and may be processed for delivery, support, security, and legal obligations; community model code receives the full input and must be trusted separately.
        subprocessors_and_routing: Replicate uses listed subprocessors and executes models supplied by Replicate or community authors. Secret fields are redacted after delivery, but a model author can still misuse a secret received by model code.
        deletion_controls: API content expires after one hour by default. Web predictions can be manually deleted, which removes their input and output data and files; users must manage training and model resources separately.
        caveat: API and web retention differ sharply, and running third-party model code creates a separate trust boundary. Do not assume a model author cannot inspect an input merely because Replicate later deletes it.
      sources:
        - {type: official_terms, title: Replicate Terms of Service, url: https://replicate.com/terms/, updated_at: "2026-04-01"}
        - {type: official_privacy, title: Replicate Privacy Policy, url: https://replicate.com/privacy/, updated_at: "2026-04-01"}
        - {type: official_docs, title: Prediction data retention, url: https://replicate.com/docs/topics/predictions/data-retention/}
        - {type: official_docs, title: Secret inputs and model-author trust, url: https://replicate.com/docs/topics/predictions/secrets}
        - {type: official_docs, title: Replicate subprocessors, url: https://replicate.com/docs/topics/site-policy/subprocessors}

    cerebras_inference:
      agreements:
        terms_of_service: https://www.cerebras.ai/terms-of-service
        privacy_policy: https://cloud.cerebras.ai/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Cerebras Cloud inference, training, and chatbot services under its public terms and Cloud privacy policy.
        prompt_retention: Cerebras states it does not retain inputs and outputs associated with inference, training, or chatbot services.
        response_retention: Same non-retention statement as prompts.
        ordinary_logging: Service logs are retained only while necessary to provide the services; account and personal data follow purpose-based retention plus legal, dispute, and enforcement exceptions.
        model_training: No ordinary retained inputs or outputs are available for model training, and no inference-content training right is disclosed in the reviewed documents.
        product_improvement: Non-content service data and user feedback may support operations and improvement; the privacy policy does not authorize using inference payloads for improvement.
        human_or_operator_access: The non-retention policy limits stored access, but transient processing, security operations, and legal obligations remain; no separate zero-operator-access promise is published.
        subprocessors_and_routing: Cerebras and its service providers process data under the privacy policy; public documentation does not describe inference as a gateway to changing external model providers.
        deletion_controls: Inference content is not retained. Other personal data can be subject to privacy-right requests and is deleted or aggregated when no longer necessary, subject to legal exceptions.
        caveat: The policy gives no precise maximum for service logs and does not call the design zero operator access. Non-retention of payloads is a first-party policy statement.
      sources:
        - {type: official_terms, title: Cerebras Terms of Service, url: https://www.cerebras.ai/terms-of-service}
        - {type: official_privacy, title: Cerebras Cloud Privacy Policy, url: https://cloud.cerebras.ai/privacy, updated_at: "2024-08-27"}
        - {type: official_legal, title: Cerebras Policies, url: https://www.cerebras.ai/policies}

    clarifai:
      agreements:
        terms_of_service: https://clarifai.com/company/terms
        privacy_policy: https://clarifai.com/company/privacy-policy
      data_governance:
        review_status: reviewed
        plan_scope: Clarifai cloud platform, inference, stored Inputs, prediction history, and customer-created training; private and Community-shared data differ.
        prompt_retention: Inputs and resulting predictions are stored by default so customers can review, search, and manage them in the portal; no universal automatic deletion period is published.
        response_retention: Prediction history is stored by default until the customer manages or deletes it.
        ordinary_logging: Request counts, compute usage, billing, monitoring, and other non-sensitive operational metadata are logged; more detailed logging features require opt-in according to current product documentation.
        model_training: Current docs say private data is not used to train Clarifai or other platform models unless the customer explicitly shares inputs and annotations with the Community.
        product_improvement: Aggregate usage and performance patterns may improve proprietary models without private input data. The general terms also authorize use of Your Content to develop and improve the service, creating a broader contractual permission than the narrower docs describe.
        human_or_operator_access: Stored private content is treated as confidential by default and is accessible under platform, support, security, and customer-authorized controls; Community sharing makes selected content public to that context.
        subprocessors_and_routing: Clarifai and its subprocessors host the platform; third-party models can carry separate manufacturer or developer licenses.
        deletion_controls: Customers can manage stored inputs and prediction history and request account or personal-data deletion, subject to fraud, dispute, fee, and legal-retention exceptions.
        caveat: There is a material scope tension between the terms' broad service-improvement right and docs' no-private-data-training promise. The audit treats explicit training as opt-in but does not treat private content as contractually excluded from all product improvement.
      sources:
        - {type: official_terms, title: Clarifai Terms of Service, url: https://clarifai.com/company/terms}
        - {type: official_privacy, title: Clarifai Data Privacy Policy, url: https://clarifai.com/company/privacy-policy}
        - {type: official_docs, title: Clarifai Data Privacy and Security, url: https://docs.clarifai.com/resources/privacy-security/}
        - {type: official_docs, title: Clarifai Inputs Manager, url: https://docs.clarifai.com/create/inputs/}

    fireworks_ai:
      agreements:
        terms_of_service: https://fireworks.ai/terms-of-service
        privacy_policy: https://fireworks.ai/privacy-policy
      data_governance:
        review_status: reviewed
        plan_scope: Fireworks AI open-model inference and Responses API; proprietary model partners, explicit logging opt-ins, and advanced features can differ.
        prompt_retention: Open-model prompts and generations exist only in volatile memory for the request by default; prompt caches may remain in volatile memory for several minutes. Responses API stores full conversation data for 30 days when store=true, which is the default.
        response_retention: Same default ZDR for ordinary inference and 30-day stored conversation policy for Responses API.
        ordinary_logging: Token counts and service-delivery metadata are logged without content under ZDR. FireOptimizer and similar advanced features can collect content only after explicit opt-in.
        model_training: Fireworks does not log or store open-model prompt or generation data for training without explicit opt-in; proprietary model partner terms must be checked separately.
        product_improvement: Metadata can support service delivery and improvement, while content use requires an explicit logging or feature opt-in under the published open-model policy.
        human_or_operator_access: Default open-model payloads are not written to persistent storage. Stored Responses conversations and opted-in feature data can be accessed for service, support, security, and legal purposes.
        subprocessors_and_routing: Fireworks hosts open models and may offer partner models under separate model-provider terms; its subprocessors are documented through the trust center.
        deletion_controls: Responses users can set store=false or immediately delete a stored response by ID; otherwise conversation data expires after 30 days.
        caveat: “ZDR by default” does not describe the Responses API default, which is store=true. Always set store=false when using Responses and review proprietary-model terms.
      sources:
        - {type: official_terms, title: Fireworks AI Terms of Service, url: https://fireworks.ai/terms-of-service}
        - {type: official_privacy, title: Fireworks AI Privacy Policy, url: https://fireworks.ai/privacy-policy}
        - {type: official_docs, title: Fireworks AI data handling and retention, url: https://docs.fireworks.ai/guides/security_compliance/data_handling}
        - {type: official_trust, title: Fireworks Trust Center, url: https://trust.fireworks.ai/}

    nebius_token_factory:
      agreements:
        terms_of_service: https://tokenfactory.nebius.com/terms
        data_processing_agreement: https://tokenfactory.nebius.com/terms
      data_governance:
        review_status: reviewed
        plan_scope: Nebius Token Factory inference and fine-tuning; speculative decoding is enabled by default unless Zero Data Retention is selected.
        prompt_retention: By default API prompts and outputs may be stored for speculative decoding, with no public maximum period stated in the quick guide. Enabling ZDR prevents post-request storage.
        response_retention: Same default speculative-decoding storage and optional ZDR as prompts.
        ordinary_logging: Account, billing, usage, security, and service metadata is processed under the integrated terms and DPA; ZDR applies to content rather than eliminating all metadata.
        model_training: Nebius states customer data is never used to train AI models. Fine-tuning data is used only for the customer's requested training.
        product_improvement: Stored default content is used for speculative decoding to improve inference speed, not model training; ZDR opts out of that use.
        human_or_operator_access: Content is processed within Nebius infrastructure and may be stored under the default; the public guide does not promise zero operator access.
        subprocessors_and_routing: Model hosting can occur in EU, Israel, or US locations shown per endpoint. Fine-tuning artifacts are stored centrally in the EU, and US processing relies on transfer mechanisms such as SCCs.
        deletion_controls: Users can enable ZDR in account settings to prevent storage after processing. Fine-tuning datasets, artifacts, and models follow separate customer resource controls.
        caveat: ZDR is opt-in and may reduce service level. Without it, input and output storage for speculative decoding has no stated maximum retention in the reviewed public guide.
      sources:
        - {type: official_terms, title: Nebius Token Factory Terms and integrated DPA, url: https://tokenfactory.nebius.com/terms}
        - {type: official_docs, title: Nebius Token Factory Legal Quick Guide, url: https://docs.tokenfactory.nebius.com/legal/legal-quick-guide}

    requesty:
      agreements:
        terms_of_service: https://www.requesty.ai/terms
        privacy_policy: https://www.requesty.ai/privacy
        data_processing_agreement: https://www.requesty.ai/dpa
      data_governance:
        review_status: reviewed
        plan_scope: Requesty LLM gateway. Self-service and Enterprise logging defaults differ, and each selected model provider has independent data rules.
        prompt_retention: Self-service plans store prompts and outputs encrypted in the EU for up to 30 days by default. Disabling logging enables gateway ZDR prospectively; Enterprise logging is disabled by default.
        response_retention: Same plan and logging-setting policy as prompts. Encrypted backups can remain up to 30 days after deletion.
        ordinary_logging: With content logging disabled, token counts, model ID, and timestamps remain for billing and statutory accounting retention of six years.
        model_training: Requesty itself does not train models on customer content. The privacy summary says free-plan accounts can be associated with models whose upstream provider trains on content; these are labeled Training Permitted Models and are free.
        product_improvement: Requesty uses aggregated, de-identified usage for performance, capacity planning, reporting, and service improvement rather than training on identifiable prompt content.
        human_or_operator_access: Logged content can be accessed for debugging, support, security, and legal obligations under access controls. ZDR prevents Requesty content logging but not upstream provider access.
        subprocessors_and_routing: Requests go to the selected third-party model provider. Requesty's EU endpoint constrains its own routing and storage geography, not necessarily the provider's inference location or policy.
        deletion_controls: Workspace admins can disable logging at any time for new requests. Logged content expires within 30 days, backups within another 30 days, and account deletion is available subject to statutory records.
        caveat: ZDR only controls Requesty's layer. Free Training Permitted Models can allow the final provider to retain or train on prompts even when gateway logging is off.
      sources:
        - {type: official_terms, title: Requesty Terms of Service, url: https://www.requesty.ai/terms, updated_at: "2026-05-11"}
        - {type: official_privacy, title: Requesty Privacy Policy, url: https://www.requesty.ai/privacy, updated_at: "2026-05-11"}
        - {type: official_dpa, title: Requesty Data Processing Agreement, url: https://www.requesty.ai/dpa}
        - {type: official_subprocessors, title: Requesty providers and subprocessors, url: https://www.requesty.ai/privacy/subprocessors}

    voyage_ai:
      agreements:
        terms_of_service: https://www.voyageai.com/tos
        privacy_policy: https://www.voyageai.com/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Voyage AI hosted embedding and reranking APIs; customer opt-out and separately negotiated agreements can replace the standard content-use defaults.
        prompt_retention: Unless the organization opts out, Voyage may store customer content under a perpetual license for service delivery, training, and improvement. Eligible organization admins with a payment method can opt out for zero-day retention on Voyage-hosted endpoints.
        response_retention: Outputs are Customer Content under the same default license and opt-out control.
        ordinary_logging: Account, API, usage, security, billing, and operational data is processed under the privacy policy; zero-day content retention does not eliminate all metadata.
        model_training: Standard terms allow Voyage to use Customer Content to train and improve the service and its AI models unless the customer opts out.
        product_improvement: Broadly permitted by default through an irrevocable, perpetual content license; the dashboard opt-out stops future storage and training use.
        human_or_operator_access: Stored opted-in content may be processed by Voyage personnel and subprocessors for service, training, improvement, security, support, and legal purposes.
        subprocessors_and_routing: Voyage and its subprocessors host the model API; separately supplied third-party integrations can add their own terms.
        deletion_controls: An organization admin with a payment method can opt out in dashboard settings for zero-day retention. The dashboard does not allow opting back in without contacting Voyage.
        caveat: The free account cannot necessarily use the documented opt-out because a payment method is required. Treat free-tier inputs as training-eligible under the standard perpetual license.
      sources:
        - {type: official_terms, title: Voyage AI Terms of Service, url: https://www.voyageai.com/tos, updated_at: "2026-05-27"}
        - {type: official_privacy, title: Voyage AI Privacy Policy, url: https://www.voyageai.com/privacy}
        - {type: official_docs, title: Voyage AI data-policy FAQ, url: https://docs.voyageai.com/docs/faq}

    deepgram:
      agreements:
        terms_of_service: https://deepgram.com/terms
        privacy_policy: https://deepgram.com/privacy
        master_services_agreement: https://static.deepgram.com/business/MSA_20240315.pdf
      data_governance:
        review_status: reviewed
        plan_scope: Deepgram hosted speech and voice APIs; Model Improvement Partnership, request opt-out, Voice Agent history, regional, dedicated, and self-hosted modes differ.
        prompt_retention: Deepgram may retain fractional samples for model improvement and support when an account participates in the Model Improvement Partnership. Requests with mip_opt_out=true retain content only for transaction processing.
        response_retention: Transcripts and generated audio follow the same program and feature controls; Voice Agent history can be disabled, while reusable configurations and other customer resources persist.
        ordinary_logging: Usage logs and request metadata are retained for up to 90 days in the console; summarized usage can persist longer.
        model_training: Only data contractually included through the voluntary Model Improvement Partnership is used in future model training. The per-request mip_opt_out parameter excludes data from the program.
        product_improvement: Participating content can improve general speech models and support; opted-out content is processed only for the request. Aggregate service metrics remain.
        human_or_operator_access: Program data can be accessed for labeling, model development, and support under RBAC, MFA, VPN, and encryption controls; opted-out content is transient.
        subprocessors_and_routing: Deepgram hosts its own models, including its hosted Whisper implementation, and offers regional, dedicated, and self-hosted endpoints; Whisper requests are not sent to OpenAI.
        deletion_controls: Set mip_opt_out=true on every applicable API request for transient-only content handling and disable Voice Agent history where needed. Dedicated and self-hosted arrangements provide stronger customer control.
        caveat: Participation and contract defaults must be checked at the account level; the safe API posture requires sending the opt-out parameter consistently on every request.
      sources:
        - {type: official_terms, title: Deepgram Terms, url: https://deepgram.com/terms}
        - {type: official_privacy, title: Deepgram Privacy Policy, url: https://deepgram.com/privacy}
        - {type: official_terms, title: Deepgram Master Services Agreement, url: https://static.deepgram.com/business/MSA_20240315.pdf}
        - {type: official_docs, title: Deepgram Model Improvement Partnership, url: https://developers.deepgram.com/docs/the-deepgram-model-improvement-partnership-program}
        - {type: official_docs, title: Deepgram usage-log retention, url: https://deepgram.com/changelog/deepgram-log-usage-data-limited-to-90-days, published_at: "2023-10-06"}

    assemblyai:
      agreements:
        terms_of_service: https://www.assemblyai.com/legal/terms-of-service
        privacy_policy: https://www.assemblyai.com/legal/privacy-policy
        data_processing_addendum: https://www.assemblyai.com/legal/data-processing-addendum
      data_governance:
        review_status: reviewed
        plan_scope: AssemblyAI async and streaming speech APIs plus LLM Gateway; training opt-out, BAA, EU, TTL, and upstream-model choices materially change handling.
        prompt_retention: Without a configured TTL or BAA, async transcripts and text prompts can be retained indefinitely; uploaded audio deletion starts after 24 hours and completes within 48 hours. Streaming offers content ZDR only when training is opted out.
        response_retention: Async transcription artifacts follow the configured TTL, which can be as low as one hour; without TTL or BAA they are indefinite. LLM Gateway content follows customer TTL and upstream-provider rules.
        ordinary_logging: Transcript metadata is retained for logging and billing even where content ZDR applies; Usage Data can be used broadly for business purposes.
        model_training: The standard terms permit training and product improvement with Customer Data. Customers can opt out; BAA customers and EU-server use are excluded from AssemblyAI model training. All LLM Gateway upstream providers are configured no-training.
        product_improvement: Unless excluded or opted out, Customer Data and de-identified derivatives can be used for development, testing, benchmarking, marketing, and model improvement.
        human_or_operator_access: Retained production and training data can be accessed under security controls for support, model development, safety, and legal purposes; training and production environments are separate.
        subprocessors_and_routing: LLM Gateway routes to Anthropic, Google, OpenAI, and Bedrock paths with provider-specific retention; OpenAI routes retain abuse logs for 30 days, while named Anthropic and Google routes can provide ZDR.
        deletion_controls: Customers can set an async TTL down to one hour, delete transcripts by API, opt out of model training, select EU processing, and use eligible Streaming or LLM Gateway ZDR configurations.
        caveat: "Free/default use should not be assumed private: training is contractually permitted and async transcripts can persist indefinitely until the user selects an opt-out and retention control."
      sources:
        - {type: official_terms, title: AssemblyAI Terms of Service, url: https://www.assemblyai.com/legal/terms-of-service}
        - {type: official_privacy, title: AssemblyAI Privacy Policy, url: https://www.assemblyai.com/legal/privacy-policy}
        - {type: official_dpa, title: AssemblyAI Data Processing Addendum, url: https://www.assemblyai.com/legal/data-processing-addendum}
        - {type: official_docs, title: Data retention and model training, url: https://www.assemblyai.com/docs/data-retention-and-model-training}

    speechmatics:
      agreements:
        terms_of_service: https://www.speechmatics.com/legal/terms-of-service
        privacy_policy: https://www.speechmatics.com/legal/privacy-policy
      data_governance:
        review_status: reviewed
        plan_scope: Speechmatics Cloud speech-to-text and text-to-speech API and portal; batch, real-time, sharing, and explicit model-improvement opt-in differ.
        prompt_retention: Batch audio, transcripts, and configuration are retained for seven days by default; portal playback and sharing require storage. Real-time processing and negotiated deployments can differ.
        response_retention: Batch transcripts follow the seven-day default and can be removed by customer action; shared portal content remains available until deleted or the account is removed.
        ordinary_logging: Account, API, billing, diagnostic, security, and service metadata is retained under the privacy policy; API keys themselves are not stored or recorded.
        model_training: Speechmatics states customer audio is not used to train models unless the customer explicitly opts in.
        product_improvement: Opted-in audio and transcripts may improve models; otherwise content use is limited to producing outputs, portal playback, customer-directed sharing, support, and legal duties.
        human_or_operator_access: Authorized employees, affiliates, representatives, and subprocessors can access content as needed to deliver the service under confidentiality and data-processing obligations.
        subprocessors_and_routing: Speechmatics acts as processor and can use notified equivalent-obligation subprocessors; customer-directed sharing exposes stored audio and transcripts to recipients.
        deletion_controls: Batch data auto-deletes after seven days and can be deleted earlier; deleting the account removes content. On termination, personal data is returned or destroyed except where law requires retention.
        caveat: Portal playback and sharing are intentionally stored features. The seven-day default is not zero retention, although model training is opt-in.
      sources:
        - {type: official_terms, title: Speechmatics Terms of Service, url: https://www.speechmatics.com/legal/terms-of-service}
        - {type: official_privacy, title: Speechmatics Privacy Policy, url: https://www.speechmatics.com/legal/privacy-policy}
        - {type: official_docs, title: Speechmatics Cloud data retention, url: https://legacy.docs.speechmatics.com/en/cloud/introduction}
        - {type: official_product, title: Speechmatics pricing and model-improvement opt-in, url: https://www.speechmatics.com/pricing}

    jina_ai_search_foundation:
      agreements:
        terms_and_conditions: https://jina.ai/legal/
        privacy_policy: https://www.elastic.co/legal/privacy-statement
        data_processing_agreement: https://www.elastic.co/legal/customer-dpa
      data_governance:
        review_status: partial
        plan_scope: Jina AI Search Foundation APIs after Elastic's October 2025 acquisition; current Elastic data terms supersede portions of Jina's legacy legal page.
        prompt_retention: Request data, inputs, prompts, and uploaded content are processed to provide the APIs; no current public per-request retention period was found.
        response_retention: not_documented
        ordinary_logging: Operational, diagnostic, and usage metadata may be retained and used in aggregated, anonymized form to operate, secure, and improve the services.
        model_training: Jina's published terms state customer request data, inputs, prompts, and uploaded content are not used to train its models.
        product_improvement: Aggregated and anonymized metadata can be used for service improvement; the terms exclude request content from model training but do not promise that all content is excluded from every support or product-development activity.
        human_or_operator_access: Access is governed by confidentiality, service delivery, security, support, and current Elastic data-processing controls; no operator-blind statement is published.
        subprocessors_and_routing: Jina AI is owned by Elastic, and current processing is governed by Elastic's DPA and privacy statement with their subprocessors.
        deletion_controls: Contract termination requires deletion or return of confidential information and data under Jina's published terms, subject to mandatory retention; current self-service per-request deletion is not documented.
        caveat: The acquisition creates a policy-transition risk, and neither legacy Jina terms nor linked Elastic terms publish a Search Foundation request-content retention duration.
      sources:
        - {type: official_terms, title: Jina AI Legal Information, url: https://jina.ai/legal/, updated_at: "2026-05-04"}
        - {type: official_privacy, title: Elastic Privacy Statement, url: https://www.elastic.co/legal/privacy-statement}
        - {type: official_dpa, title: Elastic Customer Data Processing Addendum, url: https://www.elastic.co/legal/customer-dpa}

    jina_ai_reader:
      agreements:
        terms_and_conditions: https://jina.ai/legal/
        privacy_policy: https://www.elastic.co/legal/privacy-statement
        data_processing_agreement: https://www.elastic.co/legal/customer-dpa
      data_governance:
        review_status: partial
        plan_scope: Jina Reader and anonymous utility endpoints after Elastic's October 2025 acquisition; legacy Jina statements and current Elastic processing terms both matter.
        prompt_retention: URLs, fetched pages, prompts, and transformed content are request data; the public legal documents do not state a Reader-specific retention period or whether anonymous requests receive different storage.
        response_retention: not_documented
        ordinary_logging: Operational, diagnostic, usage, IP, and security metadata may be retained and used in aggregated and anonymized form.
        model_training: Jina's terms say customer request data, inputs, prompts, and uploaded content are not used to train its models.
        product_improvement: Aggregated anonymized metadata can improve services; request-content use outside model training is not described with Reader-specific precision.
        human_or_operator_access: Content can be accessed as necessary for service delivery, security, support, and legal compliance under Elastic/Jina controls; no operator-blind promise is published.
        subprocessors_and_routing: Reader fetches third-party URLs and operates under Elastic's current DPA and subprocessor framework, creating both destination-site and service-side data flows.
        deletion_controls: No anonymous per-request deletion mechanism or content TTL was found; contractual deletion at termination does not directly help anonymous users.
        caveat: Anonymous access should not be mistaken for anonymous processing. IP and usage logs plus an undocumented content TTL make Reader unsuitable for confidential URLs or query material.
      sources:
        - {type: official_terms, title: Jina AI Legal Information, url: https://jina.ai/legal/, updated_at: "2026-05-04"}
        - {type: official_privacy, title: Elastic Privacy Statement, url: https://www.elastic.co/legal/privacy-statement}
        - {type: official_dpa, title: Elastic Customer Data Processing Addendum, url: https://www.elastic.co/legal/customer-dpa}

    mancer_ai:
      agreements:
        terms_of_use: https://mancer.tech/terms
        privacy_policy: https://mancer.tech/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Mancer AI hosted inference platform and its catalog of open and proprietary third-party models.
        prompt_retention: Customer Content may be used and retained to provide, maintain, develop, improve, secure, and enforce the service; no prompt-specific maximum retention period is published.
        response_retention: Outputs fall within Customer Content under the same unspecified retention and use policy.
        ordinary_logging: Account, request, security, fraud, usage, and service information is retained while the account exists and afterward as needed for claims, fairness records, and legal requirements.
        model_training: Mancer's standard terms allow it to train models with Customer Content and to provide content to third parties for their business purposes unless the user follows the platform's opt-out process.
        product_improvement: Customer Content may be used to develop and improve Mancer's services by default.
        human_or_operator_access: Retained content can be processed for development, improvement, security, abuse prevention, support, and legal purposes; no operator-blind commitment is offered.
        subprocessors_and_routing: Requests may use multiple open-source or proprietary models with their own licenses and restrictions; third-party content access is allowed by default unless opted out.
        deletion_controls: A platform opt-out is offered for future training and third-party business access. Account deletion does not guarantee immediate removal of all violation, claim, or legally required records.
        caveat: Training and third-party business use are opt-out, not opt-in, and the public documents provide no fixed inference-content retention period.
      sources:
        - {type: official_terms, title: Mancer AI Terms of Use, url: https://mancer.tech/terms, updated_at: "2024-06-06"}
        - {type: official_privacy, title: Mancer AI Privacy Policy, url: https://mancer.tech/privacy, updated_at: "2024-06-06"}

    novita_ai:
      agreements:
        terms_of_service: https://novita.ai/legal/terms-of-service
        privacy_policy: https://novita.ai/legal/privacy-policy
        acceptable_use_policy: https://novita.ai/legal/acceptable-use-policy
      data_governance:
        review_status: reviewed
        plan_scope: Novita AI marketplace inference APIs and web services; storage products, sandboxes, support, and legal exceptions have separate persistence.
        prompt_retention: Default inference Content is retained only for request processing and delivery. Exceptions permit retention where law requires or where necessary for service delivery or technical support.
        response_retention: Same default ZDR and support/legal exceptions as prompts.
        ordinary_logging: Account data persists for the account lifetime plus seven years; technical logs can persist two years, transaction records seven years, and communications three years without turning inference content into ordinary logs.
        model_training: Novita states it does not use Content to train its own models by default.
        product_improvement: The terms state Content is not used to improve services by default; de-identified personal and technical information may still be used indefinitely for research and statistics.
        human_or_operator_access: Content is not logged for human review under default ZDR, except where needed for service delivery, technical support, law, or security. Automated safety screening is always permitted.
        subprocessors_and_routing: Novita operates a model marketplace and uses service providers; underlying model licenses and optional storage or sandbox products can add separate rules.
        deletion_controls: Ordinary inference content is discarded after delivery. Account and stored-product data can be deleted subject to the lengthy legal, tax, support, and technical-log schedules.
        caveat: The ZDR promise applies to inference Content, not persistent storage products, paused sandboxes, support submissions, account records, or technical metadata.
      sources:
        - {type: official_terms, title: Novita AI Terms of Service, url: https://novita.ai/legal/terms-of-service, updated_at: "2026-08-05"}
        - {type: official_privacy, title: Novita AI Privacy Policy, url: https://novita.ai/legal/privacy-policy, updated_at: "2026-05-13"}
        - {type: official_aup, title: Novita AI Acceptable Use Policy, url: https://novita.ai/legal/acceptable-use-policy}

    hyperbolic:
      agreements:
        terms_of_use: https://www.hyperbolic.ai/terms
        privacy_policy: https://www.hyperbolic.ai/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Hyperbolic hosted AI inference and gateway; GPU marketplace suppliers and third-party AI tools can create independently governed processing.
        prompt_retention: Hyperbolic's AI Privacy FAQ says inference input is held only temporarily and discarded after the task. Feedback ratings cause the related conversation to be stored.
        response_retention: Outputs are not retained after ordinary inference according to the FAQ; rated conversations and other deliberately saved content are exceptions.
        ordinary_logging: Account, device, IP, usage, session replay, security, and marketplace data is logged under purpose-based retention, separate from inference payloads.
        model_training: The terms and FAQ state Hyperbolic does not use inputs or outputs to train generative AI models.
        product_improvement: The general Content license permits use for operating and improving services, while the narrower AI FAQ says inference data is used only for the request. Feedback and de-identified operational data can improve the service.
        human_or_operator_access: The terms reserve monitoring, review, and investigation rights and say users have no expectation of privacy in transmissions; the FAQ says ordinary inference data is not stored. Feedback and investigations are access exceptions.
        subprocessors_and_routing: Inference nodes receive input, and third-party AI tools or marketplace suppliers may apply their own retention and training terms.
        deletion_controls: Ordinary inference content is automatically discarded. Users can delete certain account content, while public or shared content may not be comprehensively removable.
        caveat: The broad monitoring and improvement language in the binding terms is less protective than the product FAQ. Third-party AI tools are expressly outside Hyperbolic's no-training commitment.
      sources:
        - {type: official_terms, title: Hyperbolic Terms of Use, url: https://www.hyperbolic.ai/terms, updated_at: "2025-03-24"}
        - {type: official_privacy, title: Hyperbolic Privacy Policy, url: https://www.hyperbolic.ai/privacy, updated_at: "2024-06-03"}
        - {type: official_docs, title: Hyperbolic AI Privacy FAQ, url: https://www.hyperbolic.ai/privacy/faq}

    upstage:
      agreements:
        terms_of_service: https://www.upstage.ai/terms-of-service/update-may-06-2026
        privacy_policy: https://www.upstage.ai/privacy-policy/updated-aug-26-2026
      data_governance:
        review_status: reviewed
        plan_scope: Upstage free API tier under the May 2026 terms and August 2026 privacy policy; paid, promotional, async, console, and third-party models differ.
        prompt_retention: Free-service request and response data may be stored through the end of service provision or another separately notified period for delivery, improvement, and AI research. The general paid API rule is no storage except operationally necessary state.
        response_retention: Same free-tier policy; asynchronous results are stored for 30 days, while request data remains until completion.
        ordinary_logging: Service usage records, access logs, and IP addresses are retained for three months; account, transaction, and support records have longer statutory periods.
        model_training: Current terms expressly allow free-service input and output data to be used for AI research and development, including training. Paid API data is not trained on without separate consent.
        product_improvement: Free-tier input and output data may be used for service quality improvement and R&D. Paid API content requires separate consent for logging or improvement.
        human_or_operator_access: Stored free-tier data can be accessed for service delivery, improvement, research, safety, and legal compliance under Upstage and subprocessor controls.
        subprocessors_and_routing: Some services use third-party AI providers whose policies govern content. Current privacy disclosures list Azure, OpenAI, Fireworks, and other overseas subprocessors with feature-specific retention.
        deletion_controls: Free data is destroyed when its stated retention or service-provision period ends; membership withdrawal deletes many account resources, subject to legal records. No free-tier immediate content-deletion control is documented.
        caveat: Older Upstage API terms said API data was not stored or trained on, but the current May 2026 terms create an explicit free-service exception. Do not apply the older no-training summary to today's free tier.
      sources:
        - {type: official_terms, title: Upstage Terms of Use, url: https://www.upstage.ai/terms-of-service/update-may-06-2026, updated_at: "2026-05-06"}
        - {type: official_privacy, title: Upstage Privacy Policy, url: https://www.upstage.ai/privacy-policy/updated-aug-26-2026, updated_at: "2026-08-19"}

    wavespeedai:
      agreements:
        terms_of_service: https://wavespeed.ai/static/terms
        privacy_policy: https://wavespeed.ai/static/privacy
      data_governance:
        review_status: partial
        plan_scope: WaveSpeedAI image, video, and other hosted model APIs and web interface; public and third-party models can have separate terms.
        prompt_retention: The terms permit WaveSpeedAI to store and process Customer Data as necessary to provide outputs and associated services, but no inference-specific maximum retention period is published.
        response_retention: Outputs are Customer Data under the same undefined retention; customers are responsible for keeping copies.
        ordinary_logging: Service-performance, use, account, security, and transaction information is collected, and aggregated anonymized Resultant Data can be retained and used after the service term.
        model_training: WaveSpeedAI's product materials state it does not use customer data for training, while the terms reserve service modifications and third-party-model rules. No selected-model training matrix was found.
        product_improvement: Aggregated anonymized Resultant Data can be used to improve services, development, diagnostics, and corrections; the terms do not authorize identifiable content training.
        human_or_operator_access: Content can be processed by WaveSpeedAI and necessary service providers for delivery, support, security, and legal compliance; no zero-operator-access commitment is published.
        subprocessors_and_routing: Open and third-party models are available and their licenses and privacy practices can apply; WaveSpeedAI also uses infrastructure, analytics, payment, support, and model-processing providers.
        deletion_controls: not_documented
        caveat: A no-training statement is available, but the public agreements do not state a prompt/output TTL or self-service deletion mechanism, so retention remains unresolved.
      sources:
        - {type: official_terms, title: WaveSpeedAI Terms of Service, url: https://wavespeed.ai/static/terms, updated_at: "2026-08-06"}
        - {type: official_privacy, title: WaveSpeedAI Privacy Policy, url: https://wavespeed.ai/static/privacy}
        - {type: official_product, title: WaveSpeedAI data-use statement, url: https://wavespeed.ai/landing/introduce}

    huggingface_inference_providers:
      agreements:
        terms_of_service: https://huggingface.co/terms-of-service
        privacy_policy: https://huggingface.co/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Hugging Face Inference Providers routing layer; each selected inference provider has separate data-security terms.
        prompt_retention: Hugging Face states it does not store request bodies or responses when routing requests. Debugging logs are kept up to 30 days without user data or tokens.
        response_retention: Hugging Face states routed responses are not stored.
        ordinary_logging: Debugging logs are retained up to 30 days but are described as excluding user data and tokens.
        model_training: Hugging Face states it does not store routed user data for training purposes; the selected upstream provider's separate policy still applies.
        product_improvement: No routed prompt or response use for Hugging Face training is disclosed; upstream provider policies must be reviewed separately.
        human_or_operator_access: Hugging Face says request and response bodies are not stored by its routing layer; upstream provider access remains provider-specific.
        subprocessors_and_routing: The service is a proxy that sends requests to the selected external inference provider, which is independently responsible for its security and data handling.
        deletion_controls: No routed content is said to be stored by Hugging Face; account personal-data rights are governed by the Hugging Face privacy policy.
        caveat: The Hugging Face proxy policy does not replace the policy of the final inference provider. Automatic routing can change which upstream processes a request.
      sources:
        - {type: official_terms, title: Terms of Service, url: https://huggingface.co/terms-of-service}
        - {type: official_privacy, title: Privacy Policy, url: https://huggingface.co/privacy}
        - {type: official_docs, title: Inference Providers Security and Compliance, url: https://huggingface.co/docs/inference-providers/security}

    vercel_ai_gateway:
      agreements:
        terms_of_service: https://vercel.com/legal/terms
        privacy_policy: https://vercel.com/legal/privacy-policy
        ai_product_terms: https://vercel.com/legal/ai-product-terms
        data_processing_addendum: https://vercel.com/legal/dpa
      data_governance:
        review_status: reviewed
        plan_scope: Vercel AI Gateway; provider-side retention and training depend on routing controls, provider agreements, plan, and BYOK configuration.
        prompt_retention: Vercel states the gateway layer deletes prompts, outputs, and sensitive data after a request completes. Provider-side retention is separate unless zero-data-retention routing is enabled.
        response_retention: Same gateway-layer deletion policy as prompts; final providers may apply separate retention.
        ordinary_logging: Usage, spend, request volume, model, provider, project, token counts, and performance metadata are available for observability; prompt-content logging is not described as part of ordinary gateway telemetry.
        model_training: A no-prompt-training routing control is available on all plans. Enterprise terms separately warrant that AI Gateway Customer Content is not used to train or improve Vercel AI products; other plans must enforce provider controls.
        product_improvement: Enterprise AI Gateway content receives an express no-training and no-improvement commitment. Non-enterprise handling depends on AI Product Terms and configured provider-routing controls.
        human_or_operator_access: The gateway claims immediate content deletion after request completion; the selected provider still processes content under its own terms.
        subprocessors_and_routing: Requests go to selected third-party AI providers. Team or request-level ZDR routes only to providers covered by negotiated ZDR agreements; BYOK agreements remain the customer's responsibility.
        deletion_controls: Gateway content is described as deleted after completion. Provider-side ZDR is available per request on Pro and Enterprise or team-wide as a paid control; no-training filtering is available on all plans.
        caveat: Enabling ZDR or no-training is not the same as the default behavior of every provider route. BYOK credentials are skipped under ZDR unless the customer marks its own contract compliant.
      sources:
        - {type: official_terms, title: Terms of Service, url: https://vercel.com/legal/terms}
        - {type: official_privacy, title: Privacy Notice, url: https://vercel.com/legal/privacy-policy}
        - {type: official_terms, title: AI Product Terms, url: https://vercel.com/legal/ai-product-terms}
        - {type: official_dpa, title: Data Processing Addendum, url: https://vercel.com/legal/dpa}
        - {type: official_docs, title: AI Gateway security and data routing, url: https://vercel.com/i/secure-ai-gateway}
        - {type: official_changelog, title: Zero Data Retention and no-prompt-training controls, url: https://vercel.com/changelog/zero-data-retention-no-prompt-training-on-ai-gateway, published_at: "2026-04-06"}

    opencode_zen:
      agreements:
        terms_of_service: https://opencode.ai/legal/terms-of-service
        privacy_policy: https://opencode.ai/legal/privacy-policy
      data_governance:
        review_status: reviewed
        plan_scope: OpenCode Zen gateway; data handling varies by selected model and upstream provider.
        prompt_retention: Most listed providers are described as zero retention, but OpenAI and Anthropic routes retain requests for 30 days and named free routes have separate collection terms.
        response_retention: Follows the selected upstream route's policy.
        ordinary_logging: OpenCode's public Zen documentation does not provide one universal metadata-retention period for every route.
        model_training: Most routes are described as no-training, with explicit exceptions for free models whose prompts and completions may be collected for model improvement or training.
        product_improvement: Named free routes may exchange discounted or free access for improvement use; NVIDIA trial routes log data for security and product or service improvement.
        human_or_operator_access: Depends on the selected upstream provider and route; the gateway documentation does not establish universal operator-blind processing.
        subprocessors_and_routing: Requests are routed to model providers. Their data policies and any model-specific trial terms apply in addition to OpenCode's terms.
        deletion_controls: Not documented as a single gateway-wide control for upstream retention; users can disable data-collecting models in team workspaces.
        caveat: Privacy is model-specific and can change with the live catalog. Free routes are among the explicit exceptions to the default zero-retention and no-training description.
      sources:
        - {type: official_terms, title: Terms of Service, url: https://opencode.ai/legal/terms-of-service}
        - {type: official_privacy, title: Privacy Policy, url: https://opencode.ai/legal/privacy-policy}
        - {type: official_docs, title: OpenCode Zen privacy disclosures, url: https://opencode.ai/docs/zen/}

    darkbloom:
      agreements:
        terms_of_service: https://www.darkbloom.dev/terms.html
        privacy_policy: https://www.darkbloom.dev/privacy.html
      data_governance:
        review_status: reviewed
        plan_scope: Darkbloom public-alpha consumer and provider services.
        prompt_retention: Content may be retained for shorter operational periods and longer for support, abuse review, legal compliance, or disputes; no fixed duration is published.
        response_retention: Covered by the same Content retention language as prompts.
        ordinary_logging: Coordinator is designed not to log prompt content in ordinary request logs, but logs operational metadata.
        model_training: Current policy does not grant Darkbloom the right to use Content for general-purpose model training; it says terms would be updated before any future change where law requires.
        product_improvement: Aggregated or de-identified information may be created for analytics, security, reporting, and service improvement.
        human_or_operator_access: The coordinator currently processes request payloads in plaintext transiently; relevant Content is disclosed to the selected independently operated provider under technical and contractual controls.
        subprocessors_and_routing: Third-party routers may terminate TLS and have technical access before requests reach Darkbloom; their own policies apply.
        deletion_controls: Privacy requests are available, subject to identity verification and legal, billing, security, compliance, and dispute exceptions.
        conflict: Product marketing describes encrypted requests hidden from operators, while current terms and privacy disclosures say universal end-to-end encryption is not yet present and coordinator plaintext access is technically possible.
      sources:
        - {type: official_product, title: Darkbloom, url: https://www.darkbloom.dev/}
        - {type: official_terms, title: Terms of Service, url: https://www.darkbloom.dev/terms.html, updated_at: "2026-06-10"}
        - {type: official_privacy, title: Privacy Policy, url: https://www.darkbloom.dev/privacy.html, updated_at: "2026-06-10"}

    zai:
      agreements:
        terms_of_service: https://docs.z.ai/legal-agreement/terms-of-use
        privacy_policy: https://docs.z.ai/legal-agreement/privacy-policy
      data_governance:
        review_status: reviewed
        plan_scope: Z.AI developer API; the consumer-service terms differ materially.
        prompt_retention: The API DPA says prompt and generated content is processed in real time and not stored on Z.AI servers.
        response_retention: The API DPA gives generated API content the same no-storage treatment as input.
        ordinary_logging: Performance and usage metrics such as model version, inference, timing, diagnostics, and technical data may be collected and used to improve the service.
        model_training: API End User Content is not used to develop or improve services unless the API customer explicitly agrees.
        product_improvement: Technical usage metrics may be used for improvement; content use requires explicit agreement for API customers.
        human_or_operator_access: Content is processed to deliver the service and may be handled by approved subprocessors; no zero-operator-access promise is published.
        subprocessors_and_routing: The DPA permits subcontractors and generally locates processing in Singapore; third-party model terms apply when selected.
        deletion_controls: Content is not stored under the API DPA; other customer data is deleted after termination unless law requires retention.
        caveat: Do not apply the consumer Z.AI policy—which permits broad content improvement use—to the developer API; the API Additional Terms and DPA control API content.
      sources:
        - {type: official_terms, title: Z.AI Terms and Additional Terms for API Services, url: https://docs.z.ai/legal-agreement/terms-of-use, updated_at: "2026-04-14"}
        - {type: official_privacy, title: Z.AI Privacy Policy and API DPA, url: https://docs.z.ai/legal-agreement/privacy-policy, updated_at: "2025-09-29"}

    llm7:
      data_governance:
        review_status: not_found
        caveat: The public site and developer documentation were reviewed, but no usable terms, privacy policy, prompt-retention period, training rule, deletion control, or upstream-processing agreement was found. Treat all content handling as unknown.
      sources:
        - {type: official_product, title: LLM7, url: https://llm7.io/}
        - {type: official_docs, title: LLM7 models documentation, url: https://docs.llm7.io/guides/models}

    modelscope_inference:
      agreements:
        terms_of_service: https://modelscope.cn/terms
        privacy_policy: https://modelscope.cn/privacy
      data_governance:
        review_status: partial
        plan_scope: ModelScope account and hosted API-Inference service; individual model licenses can add restrictions.
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: The general platform privacy policy covers account, device, and usage information, but no API-content logging duration was found.
        model_training: No service-specific commitment excluding API prompts or outputs from training was found in the reviewed public documents.
        product_improvement: General platform terms permit service operation and improvement uses; API-content scope is not clearly separated.
        human_or_operator_access: not_documented
        subprocessors_and_routing: ModelScope and infrastructure providers process requests; model-specific licenses and publishers may create additional trust boundaries.
        deletion_controls: General privacy requests exist, but no API prompt/output deletion control or TTL was found.
        caveat: General Chinese-language platform policies exist, but the hosted inference documentation does not publish a clear prompt-retention or training matrix.
      sources:
        - {type: official_terms, title: ModelScope Terms, url: https://modelscope.cn/terms}
        - {type: official_privacy, title: ModelScope Privacy Policy, url: https://modelscope.cn/privacy}
        - {type: official_docs, title: API-Inference introduction, url: https://modelscope.cn/docs/model-service/API-Inference/intro}

    ndif:
      data_governance:
        review_status: not_found
        caveat: NDIF documentation and project pages explain research access and instrumentation, but no public service terms or privacy policy defining prompt retention, research use, operator access, or deletion was found. Research workloads should be assumed observable unless a project agreement says otherwise.
      sources:
        - {type: official_product, title: National Deep Inference Fabric, url: https://ndif.us/}
        - {type: official_docs, title: NDIF get started, url: https://ndif.us/get-started/}

    pollinations:
      agreements:
        terms_of_service: https://pollinations.ai/terms
        privacy_policy: https://pollinations.ai/privacy
      data_governance:
        review_status: partial
        plan_scope: Pollinations public APIs and community-operated open-source services.
        prompt_retention: No fixed prompt-retention duration was found in the rendered public legal pages or API documentation.
        response_retention: not_documented
        ordinary_logging: Operational and abuse-prevention logging is not described with content/metadata separation or a fixed TTL.
        model_training: No current, service-wide no-training commitment was found for free API inputs or outputs.
        product_improvement: not_documented
        human_or_operator_access: The public service is operated through Pollinations infrastructure and integrations; no zero-operator-access commitment is published.
        subprocessors_and_routing: Requests can be served by multiple models and infrastructure paths documented in the open-source project.
        deletion_controls: not_documented
        caveat: Legal URLs exist, but the public pages did not yield a sufficiently specific API data-handling agreement. Do not submit confidential data based on the project's open-source status.
      sources:
        - {type: official_terms, title: Pollinations Terms, url: https://pollinations.ai/terms}
        - {type: official_privacy, title: Pollinations Privacy, url: https://pollinations.ai/privacy}
        - {type: official_repository, title: Pollinations repository and API documentation, url: https://github.com/pollinations/pollinations}

    puter_js:
      agreements:
        terms_of_service: https://puter.com/terms
        privacy_policy: https://puter.com/privacy
      data_governance:
        review_status: partial
        plan_scope: Puter platform and Puter.js user-pays AI integrations; selected upstream model providers also apply.
        prompt_retention: Puter's general terms permit platform processing, scanning, monitoring, and deletion of User Data, but do not give an AI-request TTL.
        response_retention: not_documented
        ordinary_logging: Account, activity, device, cookie, and usage data is collected; AI content logging is not isolated in the public policy.
        model_training: No universal no-training commitment was found for Puter.js AI requests or all selected upstream providers.
        product_improvement: Service providers may assist Puter with operation and improvement; content scope remains unclear.
        human_or_operator_access: Terms permit Puter to monitor and review User Data, including private messages, while upstream AI providers necessarily receive routed content.
        subprocessors_and_routing: Puter.js routes to third-party AI providers and charges the end user's account; both Puter and the chosen upstream are relevant processors.
        deletion_controls: Users can delete accounts and stored User Data, subject to the terms; no per-inference upstream deletion control is documented.
        caveat: The privacy policy says Puter does not collect personal information contained in User Data, while the terms permit operational processing and review. Neither document establishes a model-route-specific privacy guarantee.
      sources:
        - {type: official_terms, title: Puter Terms of Service, url: https://puter.com/terms, updated_at: "2023-02-02"}
        - {type: official_privacy, title: Puter Privacy Policy, url: https://puter.com/privacy, updated_at: "2022-07-18"}
        - {type: official_docs, title: Puter user-pays model, url: https://docs.puter.com/user-pays-model/}

    public_ai:
      data_governance:
        review_status: not_found
        caveat: The platform documentation, plans, and model catalog were reviewed, but no public terms or privacy policy specifying inference-content retention, training, operator access, deletion, or provider routing was found.
      sources:
        - {type: official_docs, title: Public AI platform documentation, url: https://platform.publicai.co/docs}
        - {type: official_product, title: Public AI plans, url: https://platform.publicai.co/plans}

    lightning_ai_model_apis:
      agreements:
        terms_of_service: https://lightning.ai/terms-of-service
        privacy_policy: https://lightning.ai/privacy-policy
      data_governance:
        review_status: partial
        plan_scope: Lightning AI platform and hosted Model APIs; model publishers and deployed applications can add separate terms.
        prompt_retention: No Model-API-specific content retention period was found in the public legal or product documentation.
        response_retention: not_documented
        ordinary_logging: Platform account, use, diagnostic, and security information may be collected; content logging is not clearly separated.
        model_training: No plan-specific commitment excluding free Model API inputs and outputs from training was found.
        product_improvement: General platform information may be used to operate and improve services; the treatment of inference content is unclear.
        human_or_operator_access: Lightning and the publisher/operator of a model endpoint may technically process content; no universal operator-access limitation is published.
        subprocessors_and_routing: Requests run on Lightning infrastructure and can target third-party-published models or apps, creating a separate publisher trust boundary.
        deletion_controls: General account/privacy rights exist, but no per-request content deletion or TTL was found.
        caveat: Security and general legal policies do not establish a uniform privacy contract for every community Model API.
      sources:
        - {type: official_terms, title: Lightning AI Terms of Service, url: https://lightning.ai/terms-of-service}
        - {type: official_privacy, title: Lightning AI Privacy Policy, url: https://lightning.ai/privacy-policy}
        - {type: official_docs, title: Lightning Model APIs, url: https://lightning.ai/docs/overview/model-apis}

    ovhcloud_ai_endpoints:
      agreements:
        terms_of_service: https://www.ovhcloud.com/en/terms-and-conditions/
        privacy_policy: https://www.ovhcloud.com/en/personal-data-protection/
      data_governance:
        review_status: partial
        plan_scope: OVHcloud AI Endpoints under OVHcloud general and service-specific cloud agreements.
        prompt_retention: No public endpoint-specific prompt retention duration was verified.
        response_retention: not_documented
        ordinary_logging: OVHcloud processes technical, security, billing, and account data; the reviewed catalog does not specify inference-content logging.
        model_training: No endpoint-specific statement authorizing or prohibiting training on prompts and outputs was found.
        product_improvement: General service telemetry may be used for operation and improvement; content scope is not documented.
        human_or_operator_access: not_documented
        subprocessors_and_routing: OVHcloud operates the endpoints in its cloud regions; catalog models may carry separate licenses.
        deletion_controls: General GDPR rights exist; no inference-content deletion control or TTL was found.
        caveat: European cloud/privacy commitments are not equivalent to a documented zero-retention or no-training promise for AI Endpoints.
      sources:
        - {type: official_terms, title: OVHcloud Terms and Conditions, url: https://www.ovhcloud.com/en/terms-and-conditions/}
        - {type: official_privacy, title: OVHcloud personal data protection, url: https://www.ovhcloud.com/en/personal-data-protection/}
        - {type: official_product, title: OVHcloud AI Endpoints catalog, url: https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/}

    inception_platform:
      data_governance:
        review_status: not_found
        caveat: Inception's public platform and developer documentation were reviewed, but no accessible governing terms or privacy disclosure specifying API-content retention, training, human access, or deletion was found.
      sources:
        - {type: official_docs, title: Inception API getting started, url: https://docs.inceptionlabs.ai/get-started/get-started}
        - {type: official_product, title: Inception Platform, url: https://platform.inceptionlabs.ai/}

    poolside_direct_api:
      data_governance:
        review_status: not_found
        caveat: Poolside's model and developer pages were reviewed, but no public API agreement or privacy policy with prompt retention, training, human access, routing, or deletion terms was found for the direct preview API.
      sources:
        - {type: official_product, title: Poolside models, url: https://poolside.ai/models}
        - {type: official_docs, title: Poolside supported models, url: https://docs.poolside.ai/get-started/supported-models}

    ai21_studio:
      agreements:
        terms_of_service: https://www.ai21.com/terms-policies/terms-of-use/
        privacy_policy: https://www.ai21.com/terms-policies/privacy-policy/
        service_specific_terms: https://lp.ai21.com/hubfs/resources/AI21-Models-Terms-of-Service.pdf
      data_governance:
        review_status: partial
        plan_scope: AI21 Studio and Models API; negotiated Traceless Operations has stronger handling than default access.
        prompt_retention: The model terms say the service is not intended as storage, but no default public retention maximum was verified. Traceless Operations can disable content retention when explicitly configured.
        response_retention: Same unresolved default; Traceless Operations applies only when contracted/configured.
        ordinary_logging: Usage and operational metadata may be retained; the public terms do not publish a complete default content-log TTL.
        model_training: No sufficiently clear plan-specific default training statement was found in the reviewed public model terms.
        product_improvement: The agreement permits service operation and improvement uses subject to customer terms; exact content scope remains unclear.
        human_or_operator_access: Support, security, and legal workflows may permit access; Traceless Operations is the documented stronger control.
        subprocessors_and_routing: AI21 and its cloud/service subprocessors process requests.
        deletion_controls: Traceless Operations supports a retain=false mode; default account deletion does not establish immediate request-content deletion.
        caveat: Do not infer Traceless Operations for free Studio access; it is a separately described negotiated/configured mode.
      sources:
        - {type: official_terms, title: AI21 Terms of Use, url: https://www.ai21.com/terms-policies/terms-of-use/}
        - {type: official_privacy, title: AI21 Privacy Policy, url: https://www.ai21.com/terms-policies/privacy-policy/}
        - {type: official_terms, title: AI21 Models Terms of Service, url: https://lp.ai21.com/hubfs/resources/AI21-Models-Terms-of-Service.pdf}

    mara_inference_cloud:
      data_governance:
        review_status: not_found
        caveat: MARA's cloud plan, landing, and live model-catalog surfaces were reviewed, but no public terms or privacy policy defining inference-content retention, training, access, routing, or deletion was found.
      sources:
        - {type: official_product, title: MARA Inference Cloud plans, url: https://cloud.mara.com/plans}
        - {type: live_catalog, title: MARA model catalog, url: https://api.cloud.mara.com/v1/models}

    waterfall:
      data_governance:
        review_status: not_found
        caveat: Waterfall's product, pricing, documentation, and trust surfaces were reviewed, but no usable public agreement specifying prompt/output retention, training, operator access, upstream providers, or deletion was found.
      sources:
        - {type: official_product, title: Waterfall trust information, url: https://www.getwaterfall.org/trust/}
        - {type: official_docs, title: Waterfall documentation, url: https://www.getwaterfall.org/docs/}

    logfare:
      agreements:
        terms_of_service: https://logfare.ai/tos
        privacy_policy: https://logfare.ai/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Logfare standard and premium free API routes under policy version 3.1.
        prompt_retention: Every request body is logged after best-effort PII scrubbing. The privacy policy defines retention by store and warns that scrubbing is imperfect.
        response_retention: Every response body is logged after the same best-effort scrub and can enter internal evaluation datasets.
        ordinary_logging: IP addresses, forwarded-for values, User-Agent, headers, timestamps, token counts, model, prompts, responses, and metadata are collected; network/client identifiers are retained up to 90 days.
        model_training: Standard-tier content is not used for training by default. Premium access requires voluntary, reversible opt-in; already incorporated training data cannot be removed from a trained model.
        product_improvement: Post-scrub content may be used in private internal evaluation and benchmarking datasets on a legitimate-interest basis, including standard-tier requests.
        human_or_operator_access: Authorized Logfare personnel can access protected internal datasets; upstream providers receive request content to perform inference.
        subprocessors_and_routing: Logfare proxies to third-party LLM providers. The policy says Logfare does not sell, license, publish, or distribute its underlying user-content datasets.
        deletion_controls: Users can object to evaluation use and withdraw future training consent; model unlearning is not offered for content already trained into a model.
        caveat: “Free” standard access explicitly funds private evaluation data collection. Premium routes exchange access for opt-in training use.
      sources:
        - {type: official_terms, title: Logfare Terms of Service, url: https://logfare.ai/tos, updated_at: "2026-06-01"}
        - {type: official_privacy, title: Logfare Privacy Policy, url: https://logfare.ai/privacy, updated_at: "2026-06-01"}

    bazaarlink:
      agreements:
        terms_of_service: https://bazaarlink.ai/terms
        privacy_policy: https://bazaarlink.ai/privacy
      data_governance:
        review_status: partial
        plan_scope: BazaarLink gateway and free API routes.
        prompt_retention: The public agreements were located, but no clear API prompt-content TTL was verified.
        response_retention: not_documented
        ordinary_logging: Account, usage, security, and service data may be collected; content/metadata separation is not clearly documented.
        model_training: No route-specific no-training guarantee was found for all free models.
        product_improvement: General improvement uses are permitted without a precise API-content scope.
        human_or_operator_access: not_documented
        subprocessors_and_routing: The gateway routes requests to model providers; upstream terms can apply independently.
        deletion_controls: General privacy requests exist, but no per-request or upstream deletion guarantee was found.
        caveat: A legal policy exists, but it does not resolve the central inference-data questions for the free gateway.
      sources:
        - {type: official_terms, title: BazaarLink Terms, url: https://bazaarlink.ai/terms}
        - {type: official_privacy, title: BazaarLink Privacy, url: https://bazaarlink.ai/privacy}
        - {type: official_security, title: BazaarLink Security, url: https://bazaarlink.ai/security}

    dreamprompting:
      agreements:
        terms_of_service: https://dreamprompting.com/terms
        privacy_policy: https://dreamprompting.com/privacy
      data_governance:
        review_status: partial
        plan_scope: DreamPrompting website, public prompt-sharing platform, and free multi-provider API gateway.
        prompt_retention: The privacy policy describes submitted prompts as stored and, for sharing features, publicly visible; it does not state a separate API-gateway prompt TTL.
        response_retention: not_documented
        ordinary_logging: IP, device, page, account, and usage information is collected; API content logging is not specifically disclosed.
        model_training: No service-wide no-training commitment was found for API inputs or outputs.
        product_improvement: Collected information and user content may be used to provide and improve the service.
        human_or_operator_access: Stored/shared prompt content is accessible to the service and can be public; gateway upstreams process API requests.
        subprocessors_and_routing: The gateway advertises routing across multiple providers with automatic failover, so the selected upstream's policy also applies.
        deletion_controls: Users can remove submitted public prompts and request account deletion; no upstream API-request deletion control is documented.
        caveat: The legal pages focus heavily on the public prompt-sharing product and do not cleanly distinguish gateway traffic. Do not assume public-prompt deletion covers routed API content.
      sources:
        - {type: official_terms, title: DreamPrompting Terms of Service, url: https://dreamprompting.com/terms, updated_at: "2025-12-21"}
        - {type: official_privacy, title: DreamPrompting Privacy Policy, url: https://dreamprompting.com/privacy}
        - {type: official_docs, title: DreamPrompting API documentation, url: https://dreamprompting.com/api-docs}

    ch_at:
      agreements:
        privacy_policy: https://ch.at/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Anonymous ch.at web, curl, DNS, SSH, and API access.
        prompt_retention: The service states “No logs”; no prompt history or account is provided.
        response_retention: The same no-logs claim applies to responses in ordinary operation.
        ordinary_logging: ch.at explicitly markets no logs and no accounts, though no separately negotiated audit or DPA was found.
        model_training: No training use is disclosed; the no-logs architecture would preclude retained-content training by ch.at in ordinary operation.
        product_improvement: No content-based improvement use is disclosed.
        human_or_operator_access: Requests are necessarily processed in plaintext by the service during inference; no zero-operator-access claim is made.
        subprocessors_and_routing: The public repository and deployed service should be checked for current model/upstream configuration; upstream handling is not fully documented in the privacy page.
        deletion_controls: No stored content is claimed, so no content-deletion control is offered; there is no user account.
        caveat: “No logs” is a first-party operational claim, not an independently audited contractual SLA, and upstream model handling remains unclear.
      sources:
        - {type: official_privacy, title: ch.at privacy response, url: https://ch.at/privacy}
        - {type: official_repository, title: ch.at source repository, url: https://github.com/Deep-ai-inc/ch.at}

    opentyphoon:
      data_governance:
        review_status: not_found
        caveat: OpenTyphoon's official documentation, FAQ, models, and API reference were reviewed, but no public service terms or privacy policy defining request retention, training, operator access, or deletion was found.
      sources:
        - {type: official_docs, title: OpenTyphoon documentation, url: https://docs.opentyphoon.ai/en/}
        - {type: official_docs, title: OpenTyphoon FAQ, url: https://docs.opentyphoon.ai/en/faq/}

    alcf_inference_endpoints:
      data_governance:
        review_status: partial
        plan_scope: Argonne Leadership Computing Facility research allocations and inference endpoints; project and Department of Energy policies apply.
        prompt_retention: No endpoint-wide public prompt-retention TTL was found.
        response_retention: not_documented
        ordinary_logging: Facility authentication, allocation, security, and operational telemetry may be logged; content logging is not documented in the endpoint guide.
        model_training: No public commitment excludes research workload content from secondary analysis or model training across all projects.
        product_improvement: Facility telemetry may support operations and research; content scope is not specified.
        human_or_operator_access: Facility administrators and project personnel may have privileged technical access under institutional controls.
        subprocessors_and_routing: Workloads run on Argonne/ALCF systems and are subject to allocation and project governance rather than consumer SaaS terms.
        deletion_controls: Project storage and account controls apply; no inference-request deletion workflow is documented.
        caveat: Institutional access approval is not a privacy guarantee. Users should rely on their allocation agreement and data-management plan before sending controlled data.
      sources:
        - {type: official_docs, title: ALCF inference endpoints, url: https://docs.alcf.anl.gov/services/inference-endpoints/}
        - {type: official_docs, title: ALCF allocation management, url: https://docs.alcf.anl.gov/account-project-management/allocation-management/}

    fikra_api:
      agreements:
        terms_of_service: https://fikraapi.co.ke/terms
        privacy_policy: https://fikraapi.co.ke/privacy
      data_governance:
        review_status: reviewed
        plan_scope: Fikra API operated by Lacesse Ventures; Groq is disclosed as inference hardware/provider.
        prompt_retention: The privacy policy states prompts are processed in memory and immediately discarded under a strict zero-data-retention policy.
        response_retention: Responses are also stated not to be stored or logged and are immediately discarded.
        ordinary_logging: Request metadata—timestamps, token counts, and model—is logged for billing; prompts and responses are excluded.
        model_training: Fikra states it does not train on prompts or responses.
        product_improvement: No content-based improvement use is disclosed.
        human_or_operator_access: No ordinary stored-content access is available under the stated design; transient processing by Fikra and Groq remains necessary.
        subprocessors_and_routing: Groq is named for inference hardware; Render, HostAfrica, Cloudflare, Supabase, and payment providers are also disclosed.
        deletion_controls: Content is claimed to be discarded immediately; account/privacy requests remain available for other personal data.
        caveat: This is a first-party contractual/operational claim and depends on Fikra's disclosed upstream configuration remaining current.
      sources:
        - {type: official_terms, title: Fikra API Terms, url: https://fikraapi.co.ke/terms}
        - {type: official_privacy, title: Fikra API Privacy Policy, url: https://fikraapi.co.ke/privacy}

    sarvam_ai:
      agreements:
        terms_of_service: https://www.sarvam.ai/terms-of-service
        privacy_policy: https://www.sarvam.ai/privacy-policy
      data_governance:
        review_status: partial
        plan_scope: Sarvam developer APIs; enterprise customer content can be governed by separate customer agreements.
        prompt_retention: The public privacy policy says the products collect user inputs, file uploads, and generated outputs, but does not publish a developer-API content TTL.
        response_retention: Outputs are included in collected product information without a fixed public retention period.
        ordinary_logging: Standard usage logs, IP/device data, traffic, cookies, diagnostics, and product use are collected.
        model_training: No clear default developer-API commitment excluding inputs and outputs from model training was found; enterprise agreements may differ.
        product_improvement: The policy permits data use to enhance user experience and improve services; beta/trial terms expressly allow diagnostic, performance, and usage analysis.
        human_or_operator_access: Sarvam and service providers may process collected content for delivery, support, safety, and legal purposes.
        subprocessors_and_routing: Service providers and infrastructure vendors may receive necessary data; enterprise processing is governed separately.
        deletion_controls: Privacy rights and account deletion requests exist, subject to legal and operational exceptions; no per-request API deletion control is published.
        caveat: Sarvam's public privacy policy excludes enterprise-customer content from its scope, but free developer access should not be assumed to receive enterprise protections.
      sources:
        - {type: official_terms, title: Sarvam Terms of Service, url: https://www.sarvam.ai/terms-of-service}
        - {type: official_privacy, title: Sarvam Privacy Policy, url: https://www.sarvam.ai/privacy-policy}
        - {type: official_security, title: Sarvam Trust Center, url: https://www.sarvam.ai/trust-center}

    byteplus_modelark:
      agreements:
        terms_of_service: https://www.byteplus.com/en/legal/terms-of-service
        privacy_policy: https://www.byteplus.com/en/legal/privacy-policy
        promotional_terms: https://docs.byteplus.com/en/docs/legal/termsandconditions_modelark_free-token_campaign
      data_governance:
        review_status: partial
        plan_scope: BytePlus ModelArk and its free-token campaign; product-specific, regional, and model-provider terms may add conditions.
        prompt_retention: No ModelArk-free-tier prompt TTL was verified in the campaign or general legal terms.
        response_retention: not_documented
        ordinary_logging: BytePlus collects service, account, device, security, and usage information; the legal pages do not clearly isolate inference content.
        model_training: No campaign-specific no-training commitment was found for ModelArk prompts and outputs.
        product_improvement: General service improvement uses are permitted; content scope remains unresolved.
        human_or_operator_access: BytePlus and service providers may process content; no zero-operator-access commitment was found.
        subprocessors_and_routing: BytePlus infrastructure and selected third-party/open models can create additional model-specific terms.
        deletion_controls: General privacy rights exist; no inference-content TTL or per-request deletion control was found.
        caveat: Promotional token terms establish price eligibility, not privacy. They must not be treated as a data-processing agreement.
      sources:
        - {type: official_terms, title: BytePlus Terms of Service, url: https://www.byteplus.com/en/legal/terms-of-service}
        - {type: official_privacy, title: BytePlus Privacy Policy, url: https://www.byteplus.com/en/legal/privacy-policy}
        - {type: official_terms, title: ModelArk free-token campaign terms, url: https://docs.byteplus.com/en/docs/legal/termsandconditions_modelark_free-token_campaign}

    tencent_hunyuan:
      data_governance:
        review_status: partial
        plan_scope: Tencent Cloud Hunyuan API under Tencent Cloud account, privacy, and product agreements; policy location varies by region/language.
        prompt_retention: No service-specific free-tier prompt retention duration was verified from the reviewed Hunyuan documentation.
        response_retention: not_documented
        ordinary_logging: Tencent Cloud collects account, security, billing, and usage records; inference-content logging was not clearly documented.
        model_training: No current Hunyuan API commitment excluding free-tier inputs or outputs from training was verified.
        product_improvement: General cloud/service improvement uses may apply; content scope is not clear enough to characterize.
        human_or_operator_access: not_documented
        subprocessors_and_routing: Tencent Cloud and its service providers process requests in selected regions; model/product terms can differ.
        deletion_controls: General account and privacy controls exist, but no API-request deletion or TTL was found.
        caveat: The product documentation establishes API use and quota, not a complete inference-data contract; obtain the applicable regional Tencent Cloud agreement before sensitive use.
      sources:
        - {type: official_docs, title: Tencent Hunyuan API documentation, url: https://cloud.tencent.com/document/product/1729/97731}
        - {type: official_docs, title: Tencent Hunyuan usage and quota documentation, url: https://cloud.tencent.com/document/product/1729/111007}

    maritaca_academic_credits:
      data_governance:
        review_status: not_found
        caveat: Maritaca's research, model, rate-limit, and pricing documentation were reviewed, but no public agreement was found that specifies academic API prompt/output retention, training use, human access, subprocessors, or deletion controls.
      sources:
        - {type: official_product, title: Maritaca research program, url: https://www.maritaca.ai/research}
        - {type: official_docs, title: Maritaca model documentation, url: https://docs.maritaca.ai/pt/modelos}

    anyrouter:
      data_governance:
        review_status: not_found
        plan_scope: "AnyRouter watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_product", title: "Free model", url: https://anyrouter.dev/free}
        - {type: "official_model_page", title: "anyrouter/free", url: https://anyrouter.dev/model/anyrouter/free}

    nexusrouter:
      data_governance:
        review_status: not_found
        plan_scope: "NexusRouter watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_product", title: "NexusRouter", url: https://nexusrouter.net/}
        - {type: "official_terms", title: "Fair use", url: https://nexusrouter.net/fair-use}

    siliconflow:
      data_governance:
        review_status: not_found
        plan_scope: "SiliconFlow / SiliconCloud watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_docs", title: "Global rate limits", url: https://docs.siliconflow.com/en/userguide/rate-limits/rate-limit-and-upgradation}
        - {type: "official_docs", title: "China rate limits", url: https://docs.siliconflow.cn/cn/userguide/rate-limits/rate-limit-and-upgradation}

    freemodel_dev:
      data_governance:
        review_status: not_found
        plan_scope: "FreeModel.dev watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_product", title: "FreeModel.dev", url: https://www.freemodel.dev/}

    arouter:
      data_governance:
        review_status: not_found
        plan_scope: "ARouter watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_docs", title: "ARouter", url: https://docs.arouter.ai/en/faq}

    baseten:
      data_governance:
        review_status: not_found
        plan_scope: "Baseten watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_pricing", title: "Pricing", url: https://www.baseten.co/pricing/}
        - {type: "official_docs", title: "Inference API overview", url: https://docs.baseten.co/reference/inference-api/overview}

    nscale_serverless_inference:
      data_governance:
        review_status: not_found
        plan_scope: "Nscale Serverless Inference watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_docs", title: "Current overview and minimum credit", url: https://docs.nscale.com/docs/getting-started/overview}
        - {type: "official_docs", title: "Current quickstart and possible promotions", url: https://docs.nscale.com/docs/getting-started/quickstart}

    bentoml_cloud:
      data_governance:
        review_status: not_found
        plan_scope: "BentoML Cloud / Bento Inference Platform watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_pricing", title: "Pricing", url: https://www.bentoml.com/pricing}
        - {type: "historical_product_post", title: "Deployment example with older credit claim", url: https://www.bentoml.com/blog/deploying-a-text-to-speech-application-with-bentoml}

    friendli_ai:
      data_governance:
        review_status: not_found
        plan_scope: "FriendliAI watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_docs", title: "Credits", url: https://friendli.ai/docs/guides/suite/credits}
        - {type: "official_pricing", title: "Pricing", url: https://friendli.ai/pricing}

    akashml:
      data_governance:
        review_status: not_found
        plan_scope: "AkashML watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_product", title: "AkashML and FAQ", url: https://akashml.com/}
        - {type: "official_docs", title: "Documentation", url: https://akashml.com/docs}

    eden_ai:
      data_governance:
        review_status: not_found
        plan_scope: "Eden AI watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_docs", title: "Eden AI overview", url: https://www.edenai.co/docs/index.md}
        - {type: "official_docs", title: "LLM quickstart", url: https://www.edenai.co/docs/v3/quickstart/first-llm-call.md}

    sambanova_cloud:
      data_governance:
        review_status: not_found
        plan_scope: "SambaNova Cloud watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_docs", title: "Model rate limits and Free Tier", url: https://docs.sambanova.ai/docs/en/models/rate-limits}
        - {type: "official_plans", title: "Contradictory plans page", url: https://cloud.sambanova.ai/plans}

    fal_ai_builder_grant:
      data_governance:
        review_status: not_found
        plan_scope: "fal Builder Grant watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_program", title: "Builder Grant", url: https://fal.ai/builder-grant}
        - {type: "official_docs", title: "Sandbox limitations", url: https://fal.ai/docs/documentation/model-apis/sandbox}

    openai_data_sharing_tokens:
      data_governance:
        review_status: reviewed
        plan_scope: "OpenAI API complimentary data-sharing tokens watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: Shared prompts and completions are deliberately provided to OpenAI for improvement. Outside the opt-in, default API abuse-monitoring logs may contain prompts and responses for up to 30 days, while stateful endpoints have separate retention.
        response_retention: Shared outputs receive the same improvement use; Responses and other stateful APIs can store application state according to endpoint settings, in addition to abuse logs.
        ordinary_logging: Default API abuse-monitoring logs can contain customer content and derived metadata for up to 30 days unless a separately approved retention control applies.
        model_training: Data sharing is off by default, but this free-token mechanism requires an organization owner to opt in to sharing inputs and outputs for evaluation and future model training.
        product_improvement: Shared traffic is used to identify usage patterns, measure model quality, and inform future evaluation and training.
        human_or_operator_access: Shared traffic enters OpenAI improvement systems. Safety exceptions can also permit retention and human review, including for certain detected content.
        subprocessors_and_routing: OpenAI processes the selected eligible API traffic; data sent through tools such as remote MCP servers is separately governed by those third parties.
        deletion_controls: Organization owners can opt out prospectively at any time. Opting out ends future sharing but does not promise deletion or model unlearning for data already used.
        caveat: Complimentary tokens apply only to eligible organizations and only to traffic intentionally shared with OpenAI. A positive account balance is required, overages are billed, and Zero Data Retention organizations cannot enroll.
      sources:
        - {type: "official_help", title: "Complimentary tokens for shared API traffic", url: https://help.openai.com/en/articles/10306912-sharing-feedback-and-api-inputs-and-outputs-with-openai}
        - {type: "official_docs", title: "API data controls", url: https://platform.openai.com/docs/models/default-usage-policies-by-endpoint}

    llm_gateway:
      data_governance:
        review_status: not_found
        plan_scope: "LLM Gateway watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_pricing", title: "LLM Gateway", url: https://llmgateway.io/pricing}
        - {type: "official_catalog", title: "LLM Gateway", url: https://llmgateway.io/models}

    baidu_qianfan:
      data_governance:
        review_status: not_found
        plan_scope: "Baidu Qianfan watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_product", title: "Baidu Qianfan", url: https://cloud.baidu.com/product/qianfan.html}
        - {type: "official_docs", title: "Baidu Qianfan", url: https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya}

    tera_promotional_credits:
      data_governance:
        review_status: not_found
        plan_scope: "Tera watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_program", title: "Credits", url: https://www.tera.gw/credits}
        - {type: "official_pricing", title: "Pricing", url: https://www.tera.gw/pricing}

    runpod_startup_program:
      data_governance:
        review_status: not_found
        plan_scope: "Runpod startup credits watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_program", title: "Startup program", url: https://www.runpod.io/startup-program}
        - {type: "official_docs", title: "Billing", url: https://docs.runpod.io/accounts-billing/billing}

    jina_ai_llm_serp:
      data_governance:
        review_status: not_found
        plan_scope: "Jina LLM-as-SERP API watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_product", title: "Jina LLM-as-SERP API", url: https://jina.ai/api-dashboard/llm-serp/}
        - {type: "company_press_release", title: "Jina LLM-as-SERP API", url: https://jina.ai/news/llm-as-serp-search-engine-result-pages-from-large-language-models/}

    neurlap:
      data_governance:
        review_status: not_found
        plan_scope: "Neurlap watchlist offer; current free delivery or eligibility remains unresolved."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "A first-party product or offer surface was reviewed, but the available evidence does not provide a complete service-specific contract for prompt/output retention, training, human access, upstream routing, and deletion. Do not treat watchlist status or free pricing as a privacy guarantee."
      sources:
        - {type: "official_product", title: "Neurlap", url: https://neurlap.ai/}

    aimlapi:
      data_governance:
        review_status: not_found
        plan_scope: "AI/ML API historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_docs", title: "Free Tier FAQ", url: https://docs.aimlapi.com/faq/free-tier.md}

    runpod:
      data_governance:
        review_status: not_found
        plan_scope: "Runpod Serverless and Public Endpoints historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_docs", title: "Billing", url: https://docs.runpod.io/accounts-billing/billing}
        - {type: "official_docs", title: "Public endpoints quickstart", url: https://docs.runpod.io/public-endpoints/quickstart}

    digitalocean_gradient_inference:
      data_governance:
        review_status: not_found
        plan_scope: "DigitalOcean Gradient serverless inference historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_pricing", title: "DigitalOcean Gradient serverless inference", url: https://docs.digitalocean.com/products/gradient-platform/details/pricing/}
        - {type: "official_api_reference", title: "DigitalOcean Gradient serverless inference", url: https://docs.digitalocean.com/reference/api/reference/serverless-inference/}

    fal_ai_general_api:
      data_governance:
        review_status: not_found
        plan_scope: "fal Model APIs historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_pricing", title: "fal Model APIs", url: https://fal.ai/docs/documentation/model-apis/pricing}
        - {type: "official_docs", title: "fal Model APIs", url: https://fal.ai/docs/documentation/model-apis/sandbox}

    featherless_ai:
      data_governance:
        review_status: not_found
        plan_scope: "Featherless AI historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_plans", title: "Featherless AI", url: https://featherless.ai/docs/plans}
        - {type: "official_pricing", title: "Featherless AI", url: https://featherless.ai/docs/request-pricing-and-credits}

    inference_net:
      data_governance:
        review_status: not_found
        plan_scope: "Inference.net historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_pricing", title: "Inference.net", url: https://inference.net/pricing/}
        - {type: "live_catalog", title: "Inference.net", url: https://api.inference.net/v1/models}

    venice_api:
      data_governance:
        review_status: not_found
        plan_scope: "Venice API historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_pricing", title: "Venice API", url: https://venice.ai/pricing}
        - {type: "official_docs", title: "Generating an API key", url: https://docs.venice.ai/guides/getting-started/generating-api-key.md}

    infermatic:
      data_governance:
        review_status: not_found
        plan_scope: "Infermatic historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_pricing", title: "Infermatic", url: https://infermatic.ai/pricing/}

    modular_model_api:
      data_governance:
        review_status: not_found
        plan_scope: "Modular shared Model API historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_pricing", title: "Modular shared Model API", url: https://www.modular.com/pricing}

    lambda_inference_api:
      data_governance:
        review_status: not_found
        plan_scope: "Lambda hosted Inference API historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_product", title: "Lambda hosted Inference API", url: https://lambda.ai/inference}

    avian_api:
      data_governance:
        review_status: not_found
        plan_scope: "Avian API historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_pricing", title: "Avian API", url: https://api.avian.io/pricing/}

    nextbit:
      data_governance:
        review_status: not_found
        plan_scope: "NextBit historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_docs", title: "NextBit", url: https://www.nextbit256.com/docs}

    wafer:
      data_governance:
        review_status: partial
        plan_scope: "Wafer historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_terms", title: "Wafer", url: https://www.wafer.ai/terms}

    liquid_ai_hosted_api:
      data_governance:
        review_status: not_found
        plan_scope: "Liquid AI hosted API historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_pricing", title: "Liquid AI hosted API", url: https://www.liquid.ai/pricing}
        - {type: "official_pricing", title: "LEAP", url: https://leap.liquid.ai/pricing}

    petals_public_chat:
      data_governance:
        review_status: not_found
        plan_scope: "Petals public chat endpoint historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_repository", title: "Petals public chat endpoint", url: https://github.com/petals-infra/chat.petals.dev}

    webinfer_resource_pool:
      data_governance:
        review_status: not_found
        plan_scope: "WebInfer resource pool historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_product", title: "WebInfer resource pool", url: https://webllm.org/providers/resource-pool}
        - {type: "official_product", title: "WebInfer resource pool", url: https://webinfer.com/providers}

    cscs_inference:
      data_governance:
        review_status: not_found
        plan_scope: "CSCS inference service historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_docs", title: "CSCS inference service", url: https://docs.cscs.ch/services/inference/api/}

    isambard_ai_inference:
      data_governance:
        review_status: not_found
        plan_scope: "Isambard-AI inference service historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_announcement", title: "Isambard-AI inference service", url: https://www.bristol.ac.uk/research/centres/bristol-supercomputing/articles/2026/one-year-of-isambard-ai.html}

    alia_spain:
      data_governance:
        review_status: not_found
        plan_scope: "ALIA Spain historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_product", title: "ALIA Spain", url: https://alia.gob.es/eng}

    falcon_tii:
      data_governance:
        review_status: not_found
        plan_scope: "Falcon / Technology Innovation Institute historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_product", title: "Falcon / Technology Innovation Institute", url: https://falconllm.tii.ae/index.html}

    ai2_playground:
      data_governance:
        review_status: not_found
        plan_scope: "Ai2 Playground historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_docs", title: "Ai2 Playground", url: https://docs.allenai.org/quick_start/apis}

    llm_jp:
      data_governance:
        review_status: not_found
        plan_scope: "LLM-jp historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_product", title: "LLM-jp", url: https://llm-jp.nii.ac.jp/release/}

    opengradient:
      data_governance:
        review_status: not_found
        plan_scope: "OpenGradient historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_docs", title: "OpenGradient", url: https://docs.opengradient.ai/about/}

    swarmllm:
      data_governance:
        review_status: not_found
        plan_scope: "SwarmLLM historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_repository", title: "SwarmLLM", url: https://github.com/enapt/SwarmLLM}

    freellmapi_co:
      data_governance:
        review_status: not_found
        plan_scope: "FreeLLMAPI.co historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_product", title: "FreeLLMAPI.co", url: https://freellmapi.co/}

    krutrim_cloud:
      data_governance:
        review_status: not_found
        plan_scope: "Krutrim Cloud AI Studio historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_docs", title: "Krutrim Cloud AI Studio", url: https://docs.cloud.olakrutrim.com/basics/ai-studio/billing-for-ai-studio}

    duckk_africa:
      data_governance:
        review_status: not_found
        plan_scope: "Duckk Africa historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_product", title: "Duckk Africa", url: https://www.duckk.org/}

    national_university_of_singapore:
      data_governance:
        review_status: not_found
        plan_scope: "National University of Singapore generic AI services (unrelated name match) historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_announcement", title: "NUS and OpenAI strategic collaboration", url: https://news.nus.edu.sg/nus-powers-education-research-and-administration-to-new-heights-with-ai-through-a-strategic-collaboration-with-openai/}
        - {type: "official_research_site", title: "NUS AI Institute", url: https://ai.nus.edu.sg/research/}

    github_models:
      data_governance:
        review_status: not_found
        plan_scope: "GitHub Models historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_retirement", title: "Full retirement announcement", url: https://github.blog/changelog/2026-07-01-github-models-is-being-fully-retired-on-july-30-2026/}
        - {type: "official_announcement", title: "Public preview", url: https://github.blog/changelog/2024-10-29-github-models-is-now-available-in-public-preview/}

    chutes:
      data_governance:
        review_status: not_found
        plan_scope: "Chutes historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_pricing", title: "Current pricing", url: https://chutes.ai/pricing}
        - {type: "official_announcement", title: "February community announcement", url: https://chutes.ai/news/community-announcement-february}

    together_ai:
      data_governance:
        review_status: not_found
        plan_scope: "Together AI historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_docs", title: "Billing and credits", url: https://docs.together.ai/docs/billing-credits}
        - {type: "official_pricing", title: "Inference pricing", url: https://docs.together.ai/docs/inference/pricing}

    deepinfra:
      data_governance:
        review_status: not_found
        plan_scope: "DeepInfra historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_pricing", title: "DeepInfra", url: https://deepinfra.com/pricing}
        - {type: "official_api_reference", title: "DeepInfra", url: https://docs.deepinfra.com/api-reference/introduction}

    qwen_code_oauth:
      data_governance:
        review_status: not_found
        plan_scope: "Qwen OAuth free access for Qwen Code historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "official_docs", title: "Qwen OAuth free access for Qwen Code", url: https://qwenlm.github.io/qwen-code-docs/en/users/configuration/auth/}

    public_ai_old_blanket_free:
      data_governance:
        review_status: not_found
        plan_scope: "Public AI old blanket-free offer historical or excluded catalog record; no current qualifying free offer is represented."
        prompt_retention: not_documented
        response_retention: not_documented
        ordinary_logging: not_documented
        model_training: not_documented
        product_improvement: not_documented
        human_or_operator_access: not_documented
        subprocessors_and_routing: not_documented
        deletion_controls: not_documented
        caveat: "This record is retained to explain an excluded, retired, paid-only, web-only, or otherwise non-qualifying offer. Its cited first-party surface does not establish a complete current data-governance profile for free API use. Review the operator's current legal terms if using another paid or successor service."
      sources:
        - {type: "historical_announcement", title: "Public AI old blanket-free offer", url: https://huggingface.co/blog/inference-providers-publicai}

    huggingface_spaces_as_a_class:
      data_governance:
        review_status: not_applicable
        plan_scope: "Hugging Face Spaces / Gradio demos"
        caveat: "This is a class of independently operated applications, not one inference provider. Each Space owner and any upstream service controls its own data handling."
      sources: []

    self_hosted_open_weights:
      data_governance:
        review_status: not_applicable
        plan_scope: "Self-hosted open-weight models"
        caveat: "This is a class of locally operated weights, not a remote provider. Data handling belongs to the deployer, infrastructure, and application chosen by the user."
      sources: []
