{
  "source_commit": "2dc3a78",
  "generated_at": "2026-09-11T00:56:04.735Z",
  "providers": [
    {
      "id": "alibaba-dashscope",
      "name": "Alibaba Model Studio (DashScope)",
      "kind": "first_party",
      "urls": {
        "docs": "https://www.alibabacloud.com/help/en/model-studio/compatibility-of-openai-with-dashscope",
        "console": "https://modelstudio.console.alibabacloud.com"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "DASHSCOPE_API_KEY"
          ],
          "key_prefix": "sk-",
          "extra_headers": {},
          "getting_credentials": "Model Studio console → API Keys. Keys are region-bound: international dashscope-intl vs China dashscope endpoints.",
          "docs": "https://www.alibabacloud.com/help/en/model-studio/first-api-call-to-qwen"
        }
      ],
      "endpoints": [
        {
          "id": "compatible-mode-chat",
          "base_url": "https://dashscope-intl.aliyuncs.com",
          "path": "/compatible-mode/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "anthropic",
          "base_url": "https://dashscope-intl.aliyuncs.com",
          "path": "/apps/anthropic",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings",
        "batch"
      ],
      "quirks": [
        {
          "text": "China base: https://dashscope.aliyuncs.com/compatible-mode/v1. Soft switches /think and /no_think appended to prompts work on open-source hybrid Qwen3 models and qwen-plus-2025-04-28+ (most recent instruction wins). Open-source hybrid models (qwen3-235b-a22b, qwen3-32b) support streaming only when enable_thinking=true — non-streaming calls error. Thinking-only variants (qwen3-next-80b-a3b-thinking, qwq-plus, kimi-k2-thinking) cannot disable. Regional OpenAI-compatible bases like https://{workspace}.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1 also exist; native DashScope endpoint .../api/v1/services/aigc/text-generation/generation with X-DashScope-SSE: enable.",
          "docs": "https://www.alibabacloud.com/help/en/model-studio/deep-thinking"
        },
        {
          "text": "Pay-as-you-go regions: China dashscope.aliyuncs.com, Singapore dashscope-intl, Virginia dashscope-us — each with /compatible-mode/v1 (OpenAI), /apps/anthropic (Anthropic), and native /api/v1. Workspace-dedicated domains https://{WorkspaceId}.{region}.maas.aliyuncs.com are recommended for production; rate-limited trial domains exist (trial.cn-beijing / trial.ap-southeast-1).",
          "docs": "https://help.aliyun.com/en/model-studio/base-url"
        },
        {
          "text": "Temporary API keys: POST /api/v1/tokens (Bearer with permanent key) returns short-lived st- tokens (default 60s, max 1800s) inheriting the parent key's permissions — for browser/mobile use. Alibaba Cloud AccessKey/STS is NOT accepted on inference endpoints.",
          "docs": "https://www.alibabacloud.com/help/en/model-studio/application-obtain-temporary-authentication-token"
        },
        {
          "text": "Savings/resource plans are billing constructs on the same keys and endpoints: AI General-purpose Savings Plan (from $150/mo, discounts up to 32%/47%, no discount for DeepSeek/Kimi/GLM/MiniMax) and LLM Savings Plan prepaid credits (no discount). QwenCloud (qwencloud.com) is a separate-account console using these same endpoints and key formats; Bailian is the China brand of Model Studio.",
          "docs": "https://www.alibabacloud.com/help/en/model-studio/savings-plan-and-resource-package"
        }
      ]
    },
    {
      "id": "alibaba-token-plan",
      "name": "Alibaba Token Plan",
      "kind": "subscription",
      "urls": {
        "docs": "https://www.alibabacloud.com/help/en/model-studio/token-plan-overview",
        "console": "https://modelstudio.console.alibabacloud.com/",
        "pricing": "https://www.alibabacloud.com/help/en/model-studio/token-plan-personal-overview"
      },
      "auth": [
        {
          "id": "plan-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "BAILIAN_TOKEN_PLAN_API_KEY"
          ],
          "key_prefix": "sk-sp-",
          "extra_headers": {},
          "getting_credentials": "Subscribe in Model Studio (Singapore region); plan keys only work on token-plan endpoints. Env var name not officially documented — verify.",
          "docs": "https://www.alibabacloud.com/help/en/model-studio/token-plan-team-quickstart"
        }
      ],
      "endpoints": [
        {
          "id": "compatible-mode",
          "base_url": "https://token-plan.ap-southeast-1.maas.aliyuncs.com",
          "path": "/compatible-mode/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "anthropic",
          "base_url": "https://token-plan.ap-southeast-1.maas.aliyuncs.com",
          "path": "/apps/anthropic",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "China base: https://token-plan.cn-beijing.maas.aliyuncs.com (same paths). Model allowlist (exact strings): qwen3.8-max-preview (10x credit discount, night pricing 22:00-08:00 UTC+8), qwen3.7-max, qwen3.7-plus, qwen3.6-flash, glm-5.2, deepseek-v4-pro, wan2.7-image(-pro), happyhorse-1.1 video models; Team edition adds Kimi k2.7-code/k2.6/k2.5, GLM 5.1/5, MiniMax-M2.5, qwen3.6-plus/flash.",
          "docs": "https://www.alibabacloud.com/help/en/model-studio/token-plan-personal-overview"
        },
        {
          "text": "Team Edition: seat = one member + one auto-generated key; deduction order seat quota then shared packs then suspension; TPS/TPM limits at primary-account level; RAM policies AliyunTokenPlanFullAccess gate purchase.",
          "docs": "https://www.alibabacloud.com/help/en/model-studio/token-plan-overview"
        }
      ],
      "plan": {
        "price_usd": 6,
        "period": "monthly",
        "quota": "Personal: Lite $6 (700 credits/5h, 2500/7d), Standard $20 (3000/10k), Pro $70 (12k/40k). Team: Standard $30/seat (25k credits), Pro $100/seat (100k), Max $200/seat (250k), shared packs $700/625k.",
        "notes": "Sliding 5-hour + 7-day windows, no monthly reset. sk-sp- keys are NOT interchangeable with Coding Plan keys (shared prefix, different product). Interactive-tool use only (Claude Code, Cursor, Qwen Code, OpenClaw); scripts/backends banned on Personal.",
        "docs": "https://www.alibabacloud.com/help/en/model-studio/token-plan-personal-overview"
      }
    },
    {
      "id": "anthropic",
      "name": "Anthropic",
      "kind": "first_party",
      "urls": {
        "docs": "https://platform.claude.com/docs/en/api/overview",
        "console": "https://console.anthropic.com",
        "status": "https://status.anthropic.com",
        "pricing": "https://platform.claude.com/docs/en/docs/about-claude/models"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "x-api-key",
          "env": [
            "ANTHROPIC_API_KEY"
          ],
          "key_prefix": "sk-ant-api",
          "extra_headers": {
            "anthropic-version": "2023-06-01"
          },
          "getting_credentials": "Console → Settings → API Keys. Keys are org-scoped; optional expiry (3h to never).",
          "docs": "https://platform.claude.com/docs/en/manage-claude/authentication"
        },
        {
          "id": "oauth",
          "type": "oauth",
          "transport": "header",
          "header": "x-api-key",
          "env": [],
          "extra_headers": {
            "anthropic-beta": "oauth-2025-04-20",
            "anthropic-version": "2023-06-01"
          },
          "getting_credentials": "Run Claude Code and sign in with a Claude subscription; OAuth tokens are sent via x-api-key (not Bearer).",
          "docs": "https://code.claude.com/docs/en/authentication",
          "flow": "authorization_code_pkce",
          "scopes": [
            "org:create_api_key",
            "user:profile",
            "user:inference"
          ],
          "token_transport": "header"
        }
      ],
      "endpoints": [
        {
          "id": "v1-messages",
          "base_url": "https://api.anthropic.com",
          "path": "/v1/messages",
          "protocol": "anthropic-messages"
        },
        {
          "id": "v1-messages-batches",
          "base_url": "https://api.anthropic.com",
          "path": "/v1/messages/batches",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "batch",
        "count_tokens",
        "prompt_caching",
        "files"
      ],
      "quirks": [
        {
          "text": "Thinking is incompatible with temperature/top_p/top_k, forced tool use, and prefill; changing thinking params invalidates prompt caches; thinking signatures must be replayed unmodified in tool-use turns.",
          "docs": "https://platform.claude.com/docs/en/docs/build-with-claude/extended-thinking"
        }
      ]
    },
    {
      "id": "aws-bedrock",
      "name": "AWS Bedrock",
      "kind": "cloud_hosted",
      "urls": {
        "docs": "https://docs.aws.amazon.com/bedrock/latest/userguide/api-keys.html",
        "console": "https://console.aws.amazon.com/bedrock"
      },
      "auth": [
        {
          "id": "bearer-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "AWS_BEARER_TOKEN_BEDROCK"
          ],
          "extra_headers": {},
          "getting_credentials": "Bedrock console → API keys. Short-term (<=12h) or long-lived; governed by bedrock:CallWithBearerToken. Not valid for bidirectional streaming, Agents, or Data Automation.",
          "docs": "https://docs.aws.amazon.com/bedrock/latest/userguide/api-keys.html"
        },
        {
          "id": "sigv4",
          "type": "sigv4",
          "transport": "request_signing",
          "env": [
            "AWS_ACCESS_KEY_ID",
            "AWS_SECRET_ACCESS_KEY",
            "AWS_SESSION_TOKEN"
          ],
          "extra_headers": {},
          "getting_credentials": "Standard AWS credentials chain (IAM user, role, SSO). Signs the request with SigV4; service 'bedrock'.",
          "docs": "https://docs.aws.amazon.com/bedrock/latest/userguide/security_iam.html"
        }
      ],
      "endpoints": [
        {
          "id": "converse",
          "base_url": "https://bedrock-runtime.{region}.amazonaws.com",
          "path": "/model/{modelId}/converse",
          "protocol": "bedrock-converse",
          "auth": "sigv4"
        },
        {
          "id": "invoke-anthropic",
          "base_url": "https://bedrock-runtime.{region}.amazonaws.com",
          "path": "/model/{modelId}/invoke",
          "protocol": "anthropic-messages",
          "auth": "sigv4"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "batch",
        "embeddings"
      ],
      "quirks": [
        {
          "text": "Cross-region inference profiles prefix model IDs with us./eu./apac./global. — e.g. us.anthropic.claude-sonnet-5. Streaming required when max_tokens > 21,333 with thinking.",
          "docs": "https://docs.aws.amazon.com/bedrock/latest/userguide/inference-profiles-support.html"
        }
      ]
    },
    {
      "id": "azure-foundry",
      "name": "Azure AI Foundry",
      "kind": "cloud_hosted",
      "urls": {
        "docs": "https://learn.microsoft.com/en-us/azure/foundry/openai/api-version-lifecycle",
        "console": "https://ai.azure.com"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "api-key",
          "env": [
            "AZURE_OPENAI_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Foundry resource → Keys and Endpoint. 32-char hex keys.",
          "docs": "https://learn.microsoft.com/en-us/azure/foundry/openai/api-version-lifecycle"
        },
        {
          "id": "entra",
          "type": "entra_bearer",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [],
          "extra_headers": {},
          "getting_credentials": "Microsoft Entra token with scope https://ai.azure.com/.default; SDKs refresh automatically. Roles like 'Cognitive Services OpenAI User'.",
          "docs": "https://learn.microsoft.com/en-us/azure/foundry/openai/api-version-lifecycle"
        }
      ],
      "endpoints": [
        {
          "id": "v1-responses",
          "base_url": "https://{resource}.openai.azure.com",
          "path": "/openai/v1/responses",
          "protocol": "openai-responses"
        },
        {
          "id": "v1-chat-completions",
          "base_url": "https://{resource}.openai.azure.com",
          "path": "/openai/v1/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings",
        "files",
        "batch",
        "fine_tuning"
      ],
      "quirks": [
        {
          "text": "The v1 route needs no api-version query param and works with unmodified OpenAI SDKs (set OPENAI_BASE_URL). Classic route uses deployment names + api-version; responses include content_filter_results.",
          "docs": "https://learn.microsoft.com/en-us/azure/foundry/openai/api-version-lifecycle"
        }
      ]
    },
    {
      "id": "baseten",
      "name": "Baseten",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.baseten.co/inference/model-apis/overview",
        "console": "https://www.baseten.co",
        "pricing": "https://www.baseten.co/pricing/"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "BASETEN_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create keys in the Baseten dashboard (Settings > API Keys). Send as Authorization: Bearer BASETEN_API_KEY; the legacy Api-Key header scheme is also accepted.",
          "docs": "https://docs.baseten.co/inference/model-apis/overview"
        }
      ],
      "endpoints": [
        {
          "id": "chat-completions",
          "base_url": "https://inference.baseten.co",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "messages",
          "base_url": "https://inference.baseten.co",
          "path": "/v1/messages",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Baseten reads Authorization only — the Anthropic SDK's default x-api-key is NOT read; override default_headers.",
          "docs": "https://docs.baseten.co/inference/model-apis/overview"
        },
        {
          "text": "reasoning_effort is validated (invalid values 400); chat_template_args.enable_thinking on opt-in models is unvalidated and fails silently.",
          "docs": "https://docs.baseten.co/inference/model-apis/reasoning"
        },
        {
          "text": "Rate limits are account-wide (Basic verified 120 RPM / 500k TPM); 529 = overloaded; GET /v1/models embeds pricing and context.",
          "docs": "https://docs.baseten.co/inference/model-apis/pricing-and-limits"
        },
        {
          "text": "The Anthropic-compatible /v1/messages endpoint is in beta; behavior may change before general availability — prefer Chat Completions for production workloads.",
          "docs": "https://docs.baseten.co/inference/model-apis/overview"
        }
      ]
    },
    {
      "id": "cohere",
      "name": "Cohere",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.cohere.com/docs/models",
        "console": "https://dashboard.cohere.com",
        "pricing": "https://www.cohere.com/pricing"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "COHERE_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create a key at dashboard.cohere.com; the same key works on the native v2 API and the OpenAI-compatibility surface.",
          "docs": "https://docs.cohere.com/docs/models"
        }
      ],
      "endpoints": [
        {
          "id": "compat-chat-completions",
          "base_url": "https://api.cohere.ai",
          "path": "/compatibility/v1/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings",
        "rerank"
      ],
      "quirks": [
        {
          "text": "Native v2 API at api.cohere.com (different request shape); the OpenAI-compat surface at api.cohere.ai/compatibility/v1 supports chat, embeddings, audio transcriptions — reasoning_effort only none|high there; no store/logit_bias/n.",
          "docs": "https://docs.cohere.com/v2"
        },
        {
          "text": "North is Cohere's enterprise agentic platform (not a model API). Embed/Rerank billed via Model Vault instances.",
          "docs": "https://www.cohere.com/north"
        },
        {
          "text": "Aya vision/translate variants and 8B Ayas retired 2026-04-04.",
          "docs": "https://docs.cohere.com/release-notes"
        }
      ]
    },
    {
      "id": "deepseek",
      "name": "DeepSeek",
      "kind": "first_party",
      "urls": {
        "docs": "https://api-docs.deepseek.com/",
        "console": "https://platform.deepseek.com"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "DEEPSEEK_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create a key at platform.deepseek.com. Works in Claude Code, Copilot, OpenCode via the Anthropic-compatible endpoint.",
          "docs": "https://api-docs.deepseek.com/"
        }
      ],
      "endpoints": [
        {
          "id": "chat-completions",
          "base_url": "https://api.deepseek.com",
          "path": "/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "anthropic",
          "base_url": "https://api.deepseek.com",
          "path": "/anthropic",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Root path /chat/completions works without /v1. Stateless OpenAI Responses API also supported. In tool-call flows, prior-turn reasoning_content must be passed back or you get a 400; without tools it is ignored. When thinking is on, temperature/top_p/presence_penalty/frequency_penalty are silently ignored. FIM only in non-thinking mode. The old deepseek-chat / deepseek-reasoner split is gone — same model names serve both modes via the thinking toggle.",
          "docs": "https://api-docs.deepseek.com/guides/thinking_mode"
        }
      ]
    },
    {
      "id": "fireworks-ai",
      "name": "Fireworks AI",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.fireworks.ai/",
        "console": "https://app.fireworks.ai",
        "pricing": "https://docs.fireworks.ai/serverless/pricing"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "FIREWORKS_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create keys in the Fireworks console (app.fireworks.ai, Account Settings > API Keys).",
          "docs": "https://docs.fireworks.ai/"
        }
      ],
      "endpoints": [
        {
          "id": "inference-v1",
          "base_url": "https://api.fireworks.ai",
          "path": "/inference/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "responses",
          "base_url": "https://api.fireworks.ai",
          "path": "/inference/v1/responses",
          "protocol": "openai-responses"
        },
        {
          "id": "anthropic",
          "base_url": "https://api.fireworks.ai",
          "path": "/inference/v1/messages",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings",
        "batch",
        "rerank"
      ],
      "quirks": [
        {
          "text": "reasoning_effort (low/medium/high) AND an Anthropic-style thinking object {type: enabled, budget_tokens >= 1024} are both accepted on chat/completions — specifying both errors.",
          "docs": "https://docs.fireworks.ai/guides/reasoning"
        },
        {
          "text": "Interleaved thinking: you MUST echo prior assistant reasoning_content when the last message is a tool result; reasoning_history: preserved retains reasoning across turns.",
          "docs": "https://docs.fireworks.ai/guides/reasoning"
        },
        {
          "text": "Batch is 50% of serverless price via the Fireworks-native batch API (not OpenAI /v1/batches); Responses API stores by default; streaming includes usage in the final chunk.",
          "docs": "https://docs.fireworks.ai/guides/reasoning"
        }
      ]
    },
    {
      "id": "google-gemini",
      "name": "Google Gemini API",
      "kind": "first_party",
      "urls": {
        "docs": "https://ai.google.dev/api",
        "console": "https://aistudio.google.com"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "x-goog-api-key",
          "env": [
            "GEMINI_API_KEY",
            "GOOGLE_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "AI Studio → Get API key. Prefer the x-goog-api-key header; the legacy ?key= query param leaks keys into logs.",
          "docs": "https://ai.google.dev/api"
        }
      ],
      "endpoints": [
        {
          "id": "generate-content",
          "base_url": "https://generativelanguage.googleapis.com",
          "path": "/v1beta/models/{model}:generateContent",
          "protocol": "google-generate-content"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings",
        "files",
        "batch",
        "count_tokens",
        "prompt_caching"
      ],
      "quirks": [
        {
          "text": "OpenAI-compatible route: use https://generativelanguage.googleapis.com/v1beta/openai/ as base URL with the same API key.",
          "docs": "https://ai.google.dev/gemini-api/docs/openai"
        }
      ]
    },
    {
      "id": "google-vertex",
      "name": "Google Vertex AI",
      "kind": "cloud_hosted",
      "urls": {
        "docs": "https://docs.cloud.google.com/vertex-ai",
        "console": "https://console.cloud.google.com"
      },
      "auth": [
        {
          "id": "adc",
          "type": "adc",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "GOOGLE_APPLICATION_CREDENTIALS"
          ],
          "extra_headers": {},
          "getting_credentials": "gcloud auth application-default login; classic endpoints reject API keys (express mode adds API-key auth for some endpoints).",
          "docs": "https://ai.google.dev/gemini-api/docs/migrate-to-cloud"
        }
      ],
      "endpoints": [
        {
          "id": "generate-content",
          "base_url": "https://{region}-aiplatform.googleapis.com",
          "path": "/v1/projects/{project}/locations/{region}/publishers/google/models/{model}:generateContent",
          "protocol": "google-generate-content"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings",
        "batch",
        "count_tokens",
        "prompt_caching"
      ],
      "quirks": []
    },
    {
      "id": "hetzner",
      "name": "Hetzner Inference (Experiments)",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.hetzner.com/general/company-and-policy/experiments/inference/",
        "console": "https://experiments.hetzner.com"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "HETZNER_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create a token in the Experiments dashboard; free while experimental, no SLA.",
          "docs": "https://docs.hetzner.com/general/company-and-policy/experiments/inference/"
        }
      ],
      "endpoints": [
        {
          "id": "chat",
          "base_url": "https://inference.hetzner.com",
          "path": "/api/v1/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Experimental 'as is' program (launched ~2026-07): free, EU-based, no production guarantees; email notice before changes. GET /api/v1/models is definitive for the lineup.",
          "docs": "https://docs.hetzner.com/general/company-and-policy/experiments/inference/"
        },
        {
          "text": "Limits per key: 10 requests/60s, 4M input and 100k output tokens/60s. Reasoning control is undocumented; a community-reported chat_template_kwargs.enable_thinking flag exists for Qwen3.6.",
          "docs": "https://docs.hetzner.com/general/company-and-policy/experiments/inference/"
        }
      ]
    },
    {
      "id": "io-intelligence",
      "name": "IO Intelligence (io.net)",
      "kind": "first_party",
      "urls": {
        "docs": "https://io.net/docs/guides/intelligence/io-intelligence-apis",
        "console": "https://ai.io.net",
        "pricing": "https://io.net/docs/guides/payment/io-intelligence-payments"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "IOINTELLIGENCE_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create keys at ai.io.net/ai/api-keys with expiry and permission scoping.",
          "docs": "https://io.net/docs/guides/intelligence/api-keys-and-secrets"
        }
      ],
      "endpoints": [
        {
          "id": "chat",
          "base_url": "https://api.intelligence.io.solutions",
          "path": "/api/v1/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Base path is /api/v1 (not /v1) on the io.solutions domain; api.intelligence.io.net does not resolve.",
          "docs": "https://io.net/docs/guides/intelligence/io-intelligence-apis"
        },
        {
          "text": "Per-token prices (USD, scientific notation) come from GET /api/v1/models; model access gated by min_access_tier 1-3 by plan (Standard free-light/Professional $15/Developer $150 daily credits).",
          "docs": "https://api.intelligence.io.solutions/api/v1/models"
        },
        {
          "text": "reasoning.effort none/low/medium/high is primary; legacy top-level reasoning_effort still accepted; MiniMax-M2.7 is always-on (none is a no-op; some models 400 on none).",
          "docs": "https://io.net/docs/guides/intelligence/reasoning-for-chat-completions"
        }
      ]
    },
    {
      "id": "kimi-coding",
      "name": "Kimi for Coding",
      "kind": "subscription",
      "urls": {
        "docs": "https://www.kimi.com/code/docs/",
        "console": "https://www.kimi.com/code/console"
      },
      "auth": [
        {
          "id": "plan-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [],
          "extra_headers": {},
          "getting_credentials": "Create keys in the Kimi Code Console (shown once). For Claude Code: ANTHROPIC_BASE_URL=https://api.kimi.com/coding/ and ANTHROPIC_API_KEY=<coding key>.",
          "docs": "https://www.kimi.com/code/docs/en/third-party-tools/claude-code.html"
        }
      ],
      "endpoints": [
        {
          "id": "coding-openai",
          "base_url": "https://api.kimi.com",
          "path": "/coding/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "coding-anthropic",
          "base_url": "https://api.kimi.com",
          "path": "/coding",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Model ids on the coding endpoints: kimi-for-coding, kimi-for-coding-highspeed, k3 (1M ctx, ~2x quota), k3-256k. The [1m] suffix (kimi-k3[1m]) is a Claude Code env convention only — raw API uses k3/kimi-k3. Claude Code /effort maps to k3 low/high/max.",
          "docs": "https://www.kimi.com/code/docs/en/third-party-tools/claude-code.html"
        },
        {
          "text": "kimi-for-coding is the K2.7-code-class model; highspeed variant doubles speed and quota consumption.",
          "docs": "https://www.kimi.com/code/docs/kimi-code/faq.html"
        }
      ],
      "plan": {
        "price_usd": 19,
        "period": "monthly",
        "quota": "Kimi Code has its own rolling 5-hour + weekly limits plus shared agent credits (60/150/360/720 by tier); concurrency-limited",
        "notes": "Included with Kimi membership — tiers Moderato $19, Allegretto $39, Allegro $99, Vivace $199 (annual discounts); K3 access from Moderato up. Coding keys are distinct from Open Platform keys and never interchangeable (top 401 cause).",
        "docs": "https://www.kimi.ai/help/membership/membership-pricing"
      }
    },
    {
      "id": "meta",
      "name": "Meta Model API",
      "kind": "first_party",
      "urls": {
        "docs": "https://dev.meta.ai/docs/overview/",
        "console": "https://dev.meta.ai",
        "pricing": "https://dev.meta.ai/docs/pricing-rate-limits/"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "MODEL_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create keys in the dev.meta.ai dashboard (API Keys tab). For Claude Code: ANTHROPIC_BASE_URL=https://api.meta.ai with ANTHROPIC_AUTH_TOKEN=<key>.",
          "docs": "https://dev.meta.ai/docs/quickstart/"
        }
      ],
      "endpoints": [
        {
          "id": "v1-responses",
          "base_url": "https://api.meta.ai",
          "path": "/v1/responses",
          "protocol": "openai-responses"
        },
        {
          "id": "v1-chat-completions",
          "base_url": "https://api.meta.ai",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "v1-messages",
          "base_url": "https://api.meta.ai",
          "path": "/v1/messages",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "files"
      ],
      "quirks": [
        {
          "text": "The Llama API (api.llama.com/compat/v1) was retired 2026-07-06 — there is no first-party hosted Llama inference; Llama models are open weights served by third parties.",
          "docs": "https://www.promptfoo.dev/docs/providers/llamaApi/"
        },
        {
          "text": "reasoning_effort none returns 400 (always-on reasoning). Chat Completions exposes reasoning_content redacted to empty for external callers; use the Responses surface for summaries and encrypted reasoning replay. No logprobs, no n>1, no stop.",
          "docs": "https://dev.meta.ai/docs/reasoning"
        },
        {
          "text": "All keys share team limits: Standard 3,000 RPM / 4M TPM. Contributor tier (muse-spark-1.2-contributor) trades a heavy discount for prompts possibly training future Meta models.",
          "docs": "https://dev.meta.ai/docs/pricing-rate-limits/"
        }
      ]
    },
    {
      "id": "minimax",
      "name": "MiniMax",
      "kind": "first_party",
      "urls": {
        "docs": "https://platform.minimax.io/docs/api-reference/api-overview",
        "console": "https://platform.minimax.io"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "MINIMAX_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "platform.minimax.io → API Keys. Subscription Keys are a separate credential type issued under Billing → Token Plan.",
          "docs": "https://platform.minimax.io/docs/api-reference/api-overview"
        }
      ],
      "endpoints": [
        {
          "id": "chat-completions",
          "base_url": "https://api.minimax.io",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "anthropic",
          "base_url": "https://api.minimax.io",
          "path": "/anthropic",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "reasoning_split: true splits thinking into message.reasoning_content; without it thinking arrives inline as <think>...</think> inside content. China base: api.minimaxi.com.",
          "docs": "https://platform.minimax.io/docs/api-reference/text-chat-openai"
        }
      ]
    },
    {
      "id": "minimax-token-plan",
      "name": "MiniMax Token Plan",
      "kind": "subscription",
      "urls": {
        "docs": "https://platform.minimax.io/docs/token-plan/intro",
        "console": "https://platform.minimax.io/subscribe/token-plan"
      },
      "auth": [
        {
          "id": "subscription-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "MINIMAX_SUBSCRIPTION_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Subscribe at platform.minimax.io/subscribe/token-plan; the Subscription Key is issued separately from standard API keys under Billing > Token Plan.",
          "docs": "https://platform.minimax.io/docs/token-plan/intro"
        }
      ],
      "endpoints": [
        {
          "id": "anthropic",
          "base_url": "https://api.minimax.io",
          "path": "/anthropic",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Anthropic-compatible endpoint for coding tools (Claude Code). China base: https://api.minimaxi.com/anthropic.",
          "docs": "https://platform.minimax.io/docs/token-plan/claude-code"
        }
      ],
      "plan": {
        "price_usd": 20,
        "period": "monthly",
        "quota": "Plus ~1.7B tokens/mo (3-4 concurrent agents); Max ~5.1B; higher tiers to ~$120",
        "notes": "Subscription Key is a separate credential type from pay-as-you-go API keys, issued under Billing > Token Plan. Sources conflict on Plus pricing ($20 vs $40) — verify at the subscribe page. For Claude Code: ANTHROPIC_BASE_URL=https://api.minimax.io/anthropic + ANTHROPIC_AUTH_TOKEN=<Subscription Key>.",
        "docs": "https://platform.minimax.io/docs/token-plan/claude-code"
      }
    },
    {
      "id": "mistral",
      "name": "Mistral",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.mistral.ai/api/endpoint/chat",
        "console": "https://console.mistral.ai"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "MISTRAL_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create a key at console.mistral.ai (La Plateforme).",
          "docs": "https://docs.mistral.ai/api/endpoint/chat"
        }
      ],
      "endpoints": [
        {
          "id": "v1-chat-completions",
          "base_url": "https://api.mistral.ai",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings",
        "files",
        "batch"
      ],
      "quirks": [
        {
          "text": "With reasoning_effort high, message.content becomes a list of chunks: ThinkChunk (type thinking) followed by TextChunk; replay the full assistant message including ThinkChunk in multi-turn or quality degrades. No Responses API. Magistral reasoning models deprecated; reasoning_effort high/none supported on mistral-small-latest and mistral-medium-3-5.",
          "docs": "https://docs.mistral.ai/capabilities/reasoning/"
        }
      ]
    },
    {
      "id": "moonshot",
      "name": "Moonshot Kimi",
      "kind": "first_party",
      "urls": {
        "docs": "https://platform.kimi.ai/docs/overview",
        "console": "https://platform.kimi.ai/console/api-keys",
        "pricing": "https://platform.kimi.ai/docs/models"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "MOONSHOT_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create keys at the Kimi Open Platform (intl) or platform.kimi.com (CN, Chinese phone required). Keys are region-bound: .ai and .cn are not interchangeable.",
          "docs": "https://platform.kimi.ai/docs/overview"
        }
      ],
      "endpoints": [
        {
          "id": "chat-completions",
          "base_url": "https://api.moonshot.ai",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "anthropic",
          "base_url": "https://api.moonshot.ai",
          "path": "/anthropic",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "files",
        "batch"
      ],
      "quirks": [
        {
          "text": "China base: https://api.moonshot.cn/v1 (console platform.kimi.com). No embeddings endpoint. Partial Mode (partial: true) and ms:// file references are Moonshot extensions. Context caching is automatic for prompts over 256 tokens. Rate limits tier by cumulative top-up.",
          "docs": "https://platform.kimi.ai/docs/overview"
        },
        {
          "text": "Sampling params are fixed on k-series models (temperature 1.0, top_p 0.95; k2.6 uses 0.6 outside thinking mode) and error if set otherwise; only moonshot-v1 allows modifying temperature. tool_choice required is k3-only.",
          "docs": "https://platform.kimi.ai/docs/guide/use-thinking-models"
        },
        {
          "text": "The /anthropic endpoint (for Claude Code et al.) uses OPEN PLATFORM keys — never Kimi Code keys. WebFetch is unsupported there; WebSearch 400s on kimi-k2.7-code with thinking off.",
          "docs": "https://platform.kimi.ai/docs/guide/claude-code-kimi"
        }
      ]
    },
    {
      "id": "near-ai",
      "name": "NEAR AI Cloud",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.near.ai/cloud/quickstart",
        "console": "https://cloud.near.ai"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "NEAR_AI_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Keys from cloud.near.ai Developer Dashboard; prepaid Usage Credits (or NEAR staking yield). The old api.near.ai went offline 2025-10-31.",
          "docs": "https://docs.near.ai/cloud/quickstart"
        }
      ],
      "endpoints": [
        {
          "id": "chat",
          "base_url": "https://cloud-api.near.ai",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "responses",
          "base_url": "https://cloud-api.near.ai",
          "path": "/v1/responses",
          "protocol": "openai-responses"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings",
        "image_gen"
      ],
      "quirks": [
        {
          "text": "Inference runs in hardware TEEs with attestation (/v1/attestation/report) and response signatures; direct per-model endpoints at {slug}.completions.near.ai terminate TLS inside the enclave.",
          "docs": "https://docs.near.ai/cloud/quickstart"
        },
        {
          "text": "Per-model prices live in GET /v1/models (public, no auth); frontier OpenAI/Anthropic/Google models are resold through the gateway in OpenAI format.",
          "docs": "https://cloud-api.near.ai/v1/models"
        },
        {
          "text": "Reasoning params vary by model: reasoning_effort (gpt-oss, always-on), chat_template_kwargs.enable_thinking (GLM/Qwen), chat_template_kwargs.thinking (Kimi — not enable_thinking).",
          "docs": "https://docs.near.ai/cloud/reasoning-models/"
        },
        {
          "text": "The OpenAI Responses-compatible /v1/responses endpoint on the gateway is in beta.",
          "docs": "https://docs.near.ai/cloud/guides/openai-compatibility"
        }
      ]
    },
    {
      "id": "nvidia",
      "name": "NVIDIA NIM API",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.api.nvidia.com/nim/reference/llm-apis",
        "console": "https://build.nvidia.com"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "NVIDIA_API_KEY"
          ],
          "key_prefix": "nvapi-",
          "extra_headers": {},
          "getting_credentials": "Generate nvapi- keys at build.nvidia.com (free with the NVIDIA Developer Program).",
          "docs": "https://build.nvidia.com"
        }
      ],
      "endpoints": [
        {
          "id": "v1-chat-completions",
          "base_url": "https://integrate.api.nvidia.com",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings"
      ],
      "quirks": [
        {
          "text": "No pay-as-you-go: hosted access is free-tier rate limited (~40 RPM soft); production requires deploying NIM yourself or NVIDIA AI Enterprise.",
          "docs": "https://forums.developer.nvidia.com/t/request-for-nvidia-nim-api-rate-limit-credits-increase/375762"
        },
        {
          "text": "The real reasoning switch is chat_template_kwargs (extra_body): enable_thinking (Qwen-style), medium_effort (Nemotron-3), reasoning_strength (muse-glimmer). reasoning_effort is accepted by the schema but silently ignored on some models.",
          "docs": "https://docs.api.nvidia.com/nim/reference/nvidia-nemotron-3-ultra-550b-a55b"
        },
        {
          "text": "Hosted temperature range is 0-1 (default 0.95), not 0-2. Old catalog entries (e.g. meta/llama2-70b) still listed but 404.",
          "docs": "https://docs.api.nvidia.com/nim/reference/meta-muse-glimmer-30b"
        }
      ]
    },
    {
      "id": "ollama-cloud",
      "name": "Ollama Cloud",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.ollama.com/cloud",
        "console": "https://ollama.com/settings/keys",
        "pricing": "https://ollama.com/cloud"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "OLLAMA_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create keys at ollama.com/settings/keys (they do not currently expire).",
          "docs": "https://docs.ollama.com/api/authentication"
        }
      ],
      "endpoints": [
        {
          "id": "v1-chat-completions",
          "base_url": "https://ollama.com",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Native API at https://ollama.com/api (chat/generate/tags) uses the think parameter (low/medium/high/max); gpt-oss models accept only low/medium/high and cannot disable thinking. An Anthropic-compatible API is also documented. https://ollama.com/api/v1 does NOT exist (404).",
          "docs": "https://docs.ollama.com/capabilities/thinking"
        },
        {
          "text": "The OpenAI-compatible /v1 accepts reasoning_effort (high/medium/low/max/none) and reasoning.effort. Reasoning returns in message.thinking on the native API — not reasoning_content. Cloud embeddings are unverified: /api/embed may not be authorized for cloud keys.",
          "docs": "https://docs.ollama.com/api/openai-compatibility"
        },
        {
          "text": "Usage levels (Low, Medium, High, Extra High) weight billing — no per-Mtok pricing. Full catalog at ollama.com/search?c=cloud (also gemma4, nemotron-3 family, mistral-large-3 — the only non-thinking cloud model).",
          "docs": "https://ollama.com/cloud"
        },
        {
          "text": "Model IDs dropped the -cloud suffix (verified 2026-08-19); DeepSeek and Qwen entries carry snapshot/variant suffixes (:0731/:0813/:397b/:preview) and full-catalog drift is visible in the daily sync.",
          "docs": "https://ollama.com/search?c=cloud"
        }
      ],
      "plan": {
        "price_usd": 20,
        "period": "monthly",
        "quota": "Free: 1 concurrent model; Pro $20: 3; Max $100: 10. Sessions reset every 5 hours plus weekly limits; usage weighted by model cost level; extra balance purchasable on Pro/Max.",
        "notes": "Zero data retention; no training on prompts; US-primary hosting.",
        "docs": "https://ollama.com/cloud"
      }
    },
    {
      "id": "openai",
      "name": "OpenAI",
      "kind": "first_party",
      "urls": {
        "docs": "https://developers.openai.com/api/reference/overview/",
        "console": "https://platform.openai.com",
        "status": "https://status.openai.com"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "OPENAI_API_KEY"
          ],
          "key_prefix": "sk-",
          "extra_headers": {},
          "getting_credentials": "platform.openai.com → API Keys. Project keys (sk-proj-) scope to a project; send OpenAI-Organization / OpenAI-Project when a key spans orgs/projects.",
          "docs": "https://developers.openai.com/api/reference/overview/"
        },
        {
          "id": "chatgpt-oauth",
          "type": "oauth",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [],
          "extra_headers": {},
          "getting_credentials": "codex login (or --device-auth). Tokens persist in ~/.codex/auth.json (or OS keyring); refresh tokens rotate and are effectively single-use. Billed against ChatGPT Plus/Pro/Business plans (5-hour + weekly windows), not API credits.",
          "docs": "https://learn.chatgpt.com/docs/auth",
          "flow": "authorization_code_pkce or device code at auth.openai.com (public client; first-party clients only)",
          "scopes": [
            "openid",
            "profile",
            "email",
            "offline_access"
          ]
        }
      ],
      "endpoints": [
        {
          "id": "v1-chat-completions",
          "base_url": "https://api.openai.com",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "v1-responses",
          "base_url": "https://api.openai.com",
          "path": "/v1/responses",
          "protocol": "openai-responses"
        },
        {
          "id": "codex-backend",
          "base_url": "https://chatgpt.com",
          "path": "/backend-api/codex/responses",
          "protocol": "openai-responses",
          "auth": "chatgpt-oauth"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings",
        "files",
        "batch",
        "fine_tuning",
        "realtime",
        "image_gen",
        "audio"
      ],
      "quirks": [
        {
          "text": "Assistants API sunsets 2026-08-26; migrate to Responses. Reasoning models require max_completion_tokens (chat) instead of max_tokens.",
          "docs": "https://developers.openai.com/api/docs/guides/migrate-to-responses"
        },
        {
          "text": "ChatGPT-plan requests go to chatgpt.com/backend-api/codex (not api.openai.com) with Bearer tokens plus ChatGPT-Account-ID and originator headers; OpenAI allow-lists first-party originators server-side — third-party clients mimicking Codex work but are officially unsupported (ban reports exist). Enterprise access tokens (CODEX_ACCESS_TOKEN) are the supported programmatic path.",
          "docs": "https://learn.chatgpt.com/docs/auth"
        },
        {
          "text": "Subscription quotas: rolling 5-hour windows shared by local messages and cloud tasks plus weekly caps (Plus/Business: gpt-5.6-luna 50-280 local messages/5h; Pro 5x-20x that; flexible-pricing enterprises scale with credits — credits per 1M tokens: Sol 125/750 in/out, Terra 50/300, Luna 5/30). GPT-5.4/5.4-mini retire from ChatGPT-plan access 2026-08-31.",
          "docs": "https://chatgpt.com/codex/pricing/"
        }
      ]
    },
    {
      "id": "opencode-go",
      "name": "OpenCode Go",
      "kind": "subscription",
      "urls": {
        "docs": "https://opencode.ai/docs/go/",
        "console": "https://opencode.ai/go"
      },
      "auth": [
        {
          "id": "plan-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "OPENCODE_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Subscribe at opencode.ai/go; uses the OpenCode Zen API key. Provider IDs are opencode-go/<model>.",
          "docs": "https://opencode.ai/docs/go/"
        }
      ],
      "endpoints": [
        {
          "id": "go-responses",
          "base_url": "https://opencode.ai",
          "path": "/zen/go/v1/responses",
          "protocol": "openai-responses"
        },
        {
          "id": "go-chat-completions",
          "base_url": "https://opencode.ai",
          "path": "/zen/go/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "go-messages",
          "base_url": "https://opencode.ai",
          "path": "/zen/go/v1/messages",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Per-model-family protocol routing: /responses for Grok/GPT, /chat/completions for GLM/Kimi/DeepSeek, /messages for MiniMax/Qwen. Offerings verified against the live GET /zen/go/v1/models list (2026-08-19).",
          "docs": "https://opencode.ai/docs/go/"
        }
      ],
      "plan": {
        "price_usd": 10,
        "period": "monthly",
        "quota": "$12 per 5 hours, $30/week, $60/month",
        "notes": "$5 first month then $10/mo, on top of the Zen console (same API key mechanism); open coding models only — Grok 4.5, GPT 5.6 Luna, GLM-5.x, Kimi K3/K2.x, MiniMax M3/M2.7, Qwen3.x, DeepSeek V4, MiMo, Hy3; optional Zen-balance fallback. Routes per protocol: /responses for Grok/GPT, /chat/completions for GLM/Kimi/DeepSeek, /messages (Anthropic) for MiniMax/Qwen.",
        "docs": "https://opencode.ai/docs/go/"
      }
    },
    {
      "id": "opencode-zen",
      "name": "OpenCode Zen",
      "kind": "aggregator",
      "urls": {
        "docs": "https://opencode.ai/docs/zen/",
        "console": "https://opencode.ai/auth"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "OPENCODE_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create a key at opencode.ai/auth (in the TUI: /connect). ~90+ curated models billed at cost, pay-as-you-go.",
          "docs": "https://opencode.ai/docs/providers/"
        }
      ],
      "endpoints": [
        {
          "id": "chat-completions",
          "base_url": "https://opencode.ai",
          "path": "/zen/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "responses",
          "base_url": "https://opencode.ai",
          "path": "/zen/v1/responses",
          "protocol": "openai-responses"
        },
        {
          "id": "messages",
          "base_url": "https://opencode.ai",
          "path": "/zen/v1/messages",
          "protocol": "anthropic-messages"
        },
        {
          "id": "gemini-style",
          "base_url": "https://opencode.ai",
          "path": "/zen/v1/models/{model-id}",
          "protocol": "google-generate-content"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Four protocol surfaces; models are assigned per surface (Responses for GPT/Grok, Anthropic Messages for Claude/Qwen, chat-completions for DeepSeek/MiniMax/GLM/Kimi, Gemini-style for Google models). No documented reasoning normalization — controls follow each upstream provider's wire format per surface; community-verified: thinking {type: enabled} and reasoning_effort (low/high/max) pass through for DeepSeek V4 on the chat surface. Offerings verified against the live GET /zen/v1/models list (2026-08-19).",
          "docs": "https://opencode.ai/docs/zen/"
        }
      ]
    },
    {
      "id": "openrouter",
      "name": "OpenRouter",
      "kind": "aggregator",
      "urls": {
        "docs": "https://openrouter.ai/docs/api-reference/overview",
        "console": "https://openrouter.ai/credits",
        "pricing": "https://openrouter.ai/models"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "OPENROUTER_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "openrouter.ai → Keys. Prepaid credits (5.5% purchase fee); optional HTTP-Referer and X-Title attribution headers.",
          "docs": "https://openrouter.ai/docs/api-reference/overview"
        }
      ],
      "endpoints": [
        {
          "id": "chat-completions",
          "base_url": "https://openrouter.ai",
          "path": "/api/v1/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Unified reasoning object: {effort | max_tokens, exclude, enabled}. Effort converts to a token budget where needed (Anthropic clamped min 1024 / max 128000). reasoning.exclude strips reasoning from responses. Routing: provider.order/allow_fallbacks/sort/require_parameters/max_price; :nitro and :floor model suffixes.",
          "docs": "https://openrouter.ai/docs/use-cases/reasoning-tokens"
        }
      ]
    },
    {
      "id": "poolside",
      "name": "poolside",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.poolside.ai",
        "console": "https://platform.poolside.ai"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "POOLSIDE_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create a key at platform.poolside.ai.",
          "docs": "https://docs.poolside.ai"
        }
      ],
      "endpoints": [
        {
          "id": "chat-completions",
          "base_url": "https://inference.poolside.ai",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Free for a limited time; model ids prefixed poolside/.",
          "docs": "https://docs.poolside.ai"
        }
      ]
    },
    {
      "id": "qwen-coding-plan",
      "name": "Qwen Coding Plan",
      "kind": "subscription",
      "urls": {
        "docs": "https://www.alibabacloud.com/help/en/model-studio/coding-plan",
        "console": "https://modelstudio.console.alibabacloud.com"
      },
      "auth": [
        {
          "id": "plan-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "BAILIAN_CODING_PLAN_API_KEY"
          ],
          "key_prefix": "sk-sp-",
          "extra_headers": {},
          "getting_credentials": "Subscribe in Model Studio; plan keys (sk-sp-...) are separate from normal DASHSCOPE_API_KEY and only work on coding-plan endpoints. Configure tools via OPENAI_BASE_URL=https://coding-intl.dashscope.aliyuncs.com/v1 or the Anthropic-compatible /apps/anthropic route. The older Qwen OAuth free tier was discontinued 2026-04-15.",
          "docs": "https://qwenlm.github.io/qwen-code-docs/en/users/configuration/auth/"
        }
      ],
      "endpoints": [
        {
          "id": "coding-chat",
          "base_url": "https://coding-intl.dashscope.aliyuncs.com",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "coding-anthropic",
          "base_url": "https://coding-intl.dashscope.aliyuncs.com",
          "path": "/apps/anthropic",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "China base: https://coding.dashscope.aliyuncs.com/v1. Third-party models on the allowlist (glm-5, glm-4.7, kimi-k2.5, MiniMax-M2.5) are served through the same endpoints; their reasoning parameters follow the upstream models' conventions — offerings seeded only where wire behavior is verified.",
          "docs": "https://www.alibabacloud.com/help/en/model-studio/coding-plan"
        }
      ],
      "plan": {
        "price_usd": 50,
        "period": "monthly",
        "quota": "Pro $50/mo: 6,000 requests / 5 rolling hours, 45,000/week, 90,000/month; slots restock daily 00:00 UTC+8",
        "notes": "Strict model allowlist: qwen3.7-plus, qwen3.6-plus, qwen3.5-plus, qwen3-coder-next, qwen3-coder-plus, glm-5, glm-4.7, kimi-k2.5, MiniMax-M2.5. Interactive coding-tool use only. Lite discontinued: new subscriptions ended 2026-03-20, renewals ended 2026-04-13. No Max or Team tier exists on the Coding Plan (Token Plan Team is a separate product).",
        "docs": "https://www.alibabacloud.com/help/en/model-studio/coding-plan"
      }
    },
    {
      "id": "stepfun",
      "name": "StepFun",
      "kind": "first_party",
      "urls": {
        "docs": "https://platform.stepfun.ai",
        "console": "https://platform.stepfun.com"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "STEPFUN_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create a key on the StepFun platform console (EN platform stepfun.ai, CN stepfun.com).",
          "docs": "https://platform.stepfun.ai"
        }
      ],
      "endpoints": [
        {
          "id": "chat-completions",
          "base_url": "https://api.stepfun.ai",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "EN platform stepfun.ai, CN stepfun.com; step-3 (321B/38B VLM, Jul 2025) is the prior generation.",
          "docs": "https://platform.stepfun.ai"
        },
        {
          "text": "Parallel Thinking (PaCoRe) reasoning on 3.5-flash; 3.7-flash has three reasoning levels + Advisor Mode.",
          "docs": "https://static.stepfun.com/blog/step-3.7-flash/"
        }
      ]
    },
    {
      "id": "synthetic",
      "name": "Synthetic",
      "kind": "subscription",
      "urls": {
        "docs": "https://dev.synthetic.new/",
        "console": "https://synthetic.new",
        "pricing": "https://synthetic.new/pricing"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "SYNTHETIC_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Subscribe at synthetic.new and create an API key in the dashboard; send as Authorization: Bearer SYNTHETIC_API_KEY.",
          "docs": "https://dev.synthetic.new/"
        }
      ],
      "endpoints": [
        {
          "id": "openai-chat",
          "base_url": "https://api.synthetic.new",
          "path": "/openai/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "anthropic",
          "base_url": "https://api.synthetic.new",
          "path": "/anthropic/v1/messages",
          "protocol": "anthropic-messages"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming",
        "embeddings"
      ],
      "quirks": [
        {
          "text": "Prefer syn: routing aliases (syn:large:text etc.) — pinned model names 404 as models rotate.",
          "docs": "https://dev.synthetic.new/docs/api/models"
        },
        {
          "text": "Reasoning controls are not documented; behavior follows upstream chat templates — verify before relying.",
          "docs": "https://dev.synthetic.new/docs/api/models"
        }
      ],
      "plan": {
        "price_usd": 30,
        "period": "monthly",
        "quota": "$1/day or $30/month per pack, stackable; 500 price-weighted requests / 5h + $24/week credits + 1 concurrent per model",
        "notes": "Subscription includes all always-on models, no per-token billing; embeddings free and exempt.",
        "docs": "https://synthetic.new/rate-limits"
      }
    },
    {
      "id": "xai",
      "name": "xAI",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.x.ai/overview",
        "console": "https://console.x.ai"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "XAI_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create a key at console.x.ai. OpenAI SDKs work by changing base_url only; also Anthropic-compatible.",
          "docs": "https://docs.x.ai/overview"
        },
        {
          "id": "oauth",
          "type": "oauth",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [],
          "extra_headers": {},
          "getting_credentials": "grok login (or grok login --device-auth) with a SuperGrok or X Premium subscription. Tokens persist in ~/.grok/auth.json with background refresh; requests use subscription quota (a weekly pool shared across Grok chat, Build, and API), not API billing.",
          "docs": "https://docs.x.ai/build/overview",
          "flow": "browser OIDC or device code at auth.x.ai (endpoints not publicly documented)"
        }
      ],
      "endpoints": [
        {
          "id": "v1-chat-completions",
          "base_url": "https://api.x.ai",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "v1-responses",
          "base_url": "https://api.x.ai",
          "path": "/v1/responses",
          "protocol": "openai-responses"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Responses is the preferred API per xAI docs. presence_penalty, frequency_penalty, and stop cannot be used with reasoning models — requests including them error. grok-4.20-multi-agent maps reasoning.effort to agent count (4 or 16), not depth.",
          "docs": "https://docs.x.ai/docs/guides/reasoning"
        },
        {
          "text": "OAuth sessions call https://api.x.ai/v1 with Bearer tokens; an entitlement-aware catalog is available at https://cli-chat-proxy.grok.com/v1/models-v2. xAI decides which accounts receive OAuth tokens — some non-Heavy tiers report 403. SuperGrok is $30/mo, SuperGrok Plus $100/mo.",
          "docs": "https://docs.x.ai/build/enterprise"
        }
      ]
    },
    {
      "id": "xiaomi-mimo",
      "name": "Xiaomi MiMo",
      "kind": "first_party",
      "urls": {
        "docs": "https://mimo.mi.com/docs/en-US/quickstart",
        "console": "https://platform.xiaomimimo.com"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "MIMO_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "Create a key at platform.xiaomimimo.com; the api-key: header is also accepted.",
          "docs": "https://mimo.mi.com/docs/en-US/quickstart"
        }
      ],
      "endpoints": [
        {
          "id": "chat-completions",
          "base_url": "https://api.xiaomimimo.com",
          "path": "/v1/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "messages",
          "base_url": "https://api.xiaomimimo.com",
          "path": "/v1/messages",
          "protocol": "anthropic-messages"
        },
        {
          "id": "responses",
          "base_url": "https://api.xiaomimimo.com",
          "path": "/v1/responses",
          "protocol": "openai-responses"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Auth accepts api-key: header or Authorization: Bearer.",
          "docs": "https://mimo.mi.com/docs/en-US/quickstart"
        },
        {
          "text": "top_p ignored in thinking mode; reasoning returns in reasoning_content and must be passed back in tool loops.",
          "docs": "https://mimo.mi.com/docs/en-US/deep-thinking"
        },
        {
          "text": "Token Plan subscriptions also offered; V2 series retired 2026-06-30; ultraspeed FP4 variant early access.",
          "docs": "https://mimo.mi.com/docs/en-US/quickstart"
        }
      ]
    },
    {
      "id": "zai",
      "name": "Z.ai",
      "kind": "first_party",
      "urls": {
        "docs": "https://docs.z.ai/guides/llm/glm-5.3",
        "console": "https://z.ai/manage-apikey/apikey-list",
        "pricing": "https://docs.z.ai/guides/develop/http/introduction"
      },
      "auth": [
        {
          "id": "api-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "ZAI_API_KEY"
          ],
          "extra_headers": {},
          "getting_credentials": "z.ai console → API Keys. Key format id.secret; JWT HS256 derivation also supported.",
          "docs": "https://docs.z.ai/guides/develop/http/introduction"
        }
      ],
      "endpoints": [
        {
          "id": "chat-completions",
          "base_url": "https://api.z.ai",
          "path": "/api/paas/v4/chat/completions",
          "protocol": "openai-chat"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "clear_thinking: false preserves reasoning_content across turns (Preserved Thinking); disabled by default on this endpoint.",
          "docs": "https://docs.z.ai/guides/capabilities/thinking-mode"
        },
        {
          "text": "GLM-5.3-Flash is the first multimodal GLM-5-series model: chat-completions accepts image/video/file inputs alongside text.",
          "docs": "https://docs.z.ai/guides/vlm/glm-5.3-flash"
        }
      ]
    },
    {
      "id": "zai-coding-plan",
      "name": "Z.ai GLM Coding Plan",
      "kind": "subscription",
      "urls": {
        "docs": "https://docs.z.ai/devpack/overview",
        "console": "https://z.ai/subscribe"
      },
      "auth": [
        {
          "id": "plan-key",
          "type": "api_key",
          "transport": "header",
          "header": "Authorization: Bearer",
          "env": [
            "ZAI_CODING_PLAN_API_KEY"
          ],
          "key_prefix": "",
          "extra_headers": {},
          "getting_credentials": "Subscribe at z.ai/subscribe; plan keys are distinct from normal API keys and only work on coding-plan endpoints.",
          "docs": "https://docs.z.ai/devpack/quick-start"
        }
      ],
      "endpoints": [
        {
          "id": "coding-chat-completions",
          "base_url": "https://api.z.ai",
          "path": "/api/coding/paas/v4/chat/completions",
          "protocol": "openai-chat"
        },
        {
          "id": "coding-anthropic",
          "base_url": "https://api.z.ai",
          "path": "/api/anthropic",
          "protocol": "anthropic-messages"
        },
        {
          "id": "coding-responses",
          "base_url": "https://api.z.ai",
          "path": "/api/v1",
          "protocol": "openai-responses"
        }
      ],
      "api_surfaces": [
        "text",
        "streaming"
      ],
      "quirks": [
        {
          "text": "Preserved Thinking is enabled by default on coding-plan endpoints (opposite of the standard API). For Claude Code: ANTHROPIC_BASE_URL=https://api.z.ai/api/anthropic, ANTHROPIC_AUTH_TOKEN=<plan key>.",
          "docs": "https://docs.z.ai/devpack/quick-start"
        }
      ],
      "plan": {
        "price_usd": 18,
        "period": "monthly",
        "quota": "Lite ~2000 credits/5h, 10000/week; Pro ~12000/5h; Max ~28000/5h",
        "notes": "Credits-based since 2026-07-30. Includes GLM-5.3, GLM-5-Turbo, GLM-4.7; GLM-5.2/5.1 auto-route to GLM-5.3.",
        "docs": "https://docs.z.ai/devpack/overview"
      }
    }
  ]
}