{
  "generated_at": "2026-09-04T10:48:22+02:00",
  "research_model": "gpt-5.6-terra",
  "providers": [
    {
      "id": "anthropic",
      "name": "Anthropic",
      "notes": "Die Basispreise gelten pro 1 Mio. Tokens; Prompt Caching und Batch-Verarbeitung haben abweichende Tarife. Claude Pro und Max sind Abonnements für Claude-Apps, keine separaten API-Modellvarianten. Bei Legacy-Modellen sind gemäß Vorgabe nur die jeweils unmittelbar abgelösten Vorgänger erfasst.",
      "pricing_url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "docs_url": "https://platform.claude.com/docs/en/models/overview",
      "researched_at": "2026-09-04T10:44:38+0200",
      "usage": {
        "input_tokens": 109443,
        "output_tokens": 8549,
        "web_searches": 11
      },
      "model_count": 9,
      "live_checked": true,
      "live_error": null
    },
    {
      "id": "xai",
      "name": "xAI (Grok)",
      "notes": "Stand 2026-09-04 führt die offizielle xAI-Preisseite sieben gehostete Text-/Bild-zu-Text-Sprachmodelle auf. Die Standardpreise gelten unter 200.000 Prompt-Tokens; ab diesem Schwellenwert gilt für sämtliche Tokens eines Requests der jeweilige Long-Context-Tarif. Priority Processing kostet laut Preisseite das Doppelte des Standardtarifs; Batch ist nur bei den entsprechend gekennzeichneten Modellen verfügbar und dort um 20 % reduziert.",
      "pricing_url": "https://docs.x.ai/developers/pricing",
      "docs_url": "https://docs.x.ai/developers/models",
      "researched_at": "2026-09-04T10:44:45+0200",
      "usage": {
        "input_tokens": 101147,
        "output_tokens": 8575,
        "web_searches": 11
      },
      "model_count": 8,
      "live_checked": false,
      "live_error": null
    },
    {
      "id": "openai",
      "name": "OpenAI",
      "notes": "Stand: 2026-09-04. Die reguläre API-Produktlinie besteht aus GPT-6 Astra (limitiert ausgerollt), der GPT-5.6-Familie sowie weiterhin buchbaren GPT-5.5-/GPT-5.4-Vorgängern und GPT-5.3-Codex. Batch und Flex sind bei den aktuellen GPT-5.6-Modellen günstiger, Fast Mode teurer; lange Kontexte können einen Aufschlag auslösen.",
      "pricing_url": "https://platform.openai.com/docs/pricing",
      "docs_url": "https://platform.openai.com/docs/models",
      "researched_at": "2026-09-04T10:44:48+0200",
      "usage": {
        "input_tokens": 100820,
        "output_tokens": 8924,
        "web_searches": 10
      },
      "model_count": 14,
      "live_checked": true,
      "live_error": null
    },
    {
      "id": "mistral",
      "name": "Mistral AI",
      "notes": "Stand: 2026-09-04. Standardpreise gelten für die globale Serverless-API; Batch-Verarbeitung halbiert laut Anbieter die Tokenpreise, regionale Inferenz kann für unterstützte Modelle 10 % Aufschlag haben. Die fünf als „legacy“ geführten Vorgänger sind nach der dokumentierten sechsmonatigen Deprecation-Frist am Stichtag noch erreichbar; Labs-Vorgänger mit nur einem Monat Frist wurden nicht aufgenommen.",
      "pricing_url": "https://mistral.ai/pricing/api/",
      "docs_url": "https://docs.mistral.ai/models",
      "researched_at": "2026-09-04T10:44:56+0200",
      "usage": {
        "input_tokens": 71094,
        "output_tokens": 10155,
        "web_searches": 7
      },
      "model_count": 15,
      "live_checked": false,
      "live_error": null
    },
    {
      "id": "google",
      "name": "Google (Gemini)",
      "notes": "Google bietet über die Gemini API einen kostenlosen Einstieg sowie kostenpflichtige Standard-, Batch-, Flex- und Priority-Inference an. Batch und Flex sind bei den meisten Tokenmodellen günstiger als Standard; Priority ist teurer. Preise können nach Kontextlänge oder Eingabemodalität variieren und sind je Modell in den Preisnotizen zusammengefasst.",
      "pricing_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "docs_url": "https://ai.google.dev/gemini-api/docs/models",
      "researched_at": "2026-09-04T10:45:13+0200",
      "usage": {
        "input_tokens": 116812,
        "output_tokens": 12240,
        "web_searches": 12
      },
      "model_count": 18,
      "live_checked": true,
      "live_error": null
    },
    {
      "id": "meta",
      "name": "Meta (Llama)",
      "notes": "Die frühere Llama API wurde durch die Meta Model API abgelöst; diese befindet sich am 4. September 2026 weiterhin im Public Preview. Muse Spark 1.3 ist das aktuelle proprietäre API-Flaggschiff; die günstigere „Contributor“-Route ist derselbe Modellcheckpoint, erlaubt Meta jedoch die Nutzung von Ein- und Ausgaben zur Produktverbesserung. Llama 4 und Muse Glimmer sind Open-Weights-Angebote ohne direkte, von Meta betriebene Token-API.",
      "pricing_url": "https://dev.meta.ai/docs/models",
      "docs_url": "https://dev.meta.ai/docs/models",
      "researched_at": "2026-09-04T10:45:31+0200",
      "usage": {
        "input_tokens": 104870,
        "output_tokens": 9101,
        "web_searches": 23
      },
      "model_count": 6,
      "live_checked": false,
      "live_error": null
    },
    {
      "id": "deepseek",
      "name": "DeepSeek",
      "notes": "Stand 2026-09-04 führt DeepSeek für seine API drei Chatmodell-IDs; die früheren IDs `deepseek-chat` und `deepseek-reasoner` wurden am 2026-07-24 um 15:59 UTC eingestellt und sind daher keine Legacy-Einträge mehr. Die unten als Standardpreise verwendeten Peak-Tarife gelten werktags 01:00–04:00 sowie 06:00–10:00 UTC; alle übrigen Zeiten werden mit 50 % Rabatt abgerechnet.",
      "pricing_url": "https://api-docs.deepseek.com/quick_start/pricing/",
      "docs_url": "https://api-docs.deepseek.com/",
      "researched_at": "2026-09-04T10:45:45+0200",
      "usage": {
        "input_tokens": 58608,
        "output_tokens": 5269,
        "web_searches": 7
      },
      "model_count": 3,
      "live_checked": true,
      "live_error": null
    },
    {
      "id": "moonshot",
      "name": "Moonshot AI (Kimi)",
      "notes": "Moonshot AI bietet auf der Kimi API Platform nutzungsbasierte Abrechnung pro Token; die angegebenen Preise verstehen sich ohne anfallende Steuern. Automatisches Context Caching senkt den Preis für Cache-Hits. Der Batch-Modus kostet 60 % des Echtzeit-Tarifs, unterstützt aber nur kimi-k2.7-code und kimi-k2.6, nicht K3.",
      "pricing_url": "https://platform.kimi.ai/docs/pricing/chat",
      "docs_url": "https://platform.kimi.ai/docs/models",
      "researched_at": "2026-09-04T10:46:20+0200",
      "usage": {
        "input_tokens": 113193,
        "output_tokens": 5254,
        "web_searches": 16
      },
      "model_count": 4,
      "live_checked": true,
      "live_error": null
    },
    {
      "id": "minimax",
      "name": "MiniMax",
      "notes": "MiniMax bietet aktuell ein 1M-kontextiges, multimodales Flaggschiff (M3), die weiterhin regulär bepreisten M2.7-Varianten sowie ältere M2.5-Varianten an. Die API-Standardpreise gelten pro 1 Mio. Tokens; M3 wird oberhalb von 512k Input-Tokens teurer, und der Priority-Service kostet 1,5× des Standardtarifs. M2-her ist weiterhin per API dokumentiert, steht jedoch nicht mehr mit einem separaten Tarif in der aktuellen Pay-as-you-go-Tabelle.",
      "pricing_url": "https://platform.minimax.io/docs/guides/pricing-paygo",
      "docs_url": "https://platform.minimax.io/docs/guides/text-generation",
      "researched_at": "2026-09-04T10:46:56+0200",
      "usage": {
        "input_tokens": 118423,
        "output_tokens": 5313,
        "web_searches": 14
      },
      "model_count": 6,
      "live_checked": false,
      "live_error": null
    },
    {
      "id": "cohere",
      "name": "Cohere",
      "notes": "Stand 4. September 2026. Cohere rechnet reguläre API-Modelle grundsätzlich tokenbasiert ab; mehrere neuere Modelle sind über die Standard-API zunächst nur rate-limitiert kostenlos nutzbar und erfordern für produktive Nutzung Model Vault bzw. ein Sales-Angebot. Datierte Cohere-Modellversionen wurden gemäß Vorgabe zu undatierten Familien-IDs normalisiert; die konkreten, aktuell dokumentierten Snapshots stehen in den Quellseiten.",
      "pricing_url": "https://cohere.com/pricing",
      "docs_url": "https://docs.cohere.com/docs/models",
      "researched_at": "2026-09-04T10:47:09+0200",
      "usage": {
        "input_tokens": 109170,
        "output_tokens": 10228,
        "web_searches": 12
      },
      "model_count": 17,
      "live_checked": false,
      "live_error": null
    },
    {
      "id": "amazon",
      "name": "Amazon (Nova)",
      "notes": "Amazon Nova wird über Amazon Bedrock nutzungsbasiert pro Token abgerechnet; die Tarifseite unterscheidet je nach Modell zwischen Standard-, Flex-, Priority- und Batch-Inferenz. Nova 2 ist die aktuelle Generation; die weiterhin API-verfügbaren Nova-1-Modelle sind hier als „legacy“ eingeordnet. Nicht enthalten sind Sonic (Live-Sprache), Embeddings, Canvas/Reel sowie Nova Act.",
      "pricing_url": "https://aws.amazon.com/bedrock/pricing/",
      "docs_url": "https://docs.aws.amazon.com/nova/latest/nova2-userguide/what-is-nova-2.html",
      "researched_at": "2026-09-04T10:47:17+0200",
      "usage": {
        "input_tokens": 122893,
        "output_tokens": 8613,
        "web_searches": 16
      },
      "model_count": 7,
      "live_checked": false,
      "live_error": null
    },
    {
      "id": "zai",
      "name": "Z.ai / Zhipu (GLM)",
      "notes": "Stand 2026-09-04. Z.AI rechnet API-Nutzung pro 1 Mio. Tokens ab; bei den kostenpflichtigen Chatmodellen ist Context-Cache-Speicherung derzeit zeitlich begrenzt kostenlos. GLM-5.3-Flash läuft aktuell zu einem bis 2026-09-09 (UTC+8) befristet um 50 % reduzierten Tarif; die unten stehenden Preise sind die aktuell berechneten Preise. Die noch angebotenen GLM-4.5-/4.5V-Modelle wurden als weiter zurückliegende Historie nicht aufgenommen; enthalten ist jeweils nur die letzte abgelöste GLM-4.6-/4.6V-Generation.",
      "pricing_url": "https://docs.z.ai/guides/overview/pricing",
      "docs_url": "https://docs.z.ai/api-reference/llm/chat-completion",
      "researched_at": "2026-09-04T10:47:22+0200",
      "usage": {
        "input_tokens": 112968,
        "output_tokens": 9856,
        "web_searches": 13
      },
      "model_count": 12,
      "live_checked": false,
      "live_error": null
    },
    {
      "id": "perplexity",
      "name": "Perplexity (Sonar)",
      "notes": "Die Sonar API ist in der aktuellen Perplexity-Dokumentation als „Legacy API“ eingeordnet, ihre vier dokumentierten Modelle sind jedoch weiterhin über den Sonar-Chat-Completion-Endpunkt nutzbar und auf der Preisseite ausgewiesen. Abrechnung erfolgt nutzungsbasiert ohne Abonnement; neben Tokenpreisen fallen bei Sonar, Sonar Pro und Sonar Reasoning Pro kontextabhängige Request-Gebühren an.",
      "pricing_url": "https://docs.perplexity.ai/docs/getting-started/pricing",
      "docs_url": "https://docs.perplexity.ai/docs/sonar/models",
      "researched_at": "2026-09-04T10:47:23+0200",
      "usage": {
        "input_tokens": 77254,
        "output_tokens": 4652,
        "web_searches": 8
      },
      "model_count": 4,
      "live_checked": false,
      "live_error": null
    },
    {
      "id": "alibaba",
      "name": "Alibaba (Qwen)",
      "notes": "Abgerechnet wird bei Alibaba Cloud Model Studio standardmäßig nutzungsbasiert pro Token; die Preise unterscheiden sich je nach Region, Eingabelänge, Thinking-Modus und Modalität. Die untenstehenden Preisfelder geben, soweit verfügbar, den regulären Pay-as-you-go-Starttarif für Global/US beziehungsweise International an; Staffel-, Cache-, Batch- und Modalitätsabweichungen stehen in price_notes. Berücksichtigt sind Qwen-Text- und multimodale Chat-/Understanding-Modelle, nicht jedoch reine Generierungs-, Speech-, Embedding-, Moderations-, Realtime- oder Robotik-Modelle.",
      "pricing_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "docs_url": "https://www.alibabacloud.com/help/en/model-studio/models",
      "researched_at": "2026-09-04T10:48:19+0200",
      "usage": {
        "input_tokens": 111323,
        "output_tokens": 22446,
        "web_searches": 14
      },
      "model_count": 57,
      "live_checked": false,
      "live_error": null
    }
  ],
  "models": [
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.8-Max",
      "api_id": "qwen3.8-max",
      "tier": "flagship",
      "status": "ga",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 2,
      "cached_input_price": null,
      "output_price": 6,
      "price_notes": "Internationaler Standardtarif. Context Caching wird für Eingabetokens rabattiert; in Beijing ist Batch-Datei-Eingabe günstiger.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Aktuelles multimodales Flaggschiff für anspruchsvolle Agenten-, Coding-, Dokument- und visuelle Reasoning-Aufgaben.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen3-8-max",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.8-2.4T-A95B",
      "api_id": "qwen3.8-2.4t-a95b",
      "tier": "flagship",
      "status": "ga",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 2,
      "cached_input_price": null,
      "output_price": 6,
      "price_notes": "Regulärer Preis der Qwen3.8-Max-Klasse; Context Caching wird rabattiert.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Direkt adressierbare großskalige Qwen3.8-MoE-Variante für Flaggschiff-Reasoning und multimodale Agentenaufgaben.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen-api-via-openai-responses",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.5-Omni-Plus",
      "api_id": "qwen3.5-omni-plus",
      "tier": "flagship",
      "status": "ga",
      "release_date": null,
      "context_window": 64000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.4,
      "cached_input_price": null,
      "output_price": 8.3,
      "price_notes": "Internationaler Standard für Text-/Bild-/Video-Input und Textausgabe. Audio-Input kostet 11 USD; bei Audio-only-Ausgabe beträgt der Outputpreis 44 USD pro Million Tokens.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video"
      ],
      "modalities_output": [
        "text",
        "audio"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Aktuelles Omni-Flaggschiff für gemeinsame Text-, Bild-, Audio- und Videoanalyse mit Text- oder Sprachausgabe.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/omni",
        "https://www.alibabacloud.com/help/en/model-studio/qwen-omni",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.8-27B",
      "api_id": "qwen3.8-27b",
      "tier": "standard",
      "status": "ga",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 0.5,
      "cached_input_price": 0.1,
      "output_price": 3,
      "price_notes": "Internationaler Standardtarif. In Beijing beträgt der Standardtarif 0,424 USD Input und 1,696 USD Output; explizites und implizites Caching ist verfügbar.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Dichtes natives Vision-Language-Modell für Coding, Büroaufgaben und zuverlässige multimodale Ausführung.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen3-8-27b",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.7-Plus",
      "api_id": "qwen3.7-plus",
      "tier": "standard",
      "status": "ga",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 0.276,
      "cached_input_price": null,
      "output_price": 1.101,
      "price_notes": "Globaler/US-Standardtarif für Eingaben bis 256K; bei 256K bis 1M steigen Input/Output auf 0,826/3,301 USD. Thinking und Non-Thinking haben denselben Outputpreis; zeitlich begrenzte Tages-/Nachtrabatte sowie Cache-Rabatte können gelten.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Ausgewogenes multimodales Agentenmodell mit 1M Kontext für Chat, Coding, Produktivität und lange Dokumente.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen3-7-plus-us",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.5-Omni-Flash",
      "api_id": "qwen3.5-omni-flash",
      "tier": "small",
      "status": "ga",
      "release_date": null,
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.4,
      "cached_input_price": null,
      "output_price": 2.2,
      "price_notes": "Internationaler Standard für Text-/Bild-/Video-Input und Textausgabe. Audio-Input kostet 3 USD; bei Audio-only-Ausgabe beträgt der Outputpreis 11,9 USD pro Million Tokens.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video"
      ],
      "modalities_output": [
        "text",
        "audio"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Schnellere und günstigere aktuelle Omni-Variante für multimodale Analyse und Sprach- oder Textantworten.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/omni",
        "https://www.alibabacloud.com/help/en/model-studio/qwen-omni",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.8-Flash",
      "api_id": "qwen3.8-flash",
      "tier": "small",
      "status": "ga",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 0.15,
      "cached_input_price": null,
      "output_price": 0.47,
      "price_notes": "Internationaler Standardtarif; Context Caching reduziert Eingabekosten. Die Preise in Beijing sind niedriger und Batch-Inferenz ist dort rabattiert.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Preisgünstiges multimodales 1M-Kontext-Modell für schnelle Chat-, Agenten- und Dokumentaufgaben.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen3-8-flash",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.7-Flash",
      "api_id": "qwen3.7-flash",
      "tier": "small",
      "status": "ga",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 0.03,
      "cached_input_price": null,
      "output_price": 0.13,
      "price_notes": "Internationaler Tarif für Eingaben bis 32K. Bei 32K bis 256K gelten 0,10/0,40 USD, bei 256K bis 1M 0,20/0,80 USD; Batch ist um 50 % rabattiert und Caching reduziert Eingabekosten.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Kosteneffizientes multimodales 1M-Kontext-Modell für schnelle agentische und visuelle Anwendungen.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen3-7-flash",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.5-OCR",
      "api_id": "qwen3.5-ocr",
      "tier": "specialized",
      "status": "ga",
      "release_date": "2026-06-16",
      "context_window": 65536,
      "max_output_tokens": 16384,
      "knowledge_cutoff": null,
      "input_price": 0.069,
      "cached_input_price": null,
      "output_price": 0.275,
      "price_notes": "Öffentlich dokumentierter Tarif für China (Beijing); die Modellseite weist keine internationalen Tokenpreise aus.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Aktuelles spezialisiertes OCR- und Dokumentverständnismodell für Extraktion, Textlokalisierung und mehrturnige Bilddialoge.",
      "source_urls": [
        "https://docs.modelstudio.console.alibabacloud.com/en/model-studio/qwen3-5-ocr",
        "https://www.alibabacloud.com/help/en/model-studio/qwen-vl-ocr-api-reference",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.6-Max Preview",
      "api_id": "qwen3.6-max-preview",
      "tier": "flagship",
      "status": "preview",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.3,
      "cached_input_price": null,
      "output_price": 7.8,
      "price_notes": "Internationaler Tarif bis 128K Input; bei 128K bis 256K steigen Input/Output auf 2/12 USD. Context Caching wird rabattiert.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Noch verfügbare Vorschau des früheren Qwen3.6-Flaggschiffs mit umschaltbarem Thinking-Modus.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.7-Max Preview",
      "api_id": "qwen3.7-max-preview",
      "tier": "reasoning",
      "status": "preview",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 2.5,
      "cached_input_price": null,
      "output_price": 7.5,
      "price_notes": "Internationaler Thinking-only-Preview-Tarif; diese ID entspricht dem Snapshot vom 2026-05-17.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Vorschauversion des Qwen3.7-Max für reines textbasiertes Thinking und allgemeine Konversationsaufgaben.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen3-7-max",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.7-Max",
      "api_id": "qwen3.7-max",
      "tier": "flagship",
      "status": "legacy",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 2.5,
      "cached_input_price": null,
      "output_price": 7.5,
      "price_notes": "Internationaler Standardtarif; die undatierte ID zeigt laut Dokumentation auf qwen3.7-max-2026-05-20. Context Caching reduziert Eingabekosten.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Noch buchbares früheres Qwen3.7-Flaggschiff mit textbasierter Agenten-, Coding- und Reasoning-Ausrichtung.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen3-7-max",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-235B-A22B",
      "api_id": "qwen3-235b-a22b",
      "tier": "flagship",
      "status": "legacy",
      "release_date": "2025-05-12",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.287,
      "cached_input_price": null,
      "output_price": 1.147,
      "price_notes": "Globaler/US-Non-Thinking-Tarif; im Thinking-Modus bleibt der Input bei 0,287 USD, der Output beträgt 2,868 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-MoE-Modell der Qwen3-Generation mit umschaltbarem Thinking für starke allgemeine Aufgaben.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-235B-A22B-Instruct",
      "api_id": "qwen3-vl-235b-a22b-instruct",
      "tier": "flagship",
      "status": "legacy",
      "release_date": null,
      "context_window": 131072,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0.287,
      "cached_input_price": null,
      "output_price": 1.147,
      "price_notes": "Globaler/US-Non-Thinking-Tarif.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Instruct-Flaggschiff für leistungsstarkes multimodales Verstehen ohne Thinking-only-Betrieb.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3-VL",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "QVQ-Max",
      "api_id": "qvq-max",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": null,
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.2,
      "cached_input_price": null,
      "output_price": 4.8,
      "price_notes": "Internationaler Standardtarif für multimodale Tokenabrechnung.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Noch verfügbares multimodales Vision-Reasoning-Modell für anspruchsvolle Bildverständnisaufgaben.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/vision-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "QwQ-Plus",
      "api_id": "qwq-plus",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": null,
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.8,
      "cached_input_price": null,
      "output_price": 2.4,
      "price_notes": "Internationaler Standardtarif; in Beijing beträgt er 0,230/0,574 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Noch buchbares spezialisiertes Text-Reasoning-Modell der QwQ-Reihe.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-235B-A22B-Thinking",
      "api_id": "qwen3-vl-235b-a22b-thinking",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": null,
      "context_window": 131072,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0.287,
      "cached_input_price": null,
      "output_price": 2.868,
      "price_notes": "Globaler/US-Thinking-only-Tarif. In Singapore beträgt der internationale Preis 0,4/4 USD.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Vision-Language-Flaggschiff für multimodales STEM-, Mathematik- und visuelles Reasoning.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3-VL",
        "https://docs.modelstudio.console.alibabacloud.com/en/model-studio/qwen3-vl-235b-a22b-thinking",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "QVQ-Plus",
      "api_id": "qvq-plus",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": null,
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.287,
      "cached_input_price": null,
      "output_price": 0.717,
      "price_notes": "Öffentlich dokumentierter Tarif für China (Beijing); kein internationaler Standardtarif ist auf der aktuellen Preisseite ausgewiesen.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Kostengünstigere noch verfügbare QVQ-Variante für visuelles Reasoning.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/vision-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-235B-A22B-Thinking-2507",
      "api_id": "qwen3-235b-a22b-thinking-2507",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": "2025-07",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.23,
      "cached_input_price": null,
      "output_price": 2.3,
      "price_notes": "Internationaler Thinking-only-Tarif. Global/US beträgt 0,23 USD Input und 2,3 USD Output.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Reasoning-Variante des 235B-A22B-Modells, bei der Thinking nicht deaktiviert werden kann.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/deep-thinking",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-32B-Thinking",
      "api_id": "qwen3-vl-32b-thinking",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": null,
      "context_window": 131072,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0.16,
      "cached_input_price": null,
      "output_price": 0.64,
      "price_notes": "Globaler/US-Thinking-only-Tarif.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-32B-Vision-Language-Modell für ausschließlich multimodales Reasoning.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3-VL",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-Next-80B-A3B-Thinking",
      "api_id": "qwen3-next-80b-a3b-thinking",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.144,
      "cached_input_price": null,
      "output_price": 1.434,
      "price_notes": "Globaler/US-Thinking-only-Tarif; International beträgt er 0,15/1,2 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-MoE-Modell mittlerer Größe für ausschließlich schrittweises Reasoning.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-30B-A3B-Thinking-2507",
      "api_id": "qwen3-30b-a3b-thinking-2507",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": "2025-07",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.108,
      "cached_input_price": null,
      "output_price": 1.076,
      "price_notes": "Globaler/US-Thinking-only-Tarif; International beträgt er 0,20/2,4 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Thinking-only-Version des 30B-A3B-MoE-Modells.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/deep-thinking",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-30B-A3B-Thinking",
      "api_id": "qwen3-vl-30b-a3b-thinking",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": null,
      "context_window": 131072,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0.108,
      "cached_input_price": null,
      "output_price": 1.076,
      "price_notes": "Globaler/US-Thinking-only-Tarif.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-MoE-Vision-Language-Modell für effizientes multimodales Reasoning.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3-VL",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-8B-Thinking",
      "api_id": "qwen3-vl-8b-thinking",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": null,
      "context_window": 131072,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0.072,
      "cached_input_price": null,
      "output_price": 0.717,
      "price_notes": "Globaler/US-Thinking-only-Tarif; International beträgt er 0,18/2,1 USD.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Kompaktes Open-Weights-Vision-Language-Modell für multimodales Thinking.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3-VL",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.6-27B",
      "api_id": "qwen3.6-27b",
      "tier": "standard",
      "status": "legacy",
      "release_date": null,
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.6,
      "cached_input_price": null,
      "output_price": 3.6,
      "price_notes": "Internationaler Standardtarif laut Preistabelle; in Beijing 0,412564/2,475384 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Noch angebotene dichte Qwen3.6-Variante für textbasierte allgemeine und Reasoning-Aufgaben.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.6-35B-A3B",
      "api_id": "qwen3.6-35b-a3b",
      "tier": "standard",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 0.375,
      "cached_input_price": null,
      "output_price": 2.25,
      "price_notes": "Internationaler Tarif bis 256K. Global/US liegt bei 0,248 USD Input und 1,485 USD Output.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Vorherige multimodale MoE-Variante mittlerer Größe für visuelle und textbasierte Agentenaufgaben.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen3-6-35b-a3b",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.6-Plus",
      "api_id": "qwen3.6-plus",
      "tier": "standard",
      "status": "legacy",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 0.276,
      "cached_input_price": null,
      "output_price": 1.651,
      "price_notes": "Globaler/US-Tarif bis 256K Input; oberhalb davon bis 1M gelten 1,101/6,602 USD. Batch-Dateien sind rabattiert; Caching-Optionen unterscheiden sich nach Region.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Letzte vorherige Plus-Generation für multimodale Chat-, Agenten- und Langkontextaufgaben.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen3-6-plus",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-235B-A22B-Instruct-2507",
      "api_id": "qwen3-235b-a22b-instruct-2507",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2025-07",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.23,
      "cached_input_price": null,
      "output_price": 0.92,
      "price_notes": "Internationaler Non-Thinking-Tarif; Global/US entspricht 0,23/0,92 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Instruct-Variante des 235B-A22B-Modells für direkte Antworten ohne Thinking-Modus.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/newly-released-models",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-32B",
      "api_id": "qwen3-32b",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2025-05-12",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.16,
      "cached_input_price": null,
      "output_price": 0.64,
      "price_notes": "Globaler/US-Non-Thinking-Tarif. Im Thinking-Modus ist der Outputpreis ebenfalls 0,64 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Dense-Modell mittlerer Größe mit hybridem Thinking für allgemeine Aufgaben.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/newly-released-models",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-32B-Instruct",
      "api_id": "qwen3-vl-32b-instruct",
      "tier": "standard",
      "status": "legacy",
      "release_date": null,
      "context_window": 131072,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0.16,
      "cached_input_price": null,
      "output_price": 0.64,
      "price_notes": "Globaler/US-Non-Thinking-Tarif.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-32B-Vision-Language-Instruct-Modell für Bild-, Video- und Textverständnis.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3-VL",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-Next-80B-A3B-Instruct",
      "api_id": "qwen3-next-80b-a3b-instruct",
      "tier": "standard",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.144,
      "cached_input_price": null,
      "output_price": 0.574,
      "price_notes": "Globaler/US-Non-Thinking-Tarif; International beträgt er 0,15/1,2 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Instruct-Variante des kompakten Qwen3-Next-MoE-Modells.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-14B",
      "api_id": "qwen3-14b",
      "tier": "standard",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.144,
      "cached_input_price": null,
      "output_price": 0.574,
      "price_notes": "Globaler/US-Non-Thinking-Tarif; Thinking-Output kostet 1,434 USD pro Million Tokens.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Dense-Modell für kompaktere allgemeine Chat- und Reasoning-Workloads.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-Plus",
      "api_id": "qwen3-vl-plus",
      "tier": "standard",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.143,
      "cached_input_price": null,
      "output_price": 1.434,
      "price_notes": "Globaler/US-Tarif bis 32K Input; bei 32K bis 128K gelten 0,215/2,15 USD, bei 128K bis 256K 0,43/4,301 USD. Context Caching reduziert Eingabekosten.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Noch buchbare vorherige kommerzielle Vision-Language-Plus-Variante mit hybridem Thinking.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/vision-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-30B-A3B",
      "api_id": "qwen3-30b-a3b",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2025-05-12",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.108,
      "cached_input_price": null,
      "output_price": 0.431,
      "price_notes": "Globaler/US-Non-Thinking-Tarif; im Thinking-Modus beträgt der Outputpreis 1,076 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-MoE-Modell mit hybridem Thinking für gute Effizienz bei allgemeinen Aufgaben.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/newly-released-models",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-30B-A3B-Instruct-2507",
      "api_id": "qwen3-30b-a3b-instruct-2507",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2025-07",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.108,
      "cached_input_price": null,
      "output_price": 0.431,
      "price_notes": "Globaler/US-Non-Thinking-Tarif; International beträgt er 0,20/0,80 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Instruct-Version des 30B-A3B-MoE-Modells für direkte Antworten.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-30B-A3B-Instruct",
      "api_id": "qwen3-vl-30b-a3b-instruct",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2025-10-03",
      "context_window": 131072,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0.108,
      "cached_input_price": null,
      "output_price": 0.431,
      "price_notes": "Globaler/US-Non-Thinking-Tarif.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-MoE-Instruct-Modell für lange multimodale Dokument- und Videoaufgaben.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3-VL",
        "https://docs.modelstudio.console.alibabacloud.com/en/model-studio/newly-released-models",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-Coder-480B-A35B-Instruct",
      "api_id": "qwen3-coder-480b-a35b-instruct",
      "tier": "coding",
      "status": "legacy",
      "release_date": "2025-07-23",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.861,
      "cached_input_price": null,
      "output_price": 3.441,
      "price_notes": "Globaler/US-Tarif bis 32K Input; bei 32K bis 128K gelten 1,291/5,161 USD, bei 128K bis 200K 2,151/8,602 USD. Der internationale Preis ist höher.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Coding-MoE-Modell mit hoher Coding-Agent-Leistung für große Softwareaufgaben.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/newly-released-models",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "https://www.alibabacloud.com/help/en/model-studio/model-telemetry"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-Coder-Plus",
      "api_id": "qwen3-coder-plus",
      "tier": "coding",
      "status": "legacy",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.574,
      "cached_input_price": null,
      "output_price": 2.294,
      "price_notes": "Globaler/US-Tarif bis 32K Input; bei 32K bis 128K gelten 0,861/3,441 USD, bei 128K bis 256K 1,434/5,735 USD und bei 256K bis 1M 2,868/28,671 USD. Context Caching reduziert Eingabekosten.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Vorheriges leistungsstarkes Coding-Agent-Modell mit 1M Kontext und Tool Calling.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-Coder-Next",
      "api_id": "qwen3-coder-next",
      "tier": "coding",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.3,
      "cached_input_price": null,
      "output_price": 1.5,
      "price_notes": "Internationaler Tarif bis 32K Input; weitere Stufen sind 0,5/2,5 USD bis 128K und 0,8/4 USD bis 256K. In Beijing gelten niedrigere Preise.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Kompaktes vorheriges Coding-Modell für agentisches Programmieren und Tool Calling.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-Coder-30B-A3B-Instruct",
      "api_id": "qwen3-coder-30b-a3b-instruct",
      "tier": "coding",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.216,
      "cached_input_price": null,
      "output_price": 0.861,
      "price_notes": "Globaler/US-Tarif bis 32K Input; bei 32K bis 128K gelten 0,323/1,291 USD und bei 128K bis 200K 0,538/2,151 USD.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Kompakteres Open-Weights-Coding-MoE-Modell für Codegenerierung und Tool-gestützte Entwicklung.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-Coder-Flash",
      "api_id": "qwen3-coder-flash",
      "tier": "coding",
      "status": "legacy",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.144,
      "cached_input_price": null,
      "output_price": 0.574,
      "price_notes": "Globaler/US-Tarif bis 32K Input; weitere Stufen sind 0,216/0,861 USD bis 128K, 0,359/1,434 USD bis 256K und 0,717/3,584 USD bis 1M. Context Caching reduziert Eingabekosten.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Schnellere und preiswertere vorherige Coding-Variante mit 1M Kontext.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-Omni-Flash",
      "api_id": "qwen3-omni-flash",
      "tier": "small",
      "status": "legacy",
      "release_date": null,
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.43,
      "cached_input_price": null,
      "output_price": 1.66,
      "price_notes": "Internationaler Text-only-Standard. Der Preis variiert nach Text-, Audio- und Bild/Video-Input sowie Text- oder Audioausgabe; bei multimodalem Input beträgt der Textoutput 3,06 USD und Audiooutput 15,11 USD.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video"
      ],
      "modalities_output": [
        "text",
        "audio"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Noch buchbare vorherige leichte Omni-Generation mit Thinking und multimodalem Ein- und Ausgabeumfang.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/omni",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3.6-Flash",
      "api_id": "qwen3.6-flash",
      "tier": "small",
      "status": "legacy",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 0.25,
      "cached_input_price": null,
      "output_price": 1.5,
      "price_notes": "Internationaler Tarif bis 256K Input; bei 256K bis 1M gelten 1/4 USD. Batch-Inferenz ist um 50 % rabattiert und Context Caching reduziert Eingabekosten.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Letzte vorherige schnelle multimodale Qwen3.6-Variante für preisbewusste Anwendungen.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-8B",
      "api_id": "qwen3-8b",
      "tier": "small",
      "status": "legacy",
      "release_date": "2025-05-12",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.072,
      "cached_input_price": null,
      "output_price": 0.287,
      "price_notes": "Globaler/US-Non-Thinking-Tarif; Thinking-Output kostet 0,717 USD. Für einige Aufrufarten ist Thinking laut Dokumentation nur mit Streaming nutzbar.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Kleinmodell mit hybridem Thinking für latenz- und kostenbewusste Textanwendungen.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/newly-released-models",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-8B-Instruct",
      "api_id": "qwen3-vl-8b-instruct",
      "tier": "small",
      "status": "legacy",
      "release_date": null,
      "context_window": 131072,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0.072,
      "cached_input_price": null,
      "output_price": 0.287,
      "price_notes": "Globaler/US-Non-Thinking-Tarif; International beträgt er 0,18/0,70 USD.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming",
        "fine_tuning"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Kompaktes Open-Weights-Vision-Language-Instruct-Modell für lange Videos, Dokumente und räumliche Aufgaben.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3-VL",
        "https://docs.modelstudio.console.alibabacloud.com/en/model-studio/qwen3-vl-8b-instruct",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-Flash",
      "api_id": "qwen3-vl-flash",
      "tier": "small",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.022,
      "cached_input_price": null,
      "output_price": 0.215,
      "price_notes": "Globaler/US-Tarif bis 32K Input; bei 32K bis 128K gelten 0,043/0,43 USD und bei 128K bis 256K 0,086/0,859 USD. Context Caching reduziert Eingabekosten.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Schnelle vorherige kommerzielle Vision-Language-Variante für kostengünstiges Bild- und Videoverständnis.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/vision-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-4B",
      "api_id": "qwen3-4b",
      "tier": "small",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Für die direkt nutzbare Modell-ID nennt die aktuelle öffentliche Pay-as-you-go-Preisseite keinen separaten Tokenpreis.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Open-Weights-Kleinmodell der Qwen3-Reihe für lokale oder kostensensitive allgemeine Textaufgaben.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-1.7B",
      "api_id": "qwen3-1.7b",
      "tier": "small",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Für die direkt nutzbare Modell-ID nennt die aktuelle öffentliche Pay-as-you-go-Preisseite keinen separaten Tokenpreis.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Sehr kleines Open-Weights-Qwen3-Modell für ressourcensparende Text- und Chat-Workloads.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-0.6B",
      "api_id": "qwen3-0.6b",
      "tier": "small",
      "status": "legacy",
      "release_date": null,
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Für die direkt nutzbare Modell-ID nennt die aktuelle öffentliche Pay-as-you-go-Preisseite keinen separaten Tokenpreis.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Kleinstes Open-Weights-Modell der Qwen3-Reihe für sehr ressourcenarme Textaufgaben.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3",
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-4B-Instruct",
      "api_id": "qwen3-vl-4b-instruct",
      "tier": "small",
      "status": "legacy",
      "release_date": null,
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Für diese Modell-ID weist die aktuelle öffentliche Pay-as-you-go-Preisseite keinen separaten Tokenpreis aus; sie ist als Modell für Bereitstellung/Fine-Tuning dokumentiert.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming",
        "fine_tuning"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Kleines Open-Weights-Vision-Language-Instruct-Modell für ressourcenbewusstes multimodales Verstehen.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3-VL",
        "https://www.alibabacloud.com/help/en/model-studio/model-training-and-deployment-billing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen3-VL-2B-Instruct",
      "api_id": "qwen3-vl-2b-instruct",
      "tier": "small",
      "status": "legacy",
      "release_date": null,
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Für diese Modell-ID weist die aktuelle öffentliche Pay-as-you-go-Preisseite keinen separaten Tokenpreis aus; sie ist als Modell für Bereitstellung dokumentiert.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Sehr kleines Open-Weights-Vision-Language-Instruct-Modell für einfache multimodale Anwendungen.",
      "source_urls": [
        "https://github.com/QwenLM/Qwen3-VL",
        "https://www.alibabacloud.com/help/en/model-studio/model-training-and-deployment-billing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen-Math-Plus",
      "api_id": "qwen-math-plus",
      "tier": "specialized",
      "status": "legacy",
      "release_date": null,
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.574,
      "cached_input_price": null,
      "output_price": 1.721,
      "price_notes": "Öffentlich dokumentierter Tarif für China (Beijing); qwen-math-plus-latest wird nicht als separates Modell gezählt.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Spezialisiertes noch buchbares Textmodell für mathematische Aufgaben und schrittweises Problemlösen.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen-Plus-Character-JA",
      "api_id": "qwen-plus-character-ja",
      "tier": "specialized",
      "status": "legacy",
      "release_date": null,
      "context_window": 32000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.5,
      "cached_input_price": null,
      "output_price": 1.4,
      "price_notes": "Internationaler, in Singapore ausgewiesener Standardtarif; Session Cache wird rabattiert.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Japanisch ausgerichtete Variante des Charakter- und Rollenspiel-Chatmodells.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen-Plus-Character",
      "api_id": "qwen-plus-character",
      "tier": "specialized",
      "status": "legacy",
      "release_date": null,
      "context_window": 32000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.115,
      "cached_input_price": null,
      "output_price": 0.287,
      "price_notes": "Globaler/US-Standardtarif; international in Singapore 0,5/1,4 USD. Session Cache wird rabattiert.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Spezialisiertes Chatmodell für Rollen-, Charakter- und dialogorientierte Interaktionen.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen-Long",
      "api_id": "qwen-long",
      "tier": "specialized",
      "status": "legacy",
      "release_date": null,
      "context_window": 10000000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.072,
      "cached_input_price": null,
      "output_price": 0.287,
      "price_notes": "Öffentlich dokumentierter Tarif für China (Beijing). qwen-long-latest verweist auf die aktuelle Version und wird hier nicht als separates Modell gezählt.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "structured_output",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Spezialmodell mit 10M Kontext für die Verarbeitung sehr langer Dokumente und umfangreicher Textsammlungen.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen-VL-OCR",
      "api_id": "qwen-vl-ocr",
      "tier": "specialized",
      "status": "legacy",
      "release_date": null,
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.043,
      "cached_input_price": null,
      "output_price": 0.072,
      "price_notes": "Globaler/US-Standardtarif; die undatierte ID entspricht laut Preisseite dem Snapshot vom 2025-11-20.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Noch buchbares vorheriges spezialisiertes OCR-Modell für Dokumente, Tabellen und Text aus Bildern.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/qwen-vl-ocr-api-reference",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "alibaba",
      "provider": "Alibaba (Qwen)",
      "name": "Qwen-Flash-Character",
      "api_id": "qwen-flash-character",
      "tier": "specialized",
      "status": "legacy",
      "release_date": null,
      "context_window": 8000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.034,
      "cached_input_price": null,
      "output_price": 0.203,
      "price_notes": "Globaler/US-Standardtarif; international in Singapore 0,05/0,4 USD. Session Cache wird rabattiert.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Schnelle, kostengünstige Charakter-Chatvariante für kurze dialogorientierte Interaktionen.",
      "source_urls": [
        "https://www.alibabacloud.com/help/en/model-studio/text-generation-model",
        "https://www.alibabacloud.com/help/en/model-studio/model-pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "amazon",
      "provider": "Amazon (Nova)",
      "name": "Amazon Nova 2 Lite",
      "api_id": "amazon.nova-2-lite-v1:0",
      "tier": "standard",
      "status": "ga",
      "release_date": "2025-12-02",
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "knowledge_cutoff": "2025-10",
      "input_price": 0.3,
      "cached_input_price": 0.075,
      "output_price": 2.5,
      "price_notes": "Standardtarif für Text-Tokens. Batch-Inferenz wird laut Bedrock-Preisseite mit 50 % Rabatt gegenüber On-Demand berechnet; Flex und Priority sind als weitere Service-Tiers verfügbar. Reasoning-Tokens zählen als Output-Tokens.",
      "modalities_input": [
        "text",
        "image",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "web_search",
        "code_execution",
        "fine_tuning",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Kosteneffizientes multimodales Reasoning-Modell für Automatisierung, Dokumentenverarbeitung, Kundenservice und agentische Aufgaben.",
      "source_urls": [
        "https://aws.amazon.com/bedrock/pricing/",
        "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-2-lite.html",
        "https://aws.amazon.com/about-aws/whats-new/2025/12/nova-2-foundation-models-amazon-bedrock/",
        "https://docs.aws.amazon.com/nova/latest/nova2-userguide/whats-new.html"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "amazon",
      "provider": "Amazon (Nova)",
      "name": "Amazon Nova 2 Pro",
      "api_id": "amazon.nova-pro-v2:0",
      "tier": "flagship",
      "status": "preview",
      "release_date": "2025-12-02",
      "context_window": 1000000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Für die zugangsbeschränkte Preview veröffentlicht AWS keinen allgemein verfügbaren Token-Standardtarif auf der Bedrock-Preisseite. Der Early Access steht Nova-Forge-Kunden über den AWS-Account-Support offen.",
      "modalities_input": [
        "text",
        "image",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "fine_tuning",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Intelligentestes Nova-Modell für komplexe mehrstufige Reasoning-, Planungs-, Videoanalyse- und Softwaremigrationsaufgaben.",
      "source_urls": [
        "https://aws.amazon.com/about-aws/whats-new/2025/12/nova-2-foundation-models-amazon-bedrock/",
        "https://aws.amazon.com/nova/models/",
        "https://docs.aws.amazon.com/general/latest/gr/bedrock.html",
        "https://aws.amazon.com/bedrock/pricing/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "amazon",
      "provider": "Amazon (Nova)",
      "name": "Amazon Nova 2 Omni",
      "api_id": "amazon.nova-2-omni-v1:0",
      "tier": "specialized",
      "status": "preview",
      "release_date": "2025-12-02",
      "context_window": 1000000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.3,
      "cached_input_price": 0.075,
      "output_price": 2.5,
      "price_notes": "Standardtarif für Text-Eingabe und Text-Ausgabe; die Bildausgabe wird zusätzlich in einer separaten, nicht tokenbasierten Preisdimension geführt und ist daher nicht in output_price enthalten. Der Zugriff ist als Nova-Forge-Early-Access-Preview eingeschränkt.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video"
      ],
      "modalities_output": [
        "text",
        "image"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "All-in-one-Previewmodell für multimodales Reasoning mit Text-, Bild-, Video- und Spracheingaben sowie Text- und Bildausgaben.",
      "source_urls": [
        "https://aws.amazon.com/about-aws/whats-new/2025/12/amazon-nova-2-omni-preview/",
        "https://docs.aws.amazon.com/nova/latest/nova2-userguide/request-response-schema.html",
        "https://aws.amazon.com/bedrock/pricing/",
        "https://builder.aws.com/content/3B2n4BYrroKX8KFkn1rF7Y2LegQ/brio-building-a-unified-creator-workspace-with-amazon-nova-on-amazon-bedrock"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "amazon",
      "provider": "Amazon (Nova)",
      "name": "Amazon Nova Premier",
      "api_id": "amazon.nova-premier-v1:0",
      "tier": "flagship",
      "status": "legacy",
      "release_date": "2025-10-31",
      "context_window": 1000000,
      "max_output_tokens": 25000,
      "knowledge_cutoff": "2024-10",
      "input_price": 2.5,
      "cached_input_price": 0.25,
      "output_price": 12.5,
      "price_notes": "Standardtarif für On-Demand-Tokeninferenz. Batch-Inferenz ist mit 50 % Rabatt gegenüber On-Demand verfügbar; Flex und Priority werden ebenfalls unterstützt. AWS kennzeichnet dieses Modell als Legacy und nennt als EOL-Datum den 14. September 2026.",
      "modalities_input": [
        "text",
        "image",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Legacy-Flaggschiff für komplexes multimodales Reasoning, agentische Workflows und Modelldestillation.",
      "source_urls": [
        "https://aws.amazon.com/bedrock/pricing/",
        "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-premier.html",
        "https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "amazon",
      "provider": "Amazon (Nova)",
      "name": "Amazon Nova Pro",
      "api_id": "amazon.nova-pro-v1:0",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2024-12-05",
      "context_window": 300000,
      "max_output_tokens": 5000,
      "knowledge_cutoff": "2024-10",
      "input_price": 0.8,
      "cached_input_price": 0.08,
      "output_price": 3.2,
      "price_notes": "Standardtarif für On-Demand-Tokeninferenz. Batch-Inferenz kostet 50 % weniger als On-Demand; Flex und Priority werden unterstützt. Prompt-Cache-Lesezugriffe erhalten gegenüber regulären Eingabetokens einen Rabatt.",
      "modalities_input": [
        "text",
        "image",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "fine_tuning",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Legacy-Allroundmodell mit ausgewogenem Verhältnis von Qualität, Geschwindigkeit und Kosten für multimodale und Code-Aufgaben.",
      "source_urls": [
        "https://aws.amazon.com/bedrock/pricing/",
        "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-pro.html",
        "https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "amazon",
      "provider": "Amazon (Nova)",
      "name": "Amazon Nova Lite",
      "api_id": "amazon.nova-lite-v1:0",
      "tier": "small",
      "status": "legacy",
      "release_date": "2024-12-05",
      "context_window": 300000,
      "max_output_tokens": 5000,
      "knowledge_cutoff": "2024-10",
      "input_price": 0.06,
      "cached_input_price": 0.006,
      "output_price": 0.24,
      "price_notes": "Standardtarif für On-Demand-Tokeninferenz. Batch-Inferenz kostet 50 % weniger als On-Demand. Prompt-Cache-Lesezugriffe erhalten gegenüber regulären Eingabetokens einen Rabatt.",
      "modalities_input": [
        "text",
        "image",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "fine_tuning",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Legacy-Kleinmodell für schnelle und günstige multimodale Verarbeitung von Text, Bildern und Video.",
      "source_urls": [
        "https://aws.amazon.com/bedrock/pricing/",
        "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-lite.html",
        "https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "amazon",
      "provider": "Amazon (Nova)",
      "name": "Amazon Nova Micro",
      "api_id": "amazon.nova-micro-v1:0",
      "tier": "small",
      "status": "legacy",
      "release_date": "2024-12-05",
      "context_window": 128000,
      "max_output_tokens": 5000,
      "knowledge_cutoff": "2024-10",
      "input_price": 0.035,
      "cached_input_price": 0.0035,
      "output_price": 0.14,
      "price_notes": "Standardtarif für On-Demand-Tokeninferenz. Batch-Inferenz kostet 50 % weniger als On-Demand. Prompt-Cache-Lesezugriffe erhalten gegenüber regulären Eingabetokens einen Rabatt.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "fine_tuning",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Legacy-Textmodell mit besonders niedriger Latenz und Kosten für Klassifikation, Extraktion, Übersetzung und Zusammenfassungen.",
      "source_urls": [
        "https://aws.amazon.com/bedrock/pricing/",
        "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-micro.html",
        "https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "anthropic",
      "provider": "Anthropic",
      "name": "Claude Fable 5.1",
      "api_id": "claude-fable-5-1",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-09-01",
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-06",
      "input_price": 10,
      "cached_input_price": 0.25,
      "output_price": 50,
      "price_notes": "Prompt-Cache-Schreiben: 12,50 USD/MTok (5 Min.) bzw. 20 USD/MTok (1 Std.); Cache-Hits kosten 0,25 USD/MTok. Die Batch API reduziert Ein- und Ausgabetokens um 50 %; bei US-Inferenzgeografie gilt auf unterstützten Plattformen ein 1,1-facher Aufschlag.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Anthropics leistungsstärkstes breit verfügbares Modell für anspruchsvolles Reasoning sowie lang laufende agentische Coding-, Recherche- und Wissensarbeit.",
      "source_urls": [
        "https://platform.claude.com/docs/en/models/fable-5-1/overview",
        "https://platform.claude.com/docs/en/about-claude/pricing",
        "https://platform.claude.com/docs/en/build-with-claude/structured-outputs"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "anthropic",
      "provider": "Anthropic",
      "name": "Claude Sonnet 5",
      "api_id": "claude-sonnet-5",
      "tier": "standard",
      "status": "ga",
      "release_date": "2026-06-30",
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-01",
      "input_price": 2,
      "cached_input_price": 0.2,
      "output_price": 10,
      "price_notes": "Prompt-Cache-Schreiben: 2,50 USD/MTok (5 Min.) bzw. 4 USD/MTok (1 Std.); Cache-Hits kosten 0,20 USD/MTok. Die Batch API reduziert Ein- und Ausgabetokens um 50 %; die zum 1. September 2026 geplante Preiserhöhung auf 3/15 USD/MTok wurde nicht umgesetzt.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Ausgewogenes Standardmodell für schnelle, intelligente Coding-, Agenten-, Analyse- und Unternehmensworkloads.",
      "source_urls": [
        "https://platform.claude.com/docs/en/models/sonnet-5/overview",
        "https://platform.claude.com/docs/en/about-claude/pricing",
        "https://platform.claude.com/docs/en/build-with-claude/structured-outputs"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "anthropic",
      "provider": "Anthropic",
      "name": "Claude Opus 5",
      "api_id": "claude-opus-5",
      "tier": "coding",
      "status": "ga",
      "release_date": "2026-07-24",
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-05",
      "input_price": 5,
      "cached_input_price": 0.5,
      "output_price": 25,
      "price_notes": "Prompt-Cache-Schreiben: 6,25 USD/MTok (5 Min.) bzw. 10 USD/MTok (1 Std.); Cache-Hits kosten 0,50 USD/MTok. Batch API bietet 50 % Rabatt; Fast Mode ist als Research Preview auf der First-Party-API verfügbar und kostet 10 USD Input bzw. 50 USD Output pro MTok; im Batch-Modus ist Fast Mode nicht verfügbar.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Leistungsstarkes Opus-Modell für komplexes agentisches Coding, Enterprise-Aufgaben und tiefes adaptives Reasoning.",
      "source_urls": [
        "https://platform.claude.com/docs/en/models/opus-5/overview",
        "https://platform.claude.com/docs/en/about-claude/pricing",
        "https://platform.claude.com/docs/en/build-with-claude/fast-mode"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "anthropic",
      "provider": "Anthropic",
      "name": "Claude Haiku 4.5",
      "api_id": "claude-haiku-4-5",
      "tier": "small",
      "status": "ga",
      "release_date": "2025-10-15",
      "context_window": 200000,
      "max_output_tokens": 64000,
      "knowledge_cutoff": "2025-02",
      "input_price": 1,
      "cached_input_price": 0.1,
      "output_price": 5,
      "price_notes": "Die undatierte API-ID ist ein Alias für den aktuellen festen Snapshot. Prompt-Cache-Schreiben: 1,25 USD/MTok (5 Min.) bzw. 2 USD/MTok (1 Std.); Cache-Hits kosten 0,10 USD/MTok; Batch API bietet 50 % Rabatt.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Kleines, besonders schnelles und kostengünstiges Modell für hohe Volumina mit manuell steuerbarem Extended Thinking.",
      "source_urls": [
        "https://platform.claude.com/docs/en/models/haiku-4-5/overview",
        "https://platform.claude.com/docs/en/about-claude/models/model-ids-and-versions",
        "https://platform.claude.com/docs/en/about-claude/pricing"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "anthropic",
      "provider": "Anthropic",
      "name": "Claude Mythos 5.1",
      "api_id": "claude-mythos-5-1",
      "tier": "specialized",
      "status": "ga",
      "release_date": "2026-09-01",
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-06",
      "input_price": 10,
      "cached_input_price": 0.25,
      "output_price": 50,
      "price_notes": "Nur auf Einladung für Project-Glasswing-Teilnehmende verfügbar. Prompt-Cache-Schreiben: 12,50 USD/MTok (5 Min.) bzw. 20 USD/MTok (1 Std.); Cache-Hits kosten 0,25 USD/MTok; Batch API mit 50 % Rabatt auf Input und Output.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Zugangsbeschränkte Project-Glasswing-Variante von Fable 5.1 mit gleichen Spezifikationen für spezialisierte Sicherheitsforschung.",
      "source_urls": [
        "https://platform.claude.com/docs/en/models/mythos-5-1/overview",
        "https://platform.claude.com/docs/en/about-claude/pricing",
        "https://platform.claude.com/docs/en/build-with-claude/structured-outputs"
      ],
      "verified_in_api": false
    },
    {
      "provider_id": "anthropic",
      "provider": "Anthropic",
      "name": "Claude Mythos 5",
      "api_id": "claude-mythos-5",
      "tier": "specialized",
      "status": "ga",
      "release_date": "2026-06-09",
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-01",
      "input_price": 10,
      "cached_input_price": 1,
      "output_price": 50,
      "price_notes": "Nur auf Einladung im Rahmen von Project Glasswing verfügbar und inzwischen durch Mythos 5.1 ergänzt. Prompt-Cache-Schreiben: 12,50 USD/MTok (5 Min.) bzw. 20 USD/MTok (1 Std.); Cache-Hits kosten 1 USD/MTok; Batch API bietet 50 % Rabatt.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Zugangsbeschränktes Spezialmodell für defensive Cybersicherheits- und Biologieforschung ohne die Safety-Classifier von Fable 5.",
      "source_urls": [
        "https://platform.claude.com/docs/en/models/mythos-5/overview",
        "https://platform.claude.com/docs/en/about-claude/pricing",
        "https://platform.claude.com/docs/en/build-with-claude/structured-outputs"
      ],
      "verified_in_api": false
    },
    {
      "provider_id": "anthropic",
      "provider": "Anthropic",
      "name": "Claude Fable 5",
      "api_id": "claude-fable-5",
      "tier": "flagship",
      "status": "legacy",
      "release_date": "2026-06-09",
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-01",
      "input_price": 10,
      "cached_input_price": 1,
      "output_price": 50,
      "price_notes": "Weiter verfügbar, aber von Fable 5.1 abgelöst. Prompt-Cache-Schreiben: 12,50 USD/MTok (5 Min.) bzw. 20 USD/MTok (1 Std.); Cache-Hits kosten 1 USD/MTok; Batch API bietet 50 % Rabatt.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Älteres Flaggschiff für lang laufende agentische Aufgaben und anspruchsvolles Reasoning mit stets aktiviertem adaptivem Thinking.",
      "source_urls": [
        "https://platform.claude.com/docs/en/models/fable-5/overview",
        "https://platform.claude.com/docs/en/about-claude/pricing",
        "https://platform.claude.com/docs/en/build-with-claude/structured-outputs"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "anthropic",
      "provider": "Anthropic",
      "name": "Claude Sonnet 4.6",
      "api_id": "claude-sonnet-4-6",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2026-02-17",
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2025-08",
      "input_price": 3,
      "cached_input_price": 0.3,
      "output_price": 15,
      "price_notes": "Weiter verfügbar, aber von Sonnet 5 abgelöst. Prompt-Cache-Schreiben: 3,75 USD/MTok (5 Min.) bzw. 6 USD/MTok (1 Std.); Cache-Hits kosten 0,30 USD/MTok; Batch API bietet 50 % Rabatt.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Unmittelbarer Sonnet-Vorgänger für ausgewogene Unternehmens-, Coding- und Agentenaufgaben mit adaptivem Thinking.",
      "source_urls": [
        "https://platform.claude.com/docs/en/models/sonnet-4-6/overview",
        "https://platform.claude.com/docs/en/about-claude/pricing",
        "https://platform.claude.com/docs/en/build-with-claude/structured-outputs"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "anthropic",
      "provider": "Anthropic",
      "name": "Claude Opus 4.8",
      "api_id": "claude-opus-4-8",
      "tier": "coding",
      "status": "legacy",
      "release_date": "2026-05-28",
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-01",
      "input_price": 5,
      "cached_input_price": 0.5,
      "output_price": 25,
      "price_notes": "Weiter verfügbar, aber von Opus 5 abgelöst. Prompt-Cache-Schreiben: 6,25 USD/MTok (5 Min.) bzw. 10 USD/MTok (1 Std.); Cache-Hits kosten 0,50 USD/MTok; Batch API bietet 50 % Rabatt. Fast Mode ist als Research Preview auf der First-Party-API für 10 USD Input bzw. 50 USD Output pro MTok verfügbar und nicht mit Batch kombinierbar.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Unmittelbarer Opus-Vorgänger für anspruchsvolles agentisches Coding und adaptives Reasoning.",
      "source_urls": [
        "https://platform.claude.com/docs/en/models/opus-4-8/overview",
        "https://platform.claude.com/docs/en/about-claude/pricing",
        "https://platform.claude.com/docs/en/build-with-claude/fast-mode"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Command A",
      "api_id": "command-a",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2025-03-13",
      "context_window": 256000,
      "max_output_tokens": 8000,
      "knowledge_cutoff": "2024-06-01",
      "input_price": 2.5,
      "cached_input_price": null,
      "output_price": 10,
      "price_notes": "Regulärer Standardtarif pro 1 Mio. Tokens.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC mit Acceptable-Use-Addendum",
      "description": "Leistungsstarkes Enterprise-Modell für Tool-Nutzung, RAG, Agenten und mehrsprachige Textaufgaben.",
      "source_urls": [
        "https://docs.cohere.com/docs/command-a",
        "https://cohere.com/blog/command-a",
        "https://cohere.com/research/papers/command-a-technical-report.pdf"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Command A+",
      "api_id": "command-a-plus",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-05-20",
      "context_window": 128000,
      "max_output_tokens": 64000,
      "knowledge_cutoff": "2025-04-01",
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Kein regulärer Token-Standardtarif veröffentlicht. Laut Dokumentation für Trial- und Production-Keys bis zum Rate-Limit kostenlos; produktiver Einsatz über Model Vault.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Cohere-Flaggschiff als MoE-Modell für multimodale, mehrsprachige und reasoning-intensiven Agenten-Workloads.",
      "source_urls": [
        "https://docs.cohere.com/docs/command-a-plus",
        "https://docs.cohere.com/v2/changelog",
        "https://cohere.com/models-overview"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Command A Reasoning",
      "api_id": "command-a-reasoning",
      "tier": "reasoning",
      "status": "ga",
      "release_date": "2025-08-21",
      "context_window": 256000,
      "max_output_tokens": 32000,
      "knowledge_cutoff": "2024-06-01",
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Kein regulärer Token-Standardtarif veröffentlicht. Laut Dokumentation bis zum Rate-Limit kostenlos; produktive Nutzung über Model Vault.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Hybrid-Reasoning-Modell für komplexe, mehrstufige Agenten- und Problemlösungsaufgaben.",
      "source_urls": [
        "https://docs.cohere.com/docs/command-a-reasoning",
        "https://docs.cohere.com/v2/changelog"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Aya Expanse 32B",
      "api_id": "c4ai-aya-expanse-32b",
      "tier": "standard",
      "status": "ga",
      "release_date": "2024-10-24",
      "context_window": 128000,
      "max_output_tokens": 4000,
      "knowledge_cutoff": null,
      "input_price": 0.5,
      "cached_input_price": null,
      "output_price": 1.5,
      "price_notes": "Der offizielle Pricing-FAQ nennt diesen Tarif für Aya Expanse 8B und 32B; das 8B-API-Modell wurde am 4. April 2026 eingestellt.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Mehrsprachiges 32B-Modell für Textgenerierung, Zusammenfassung und Übersetzung in 23 Sprachen.",
      "source_urls": [
        "https://docs.cohere.com/docs/aya-expanse",
        "https://cohere.com/pricing",
        "https://cohere.com/models-overview",
        "https://cohere.com/blog/tag/research"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Aya Vision 32B",
      "api_id": "c4ai-aya-vision-32b",
      "tier": "standard",
      "status": "ga",
      "release_date": "2025-03-04",
      "context_window": 16000,
      "max_output_tokens": 4000,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Kein Token-Standardtarif für Aya Vision 32B in den offiziellen Preisangaben belegt.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Mehrsprachiges multimodales 32B-Modell für Bildbeschreibung, visuelle Fragen, Textgenerierung und Übersetzung.",
      "source_urls": [
        "https://docs.cohere.com/docs/aya-vision",
        "https://cohere.com/models-overview",
        "https://cohere.com/blog/tag/research"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "North Mini Code",
      "api_id": "north-mini-code-1-0",
      "tier": "coding",
      "status": "ga",
      "release_date": "2026-06-09",
      "context_window": 256000,
      "max_output_tokens": 64000,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Kein regulärer Token-Standardtarif veröffentlicht. Bis zum Rate-Limit kostenlos; produktive Nutzung über Model Vault.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Effizientes MoE-Coding-Modell für agentisches Programmieren und terminalbasierte Software-Engineering-Aufgaben.",
      "source_urls": [
        "https://docs.cohere.com/docs/north-mini-code-1.0",
        "https://docs.cohere.com/v2/changelog",
        "https://cohere.com/models-overview"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Command R7B",
      "api_id": "command-r7b",
      "tier": "small",
      "status": "ga",
      "release_date": "2024-12-13",
      "context_window": 128000,
      "max_output_tokens": 4000,
      "knowledge_cutoff": "2024-06-01",
      "input_price": 0.0375,
      "cached_input_price": null,
      "output_price": 0.15,
      "price_notes": "Regulärer Standardtarif pro 1 Mio. Tokens.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Kleines und schnelles Enterprise-Modell für kostensensitive RAG-, Tool-Use- und Agenten-Anwendungen.",
      "source_urls": [
        "https://docs.cohere.com/docs/command-r7b",
        "https://docs.cohere.com/changelog/command-r-7b/",
        "https://cohere.com/models-overview"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Tiny Aya Base",
      "api_id": null,
      "tier": "small",
      "status": "ga",
      "release_date": "2026-02-17",
      "context_window": 8000,
      "max_output_tokens": 8000,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Pretrained Open-Weights-Basismodell ohne Cohere-Chat-API-Zugang; daher kein API-Tokenpreis.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Vortrainiertes kompaktes 3,35B-Basismodell für mehrsprachige Anwendungen in 70 Sprachen.",
      "source_urls": [
        "https://docs.cohere.com/docs/tiny-aya",
        "https://cohere.com/blog/cohere-labs-tiny-aya"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Tiny Aya Global",
      "api_id": "tiny-aya-global",
      "tier": "small",
      "status": "ga",
      "release_date": "2026-02-17",
      "context_window": 8000,
      "max_output_tokens": 8000,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Kein Token-Standardtarif in den offiziellen Cohere-Preisangaben belegt.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Kompaktes instruktionsgetuntes 3,35B-Modell mit ausgewogener Leistung über Regionen und 70 Sprachen.",
      "source_urls": [
        "https://docs.cohere.com/docs/tiny-aya",
        "https://docs.cohere.com/docs/models",
        "https://cohere.com/models-overview"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Tiny Aya Earth",
      "api_id": "tiny-aya-earth",
      "tier": "small",
      "status": "ga",
      "release_date": "2026-02-17",
      "context_window": 8000,
      "max_output_tokens": 8000,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Kein Token-Standardtarif in den offiziellen Cohere-Preisangaben belegt.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Kompaktes instruktionsgetuntes 3,35B-Modell, besonders auf westasiatische und afrikanische Sprachen ausgerichtet.",
      "source_urls": [
        "https://docs.cohere.com/docs/tiny-aya",
        "https://docs.cohere.com/docs/models",
        "https://cohere.com/models-overview"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Tiny Aya Fire",
      "api_id": "tiny-aya-fire",
      "tier": "small",
      "status": "ga",
      "release_date": "2026-02-17",
      "context_window": 8000,
      "max_output_tokens": 8000,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Kein Token-Standardtarif in den offiziellen Cohere-Preisangaben belegt.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Kompaktes instruktionsgetuntes 3,35B-Modell, besonders auf südasiatische Sprachen ausgerichtet.",
      "source_urls": [
        "https://docs.cohere.com/docs/tiny-aya",
        "https://docs.cohere.com/docs/models",
        "https://cohere.com/models-overview"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Tiny Aya Water",
      "api_id": "tiny-aya-water",
      "tier": "small",
      "status": "ga",
      "release_date": "2026-02-17",
      "context_window": 8000,
      "max_output_tokens": 8000,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Kein Token-Standardtarif in den offiziellen Cohere-Preisangaben belegt.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Kompaktes instruktionsgetuntes 3,35B-Modell, besonders auf europäische und asiatisch-pazifische Sprachen ausgerichtet.",
      "source_urls": [
        "https://docs.cohere.com/docs/tiny-aya",
        "https://docs.cohere.com/docs/models",
        "https://cohere.com/models-overview"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Command A Translate",
      "api_id": "command-a-translate",
      "tier": "specialized",
      "status": "ga",
      "release_date": "2025-08-28",
      "context_window": 8000,
      "max_output_tokens": 8000,
      "knowledge_cutoff": "2024-06-01",
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Kein regulärer Token-Standardtarif veröffentlicht. Bis zum Rate-Limit kostenlos; für Produktion verweist Cohere auf den Vertrieb.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Spezialisiertes Modell für kontextbewusste maschinelle Übersetzung in 23 Sprachen.",
      "source_urls": [
        "https://docs.cohere.com/docs/command-a-translate",
        "https://docs.cohere.com/v2/changelog",
        "https://cohere.com/models-overview"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Command A Vision",
      "api_id": "command-a-vision",
      "tier": "specialized",
      "status": "ga",
      "release_date": "2025-07-31",
      "context_window": 128000,
      "max_output_tokens": 8000,
      "knowledge_cutoff": "2024-06-01",
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Kein regulärer Token-Standardtarif veröffentlicht. Bis zum Rate-Limit kostenlos; für Produktion verweist Cohere auf den Vertrieb.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Multimodales Enterprise-Modell für Bild- und Dokumentverständnis, OCR, Diagramme und Tabellen.",
      "source_urls": [
        "https://docs.cohere.com/docs/command-a-vision",
        "https://docs.cohere.com/v1/changelog/2025-07-31-command-a-vision",
        "https://cohere.com/models-overview"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "North Micro Vision",
      "api_id": null,
      "tier": "specialized",
      "status": "ga",
      "release_date": null,
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Open-Weights-Angebot ohne in den offiziellen Quellen dokumentierten Cohere-Chat-API-Tarif oder API-Modell-ID.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Kompaktes Vision-Language-Modell für anpassbares Bild- und Dokumentverständnis mit nativer Bildauflösung.",
      "source_urls": [
        "https://cohere.com/models-overview"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Command R+",
      "api_id": "command-r-plus",
      "tier": "flagship",
      "status": "legacy",
      "release_date": "2024-08-30",
      "context_window": 128000,
      "max_output_tokens": 4000,
      "knowledge_cutoff": "2024-06-01",
      "input_price": 2.5,
      "cached_input_price": null,
      "output_price": 10,
      "price_notes": "Regulärer Standardtarif pro 1 Mio. Tokens. Als Live-Snapshot gelistet, aber von Cohere für die meisten Fälle durch Command A abgelöst.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Älteres großes Modell für dialogorientierte Langkontext-, RAG- und mehrstufige Tool-Use-Workflows.",
      "source_urls": [
        "https://docs.cohere.com/docs/command-r-plus",
        "https://docs.cohere.com/v1/changelog/command-gets-refreshed",
        "https://cohere.com/pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "cohere",
      "provider": "Cohere",
      "name": "Command R",
      "api_id": "command-r",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2024-08-30",
      "context_window": 128000,
      "max_output_tokens": 4000,
      "knowledge_cutoff": "2024-06-01",
      "input_price": 0.15,
      "cached_input_price": null,
      "output_price": 0.6,
      "price_notes": "Regulärer Standardtarif pro 1 Mio. Tokens. Als Live-Snapshot gelistet, aber von Cohere für die meisten Fälle durch Command A abgelöst.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "streaming"
      ],
      "open_weights": true,
      "license": "CC-BY-NC-4.0",
      "description": "Älteres skalierbares Modell für dialogorientierte Langkontext-, RAG- und Tool-Use-Aufgaben.",
      "source_urls": [
        "https://docs.cohere.com/docs/command-r",
        "https://docs.cohere.com/v1/changelog/command-gets-refreshed",
        "https://cohere.com/pricing"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "deepseek",
      "provider": "DeepSeek",
      "name": "DeepSeek-V4-Pro",
      "api_id": "deepseek-v4-pro",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-08-13",
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "knowledge_cutoff": null,
      "input_price": 1.32,
      "cached_input_price": 0.044,
      "output_price": 3.96,
      "price_notes": "input_price, cached_input_price und output_price entsprechen dem nicht rabattierten Peak-Tarif in USD pro 1 Mio. Tokens. Außerhalb der Peak-Zeiten gelten jeweils die halbierten Off-Peak-Preise: 0,66 USD Input-Cache-Miss, 0,022 USD Cache-Hit und 1,98 USD Output.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT License",
      "description": "Flaggschiffmodell für anspruchsvolle Reasoning-, Agenten- und Coding-Workflows mit einstellbarem Reasoning-Aufwand bis Max.",
      "source_urls": [
        "https://api-docs.deepseek.com/quick_start/pricing/",
        "https://api-docs.deepseek.com/news/news260813/",
        "https://api-docs.deepseek.com/guides/thinking_mode/",
        "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "deepseek",
      "provider": "DeepSeek",
      "name": "DeepSeek-V4-Flash",
      "api_id": "deepseek-v4-flash",
      "tier": "standard",
      "status": "preview",
      "release_date": "2026-07-31",
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "knowledge_cutoff": null,
      "input_price": 0.44,
      "cached_input_price": 0.014,
      "output_price": 1.32,
      "price_notes": "input_price, cached_input_price und output_price entsprechen dem nicht rabattierten Peak-Tarif in USD pro 1 Mio. Tokens. Außerhalb der Peak-Zeiten gelten jeweils die halbierten Off-Peak-Preise: 0,22 USD Input-Cache-Miss, 0,007 USD Cache-Hit und 0,66 USD Output.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT License",
      "description": "Schnelles und kosteneffizientes V4-Textmodell für allgemeine Chat-, Agenten-, Coding- und Reasoning-Aufgaben mit umschaltbarem Thinking-Modus.",
      "source_urls": [
        "https://api-docs.deepseek.com/quick_start/pricing/",
        "https://api-docs.deepseek.com/news/news260424/",
        "https://api-docs.deepseek.com/updates/",
        "https://api-docs.deepseek.com/guides/thinking_mode/",
        "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "deepseek",
      "provider": "DeepSeek",
      "name": "DeepSeek-V4-Flash-Vision-Exp",
      "api_id": "deepseek-v4-flash-vision-exp",
      "tier": "specialized",
      "status": "preview",
      "release_date": "2026-08-21",
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "knowledge_cutoff": null,
      "input_price": 0.44,
      "cached_input_price": 0.014,
      "output_price": 1.32,
      "price_notes": "input_price, cached_input_price und output_price entsprechen dem nicht rabattierten Peak-Tarif in USD pro 1 Mio. Tokens; außerhalb der Peak-Zeiten gelten jeweils die halbierten Off-Peak-Preise. Eingabebilder werden abhängig von ihren Abmessungen in Tokens umgerechnet und wie Input-Tokens zum V4-Flash-Tarif abgerechnet.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT License",
      "description": "Experimentelles multimodales V4-Flash-Modell für Text- und Bildverständnis in Chat- und Agenten-Workflows.",
      "source_urls": [
        "https://api-docs.deepseek.com/quick_start/pricing/",
        "https://api-docs.deepseek.com/news/news260821/",
        "https://api-docs.deepseek.com/guides/vision/",
        "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 3.8 Flash",
      "api_id": "gemini-3.8-flash",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-09-02",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 0.75,
      "cached_input_price": 0.075,
      "output_price": 3.75,
      "price_notes": "Einführungspreis bis einschließlich 2026-12-31; ab 2027-01-01 1,50 USD Input und 7,50 USD Output. Batch und Flex kosten jeweils 50 % des Standardtarifs, Priority 180 %; Cache-Speicherung wird zusätzlich berechnet.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Googles leistungsstärkstes Flash-Modell für langlaufende Softwareentwicklung, autonome Agenten und komplexe Unternehmensabläufe.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-3.8-flash",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/changelog"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemma 4 31B IT",
      "api_id": "gemma-4-31b-it",
      "tier": "standard",
      "status": "ga",
      "release_date": null,
      "context_window": 262144,
      "max_output_tokens": null,
      "knowledge_cutoff": "2025-01",
      "input_price": 0,
      "cached_input_price": null,
      "output_price": 0,
      "price_notes": "Über die Gemini API ist das gehostete Gemma-4-Angebot im Free Tier kostenlos; für den Paid Tier weist Google keinen Tokenpreis aus.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "web_search",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Offenes dichtes 31B-Instruct-Modell für multimodales Text- und Bildverständnis, Coding und Reasoning.",
      "source_urls": [
        "https://ai.google.dev/gemma/docs/core/gemma_on_gemini_api",
        "https://ai.google.dev/gemma/docs/core/model_card_4",
        "https://ai.google.dev/gemini-api/docs/pricing"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 3.7 Flash",
      "api_id": "gemini-3.7-flash",
      "tier": "coding",
      "status": "ga",
      "release_date": "2026-08-13",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 0.75,
      "cached_input_price": 0.075,
      "output_price": 3.75,
      "price_notes": "Einführungspreis bis einschließlich 2026-12-31; ab 2027-01-01 1,50 USD Input und 7,50 USD Output. Batch und Flex kosten jeweils 50 % des Standardtarifs, Priority 180 %; Cache-Speicherung wird zusätzlich berechnet.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Schnelles multimodales Reasoning-Modell für Coding, Tool-Nutzung und zuverlässige mehrstufige Agentenabläufe.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-3.7-flash",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/changelog"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 3.6 Flash",
      "api_id": "gemini-3.6-flash",
      "tier": "coding",
      "status": "ga",
      "release_date": "2026-07-21",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 0.75,
      "cached_input_price": 0.075,
      "output_price": 3.75,
      "price_notes": "Einführungspreis bis einschließlich 2026-12-31; ab 2027-01-01 1,50 USD Input und 7,50 USD Output. Batch und Flex kosten jeweils 50 % des Standardtarifs, Priority 180 %; Cache-Speicherung wird zusätzlich berechnet.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Schnelles Frontier-Modell für Codegenerierung, agentische Ausführung und iterative Coding-Zyklen.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-3.6-flash",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/changelog"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 3.5 Flash-Lite",
      "api_id": "gemini-3.5-flash-lite",
      "tier": "small",
      "status": "ga",
      "release_date": "2026-07-21",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 0.3,
      "cached_input_price": 0.03,
      "output_price": 2.5,
      "price_notes": "Standardpreis gilt für Text-, Bild-, Video- und Audioeingaben. Batch und Flex: 0,15 USD Input sowie 1,25 USD Output; Priority: 0,54 USD Input sowie 4,50 USD Output. Cache-Speicherung wird zusätzlich berechnet.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Kleines, multimodales Low-Latency-Modell für kostengünstige Hochdurchsatz- und Subagentenaufgaben.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-3.5-flash-lite",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/changelog"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemma 4 26B A4B IT",
      "api_id": "gemma-4-26b-a4b-it",
      "tier": "small",
      "status": "ga",
      "release_date": null,
      "context_window": 262144,
      "max_output_tokens": null,
      "knowledge_cutoff": "2025-01",
      "input_price": 0,
      "cached_input_price": null,
      "output_price": 0,
      "price_notes": "Über die Gemini API ist das gehostete Gemma-4-Angebot im Free Tier kostenlos; für den Paid Tier weist Google keinen Tokenpreis aus.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "web_search",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Offenes 26B-MoE-Instruct-Modell mit 3,8B aktiven Parametern für effizientes multimodales Reasoning und Coding.",
      "source_urls": [
        "https://ai.google.dev/gemma/docs/core/gemma_on_gemini_api",
        "https://ai.google.dev/gemma/docs/core/model_card_4",
        "https://ai.google.dev/gemini-api/docs/pricing"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 3.1 Pro Preview",
      "api_id": "gemini-3.1-pro-preview",
      "tier": "flagship",
      "status": "preview",
      "release_date": "2026-02-19",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 2,
      "cached_input_price": 0.2,
      "output_price": 12,
      "price_notes": "Standardtarif für Prompts bis 200.000 Tokens; oberhalb davon 4,00 USD Input, 0,40 USD Cache-Input und 18,00 USD Output. Batch und Flex kosten bei bis zu 200.000 Tokens 1,00 USD Input und 6,00 USD Output; Priority kostet 3,60 USD Input und 21,60 USD Output.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Preview-Pro-Modell für präzises multimodales Reasoning, Software Engineering und zuverlässige Tool-gestützte Agentenabläufe.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-pro-preview",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/deprecations"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 3 Flash Preview",
      "api_id": "gemini-3-flash-preview",
      "tier": "reasoning",
      "status": "preview",
      "release_date": "2025-12-17",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 0.5,
      "cached_input_price": 0.05,
      "output_price": 3,
      "price_notes": "Standardpreis gilt für Text-, Bild- und Videoeingaben; Audio kostet 1,00 USD Input und 0,10 USD für gecachten Input. Batch und Flex: 0,25 USD Input für Text/Bild/Video sowie 1,50 USD Output; Priority: 0,90 USD Input für Text/Bild/Video und 5,40 USD Output.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Älteres Preview-Modell für multimodales Verständnis sowie agentische und Coding-orientierte Aufgaben.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-3-flash-preview",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/deprecations"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 3.1 Pro Preview Custom Tools",
      "api_id": "gemini-3.1-pro-preview-customtools",
      "tier": "coding",
      "status": "preview",
      "release_date": "2026-02-19",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 2,
      "cached_input_price": 0.2,
      "output_price": 12,
      "price_notes": "Separater API-Endpunkt derselben Pro-Preview-Familie mit identischem Tarif; optimiert für die Priorisierung eigener Werkzeuge. Für Prompts über 200.000 Tokens gelten 4,00 USD Input und 18,00 USD Output; Batch, Flex und Priority weichen wie bei Gemini 3.1 Pro Preview ab.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Spezialisierter Pro-Preview-Endpunkt für Coding-Agenten, die Bash- und eigene Funktionswerkzeuge priorisieren sollen.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-pro-preview",
        "https://ai.google.dev/gemini-api/docs/pricing"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini Deep Research",
      "api_id": "deep-research-preview",
      "tier": "specialized",
      "status": "preview",
      "release_date": "2026-04-21",
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Google berechnet die tatsächliche Nutzung anhand der zugrunde liegenden Gemini-Inferenz einschließlich Zwischen- und Reasoning-Tokens sowie verwendeter Tools; kein fester modellbezogener Tokenpreis ist ausgewiesen.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "web_search",
        "code_execution",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Preview-Forschungsagent für effiziente mehrstufige Web-Recherche und das Streamen von Ergebnissen an Client-Oberflächen.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/deep-research",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/changelog"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini Deep Research Max",
      "api_id": "deep-research-max-preview",
      "tier": "specialized",
      "status": "preview",
      "release_date": "2026-04-21",
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Google berechnet die tatsächliche Nutzung anhand der zugrunde liegenden Gemini-Inferenz einschließlich Zwischen- und Reasoning-Tokens sowie verwendeter Tools; kein fester modellbezogener Tokenpreis ist ausgewiesen.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "web_search",
        "code_execution",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Preview-Max-Variante des Forschungsagenten für besonders umfassende Kontextsammlung, Wettbewerbsanalysen und Due Diligence.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/deep-research",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/changelog"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Antigravity Agent",
      "api_id": "antigravity-preview",
      "tier": "specialized",
      "status": "preview",
      "release_date": "2026-05-19",
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Abrechnung nach den Tokenpreisen des gewählten zugrunde liegenden Gemini-Modells und der Tool-Nutzung; während der Preview berechnet Google keine Sandbox-Compute-Ressourcen. Standardmäßig verwendet der Agent Gemini 3.8 Flash, das Modell ist aber konfigurierbar.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "web_search",
        "code_execution",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Preview-Managed-Agent mit Google-gehosteter Linux-Sandbox für Planung, Codeausführung, Dateiverwaltung und Web-Browsing.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/antigravity-agent",
        "https://ai.google.dev/gemini-api/docs/agents",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/changelog"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 2.5 Pro",
      "api_id": "gemini-2.5-pro",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": "2025-06-17",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": "2025-01",
      "input_price": 1.25,
      "cached_input_price": 0.125,
      "output_price": 10,
      "price_notes": "Weiterhin verfügbares Vorgänger-Pro-Modell. Standardtarif gilt bis 200.000 Prompt-Tokens; darüber 2,50 USD Input, 0,25 USD Cache-Input und 15,00 USD Output. Batch und Flex: 0,625/5,00 USD bei bis zu 200.000 Tokens; Priority: 2,25/18,00 USD.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Älteres Pro-Reasoning-Modell für komplexe Code-, Mathematik-, STEM- und Long-Context-Aufgaben.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-pro",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/deprecations"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 3.5 Flash",
      "api_id": "gemini-3.5-flash",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2026-05-19",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 1.5,
      "cached_input_price": 0.15,
      "output_price": 9,
      "price_notes": "Weiterhin buchbares, in der Modellübersicht als Legacy bezeichnetes Flash-Modell. Batch und Flex: 0,75 USD Input sowie 4,50 USD Output; Priority: 2,70 USD Input sowie 16,20 USD Output. Cache-Speicherung wird zusätzlich berechnet.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Vorgänger-Flash-Modell für skalierbare Subagenten, mehrstufige Workflows und langlaufende Coding-Aufgaben.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-3.5-flash",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/changelog"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 2.5 Flash",
      "api_id": "gemini-2.5-flash",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2025-06-17",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": "2025-01",
      "input_price": 0.3,
      "cached_input_price": 0.03,
      "output_price": 2.5,
      "price_notes": "Weiterhin verfügbares Vorgänger-Flash-Modell. Standardpreis gilt für Text-, Bild- und Videoeingaben; Audio kostet 1,00 USD Input und 0,10 USD für gecachten Input. Batch und Flex: 0,15 USD Input für Text/Bild/Video sowie 1,25 USD Output; Priority: 0,54 USD Input und 4,50 USD Output.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Älteres preisleistungsstarkes multimodales Hybrid-Reasoning-Modell für hohe Volumina und Agenten.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-flash",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/deprecations"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 3.1 Flash-Lite",
      "api_id": "gemini-3.1-flash-lite",
      "tier": "small",
      "status": "legacy",
      "release_date": "2026-05-07",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": null,
      "input_price": 0.25,
      "cached_input_price": 0.025,
      "output_price": 1.5,
      "price_notes": "Standardpreis gilt für Text-, Bild- und Videoeingaben; Audio kostet 0,50 USD Input und 0,05 USD für gecachten Input. Batch und Flex halbieren die Tokenpreise; Priority kostet 0,45 USD Input für Text/Bild/Video bzw. 0,90 USD für Audio sowie 2,70 USD Output. Google nennt als frühestes Abschaltdatum 2027-05-07 und empfiehlt Gemini 3.5 Flash-Lite.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Älteres kleines multimodales Modell für einfache, sehr häufige und latenzkritische Aufgaben.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/deprecations"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 2.5 Flash-Lite",
      "api_id": "gemini-2.5-flash-lite",
      "tier": "small",
      "status": "legacy",
      "release_date": "2025-07-22",
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "knowledge_cutoff": "2025-01",
      "input_price": 0.1,
      "cached_input_price": 0.01,
      "output_price": 0.4,
      "price_notes": "Weiterhin verfügbares Vorgänger-Kleinmodell. Standardpreis gilt für Text-, Bild- und Videoeingaben; Audio kostet 0,30 USD Input und 0,03 USD für gecachten Input. Batch und Flex halbieren die Tokenpreise; Priority ist teurer.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Älteres sehr kostengünstiges multimodales Modell für Klassifikation, Extraktion und extrem niedrige Latenz.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-flash-lite",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/deprecations"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "google",
      "provider": "Google (Gemini)",
      "name": "Gemini 2.5 Computer Use Preview",
      "api_id": "gemini-2.5-computer-use-preview",
      "tier": "specialized",
      "status": "legacy",
      "release_date": "2025-10-07",
      "context_window": null,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.25,
      "cached_input_price": null,
      "output_price": 10,
      "price_notes": "Älteres spezialisiertes Computer-Use-Modell. Tarif bis 200.000 Prompt-Tokens; oberhalb davon 2,50 USD Input und 15,00 USD Output. Google empfiehlt inzwischen Computer Use mit Gemini-3.x-Modellen.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "computer_use",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Legacy-Modell zur Steuerung von Browser-, Mobil- und Desktop-Oberflächen anhand von Screenshots und generierten UI-Aktionen.",
      "source_urls": [
        "https://ai.google.dev/gemini-api/docs/computer-use",
        "https://ai.google.dev/gemini-api/docs/pricing",
        "https://ai.google.dev/gemini-api/docs/changelog"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "meta",
      "provider": "Meta (Llama)",
      "name": "Llama 4 Maverick",
      "api_id": null,
      "tier": "flagship",
      "status": "ga",
      "release_date": "2025-04-05",
      "context_window": 1000000,
      "max_output_tokens": null,
      "knowledge_cutoff": "2024-08",
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Open Weights ohne direkte Meta-Token-API; Meta veröffentlicht hierfür keinen eigenen API-Standardtarif.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "fine_tuning"
      ],
      "open_weights": true,
      "license": "Llama 4 Community License",
      "description": "Das leistungsstärkere offene Llama-4-Multimodalmodell für Bild- und Textverständnis sowie allgemeine Chat- und Agentenaufgaben.",
      "source_urls": [
        "https://ai.meta.com/llama/get-started/",
        "https://github.com/meta-llama/llama-models/blob/main/models/llama4/MODEL_CARD.md",
        "https://github.com/meta-llama/llama-models/blob/main/models/llama4/LICENSE"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "meta",
      "provider": "Meta (Llama)",
      "name": "Llama 4 Scout",
      "api_id": null,
      "tier": "standard",
      "status": "ga",
      "release_date": "2025-04-05",
      "context_window": 10000000,
      "max_output_tokens": null,
      "knowledge_cutoff": "2024-08",
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Open Weights ohne direkte Meta-Token-API; Meta veröffentlicht hierfür keinen eigenen API-Standardtarif.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "fine_tuning"
      ],
      "open_weights": true,
      "license": "Llama 4 Community License",
      "description": "Ein effizientes offenes Llama-4-Multimodalmodell mit sehr großem Kontextfenster, das auf den Betrieb mit einer einzelnen H100 ausgelegt ist.",
      "source_urls": [
        "https://ai.meta.com/llama/get-started/",
        "https://github.com/meta-llama/llama-models/blob/main/models/llama4/MODEL_CARD.md",
        "https://github.com/meta-llama/llama-models/blob/main/models/llama4/LICENSE"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "meta",
      "provider": "Meta (Llama)",
      "name": "Muse Glimmer 30B",
      "api_id": null,
      "tier": "small",
      "status": "ga",
      "release_date": "2026-08-10",
      "context_window": 131072,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Keine direkte, von Meta betriebene Token-API für dieses Open-Weights-Modell; die Kosten hängen bei Selbstbetrieb von der eigenen Infrastruktur ab.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "fine_tuning"
      ],
      "open_weights": true,
      "license": "Apache License 2.0",
      "description": "Ein lokal ausführbares, 30B großes multimodales Agentenmodell für mehrstufige Tool- und Automatisierungsaufgaben auf eigener Hardware.",
      "source_urls": [
        "https://huggingface.co/meta-models/Muse-Glimmer-30B",
        "https://github.com/meta-models/meta-oss-cookbook",
        "https://ai.meta.com/ai-for-good/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "meta",
      "provider": "Meta (Llama)",
      "name": "Muse Spark 1.3",
      "api_id": "muse-spark-1.3",
      "tier": "flagship",
      "status": "preview",
      "release_date": "2026-09-02",
      "context_window": 1048576,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 1.25,
      "cached_input_price": 0.15,
      "output_price": 4.25,
      "price_notes": "Standardtarif. Die alternative API-ID „muse-spark-1.3-contributor“ bezeichnet denselben Checkpoint, kostet 0,10 USD Input, 0,002 USD Cached Input und 0,20 USD Output pro 1 Mio. Tokens; dabei dürfen Eingaben und Ausgaben zur Verbesserung von Meta-Produkten verwendet werden. Reasoning-Tokens werden zum Output-Tarif abgerechnet.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "prompt_caching"
      ],
      "open_weights": false,
      "license": null,
      "description": "Metas aktuelles multimodales Flaggschiff für langlaufende agentische Aufgaben, Tool-Nutzung und Coding-Workflows.",
      "source_urls": [
        "https://research.meta.ai/blog/introducing-muse-spark-1-3",
        "https://dev.meta.ai/docs/models"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "meta",
      "provider": "Meta (Llama)",
      "name": "Llama 3.3 70B Instruct",
      "api_id": null,
      "tier": "standard",
      "status": "legacy",
      "release_date": "2024-12-06",
      "context_window": 131072,
      "max_output_tokens": null,
      "knowledge_cutoff": "2023-12",
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Open Weights ohne direkte Meta-Token-API; Meta veröffentlicht hierfür keinen eigenen API-Standardtarif.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "fine_tuning"
      ],
      "open_weights": true,
      "license": "Llama 3.3 Community License",
      "description": "Der weiterhin verfügbare Text-only-Vorgänger für mehrsprachige Assistenz-, Dialog- und Code-Workloads.",
      "source_urls": [
        "https://github.com/meta-llama/llama-models/blob/main/models/llama3_3/MODEL_CARD.md",
        "https://github.com/meta-llama/llama-models/blob/main/models/llama3_3/LICENSE"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "meta",
      "provider": "Meta (Llama)",
      "name": "Muse Spark 1.2",
      "api_id": "muse-spark-1.2",
      "tier": "coding",
      "status": "legacy",
      "release_date": "2026-08-05",
      "context_window": 1048576,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 1.25,
      "cached_input_price": 0.15,
      "output_price": 4.25,
      "price_notes": "Standardtarif. Die alternative API-ID „muse-spark-1.2-contributor“ ist derselbe Checkpoint mit 0,10 USD Input, 0,002 USD Cached Input und 0,20 USD Output pro 1 Mio. Tokens; Metas Produktverbesserungsnutzung der Anfrage- und Antwortdaten ist dafür zulässig. Reasoning-Tokens gelten als Output-Tokens.",
      "modalities_input": [
        "text",
        "image",
        "audio",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "prompt_caching"
      ],
      "open_weights": false,
      "license": null,
      "description": "Weiterhin buchbarer Vorgänger von Muse Spark 1.3, der besonders für Codegenerierung, komplexes Debugging und große Repository-Workflows optimiert wurde.",
      "source_urls": [
        "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
        "https://dev.meta.ai/docs/models"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "minimax",
      "provider": "MiniMax",
      "name": "MiniMax M3",
      "api_id": "MiniMax-M3",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-06-01",
      "context_window": 1000000,
      "max_output_tokens": 524288,
      "knowledge_cutoff": null,
      "input_price": 0.3,
      "cached_input_price": 0.06,
      "output_price": 1.2,
      "price_notes": "Standardtarif für bis zu 512k Input-Tokens. Bei mehr als 512k Input-Tokens: 0,60 USD Input, 0,12 USD Cache-Read und 2,40 USD Output pro 1 Mio. Tokens. Priority-Service kostet jeweils 1,5× des Standardtarifs.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MiniMax Community License",
      "description": "MiniMax’ multimodales Flaggschiff für agentisches Reasoning, Tool-Nutzung, Coding sowie sehr lange Kontexte bis 1 Mio. Tokens.",
      "source_urls": [
        "https://platform.minimax.io/docs/guides/pricing-paygo",
        "https://platform.minimax.io/docs/guides/text-generation",
        "https://platform.minimax.io/docs/api-reference/text-chat-openai",
        "https://platform.minimax.io/docs/release-notes/models",
        "https://platform.minimax.io/docs/guides/local-deploy-m3"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "minimax",
      "provider": "MiniMax",
      "name": "MiniMax M2.7 Highspeed",
      "api_id": "MiniMax-M2.7-highspeed",
      "tier": "coding",
      "status": "ga",
      "release_date": "2026-03-18",
      "context_window": 204800,
      "max_output_tokens": 204800,
      "knowledge_cutoff": null,
      "input_price": 0.6,
      "cached_input_price": 0.06,
      "output_price": 2.4,
      "price_notes": "Schnellere Variante mit laut MiniMax gleicher Modellleistung; Prompt-Cache-Write kostet 0,375 USD pro 1 Mio. Tokens.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "NON-COMMERCIAL LICENSE",
      "description": "Die auf niedrigere Latenz und etwa 100 Tokens pro Sekunde ausgelegte Highspeed-Variante von MiniMax M2.7.",
      "source_urls": [
        "https://platform.minimax.io/docs/guides/pricing-paygo",
        "https://platform.minimax.io/docs/guides/text-generation",
        "https://platform.minimax.io/docs/release-notes/models",
        "https://www.minimax.io/models/text/m27",
        "https://huggingface.co/MiniMaxAI/MiniMax-M2.7/blob/main/LICENSE"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "minimax",
      "provider": "MiniMax",
      "name": "MiniMax M2.7",
      "api_id": "MiniMax-M2.7",
      "tier": "coding",
      "status": "ga",
      "release_date": "2026-03-18",
      "context_window": 204800,
      "max_output_tokens": 204800,
      "knowledge_cutoff": null,
      "input_price": 0.3,
      "cached_input_price": 0.06,
      "output_price": 1.2,
      "price_notes": "Prompt-Cache-Write kostet 0,375 USD pro 1 Mio. Tokens.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "NON-COMMERCIAL LICENSE",
      "description": "Ein auf Software Engineering, komplexe Agentenaufgaben, Tool-Nutzung und produktive Wissensarbeit ausgerichtetes Reasoning- und Coding-Modell.",
      "source_urls": [
        "https://platform.minimax.io/docs/guides/pricing-paygo",
        "https://platform.minimax.io/docs/guides/text-generation",
        "https://platform.minimax.io/docs/release-notes/models",
        "https://platform.minimax.io/docs/guides/local-deploy-m2-7",
        "https://huggingface.co/MiniMaxAI/MiniMax-M2.7/blob/main/LICENSE"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "minimax",
      "provider": "MiniMax",
      "name": "MiniMax M2.5 Highspeed",
      "api_id": "MiniMax-M2.5-highspeed",
      "tier": "coding",
      "status": "legacy",
      "release_date": "2026-02",
      "context_window": 204800,
      "max_output_tokens": 204800,
      "knowledge_cutoff": null,
      "input_price": 0.6,
      "cached_input_price": 0.03,
      "output_price": 2.4,
      "price_notes": "Als Legacy-Modell weiterhin buchbar; schnellere Variante mit gleicher Modellleistung. Prompt-Cache-Write kostet 0,375 USD pro 1 Mio. Tokens.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MiniMax Model License",
      "description": "Die schnellere, weiterhin verfügbare Legacy-Variante von M2.5 für Coding- und Agent-Aufgaben.",
      "source_urls": [
        "https://platform.minimax.io/docs/guides/pricing-paygo",
        "https://platform.minimax.io/docs/guides/text-generation",
        "https://platform.minimax.io/docs/release-notes/models",
        "https://www.minimax.io/models/text",
        "https://huggingface.co/MiniMaxAI/MiniMax-M2.5/blob/main/LICENSE-MODEL"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "minimax",
      "provider": "MiniMax",
      "name": "MiniMax M2.5",
      "api_id": "MiniMax-M2.5",
      "tier": "coding",
      "status": "legacy",
      "release_date": "2026-02",
      "context_window": 204800,
      "max_output_tokens": 204800,
      "knowledge_cutoff": null,
      "input_price": 0.3,
      "cached_input_price": 0.03,
      "output_price": 1.2,
      "price_notes": "Als Legacy-Modell weiterhin buchbar; Prompt-Cache-Write kostet 0,375 USD pro 1 Mio. Tokens.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MiniMax Model License",
      "description": "Die letzte noch buchbare Vorgängergeneration für Coding, Agent-Workflows, Suche und komplexe Office-Aufgaben.",
      "source_urls": [
        "https://platform.minimax.io/docs/guides/pricing-paygo",
        "https://platform.minimax.io/docs/guides/text-generation",
        "https://platform.minimax.io/docs/release-notes/models",
        "https://www.minimax.io/models/text",
        "https://huggingface.co/MiniMaxAI/MiniMax-M2.5/blob/main/LICENSE-MODEL"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "minimax",
      "provider": "MiniMax",
      "name": "M2-her",
      "api_id": "M2-her",
      "tier": "specialized",
      "status": "legacy",
      "release_date": null,
      "context_window": 64000,
      "max_output_tokens": 2048,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "In der aktuellen Pay-as-you-go-Preistabelle ist kein separater Tarif mehr ausgewiesen; daher keine Preisangabe. Das Modell ist weiterhin per API-Aufruf mit der festen Modell-ID „M2-her“ dokumentiert.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Ein spezialisiertes Text-Chatmodell für Rollenspiel, reichhaltige Rollenprofile und lange Mehrturn-Dialoge.",
      "source_urls": [
        "https://platform.minimax.io/docs/guides/text-chat",
        "https://platform.minimax.io/docs/guides/text-generation",
        "https://platform.minimax.io/docs/guides/pricing-paygo"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Mistral Medium 3.5",
      "api_id": "mistral-medium-3-5",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-04-28",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.5,
      "cached_input_price": 0.15,
      "output_price": 7.5,
      "price_notes": "Standardtarif. Batch kostet 50 % weniger; regionale Inferenz kostet für unterstützte Modelle 10 % mehr. Cache-Hits kosten 0,15 USD pro 1 Mio. Input-Tokens.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching"
      ],
      "open_weights": true,
      "license": "Modified MIT",
      "description": "Frontier-Multimodalmodell für langlaufende agentische Aufgaben, einstellbares Reasoning und Softwareentwicklung.",
      "source_urls": [
        "https://docs.mistral.ai/models/mistral-medium-3-5-26-04",
        "https://docs.mistral.ai/inference/pricing",
        "https://mistral.ai/pricing/api/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Mistral Large 3",
      "api_id": "mistral-large-3",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2025-12-02",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.5,
      "cached_input_price": 0.05,
      "output_price": 1.5,
      "price_notes": "Standardtarif. Batch kostet 50 % weniger; regionale Inferenz kostet für unterstützte Modelle 10 % mehr. Cache-Hits kosten 0,05 USD pro 1 Mio. Input-Tokens.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Offenes multimodales Flaggschiff-Allzweckmodell mit granularer Mixture-of-Experts-Architektur.",
      "source_urls": [
        "https://docs.mistral.ai/models/mistral-large-3-25-12",
        "https://docs.mistral.ai/inference/pricing",
        "https://mistral.ai/pricing/api/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Codestral",
      "api_id": "codestral",
      "tier": "coding",
      "status": "ga",
      "release_date": "2025-07-30",
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.3,
      "cached_input_price": 0.03,
      "output_price": 0.9,
      "price_notes": "Standardtarif. Batch kostet 50 % weniger; regionale Inferenz kostet für unterstützte Modelle 10 % mehr. Cache-Hits kosten 0,03 USD pro 1 Mio. Input-Tokens.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "batch",
        "prompt_caching"
      ],
      "open_weights": false,
      "license": null,
      "description": "Niedrig-latentes Coding-Modell für häufige Code-Vervollständigung, Fill-in-the-Middle und Codegenerierung.",
      "source_urls": [
        "https://docs.mistral.ai/models/codestral-25-08",
        "https://docs.mistral.ai/inference/pricing",
        "https://mistral.ai/pricing/api/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Ministral 3 14B",
      "api_id": "ministral-3-14b",
      "tier": "small",
      "status": "ga",
      "release_date": "2025-12-02",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.2,
      "cached_input_price": 0.02,
      "output_price": 0.2,
      "price_notes": "Standardtarif. Batch kostet 50 % weniger; regionale Inferenz kostet für unterstützte Modelle 10 % mehr. Cache-Hits kosten 0,02 USD pro 1 Mio. Input-Tokens.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "batch",
        "prompt_caching"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Größtes Modell der edge-orientierten Ministral-3-Familie für leistungsfähige lokale Text- und Vision-Anwendungen.",
      "source_urls": [
        "https://docs.mistral.ai/models/ministral-3-14b-25-12",
        "https://docs.mistral.ai/inference/pricing",
        "https://mistral.ai/pricing/api/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Mistral Small 4",
      "api_id": "mistral-small-4",
      "tier": "small",
      "status": "ga",
      "release_date": "2026-03-16",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.15,
      "cached_input_price": 0.015,
      "output_price": 0.6,
      "price_notes": "Standardtarif. Batch kostet 50 % weniger; regionale Inferenz kostet für unterstützte Modelle 10 % mehr. Cache-Hits kosten 0,015 USD pro 1 Mio. Input-Tokens.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Effizientes multimodales Hybridmodell, das Instruction-Following, Reasoning und Coding in einem Modell vereint.",
      "source_urls": [
        "https://docs.mistral.ai/models/mistral-small-4-0-26-03",
        "https://docs.mistral.ai/inference/pricing",
        "https://mistral.ai/pricing/api/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Ministral 3 8B",
      "api_id": "ministral-3-8b",
      "tier": "small",
      "status": "ga",
      "release_date": "2025-12-02",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.15,
      "cached_input_price": 0.015,
      "output_price": 0.15,
      "price_notes": "Standardtarif. Batch kostet 50 % weniger; regionale Inferenz kostet für unterstützte Modelle 10 % mehr. Cache-Hits kosten 0,015 USD pro 1 Mio. Input-Tokens.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "batch",
        "prompt_caching"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Kompaktes und effizientes Edge-Modell mit Text- und Vision-Fähigkeiten.",
      "source_urls": [
        "https://docs.mistral.ai/models/ministral-3-8b-25-12",
        "https://docs.mistral.ai/inference/pricing",
        "https://mistral.ai/pricing/api/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Ministral 3 3B",
      "api_id": "ministral-3-3b",
      "tier": "small",
      "status": "ga",
      "release_date": "2025-12-02",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.1,
      "cached_input_price": 0.01,
      "output_price": 0.1,
      "price_notes": "Standardtarif. Batch kostet 50 % weniger; regionale Inferenz kostet für unterstützte Modelle 10 % mehr. Cache-Hits kosten 0,01 USD pro 1 Mio. Input-Tokens.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "batch",
        "prompt_caching"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Kleinstes und besonders effizientes Modell der Ministral-3-Familie für lokale Text- und Vision-Aufgaben.",
      "source_urls": [
        "https://docs.mistral.ai/models/ministral-3-3b-25-12",
        "https://docs.mistral.ai/inference/pricing",
        "https://mistral.ai/pricing/api/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Voxtral Small",
      "api_id": "voxtral-small",
      "tier": "specialized",
      "status": "ga",
      "release_date": "2025-07-15",
      "context_window": 32000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.1,
      "cached_input_price": null,
      "output_price": 0.4,
      "price_notes": "Der Text-Input kostet standardmäßig 0,10 USD pro 1 Mio. Tokens; Audio-Input wird getrennt mit 0,004 USD pro Minute abgerechnet. Batch kostet 50 % weniger; ein konkreter Cache-Tarif für dieses Modell ist nicht ausgewiesen.",
      "modalities_input": [
        "text",
        "audio"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "batch"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Multimodales Instruct-Modell mit Audioeingabe für Sprachverständnis und textbasierte Antworten.",
      "source_urls": [
        "https://docs.mistral.ai/models/voxtral-small-25-07",
        "https://mistral.ai/pricing/api/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Z.ai GLM 5.2",
      "api_id": "zai-glm-5-2",
      "tier": "coding",
      "status": "preview",
      "release_date": "2026-08-06",
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": null,
      "input_price": 1.4,
      "cached_input_price": 0.14,
      "output_price": 4.4,
      "price_notes": "Standardtarif für das von Mistral gehostete Drittanbieter-Modell. Batch kostet 50 % weniger; regionale Inferenz kostet für unterstützte Modelle 10 % mehr. Cache-Hits kosten 0,14 USD pro 1 Mio. Input-Tokens.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "batch",
        "prompt_caching"
      ],
      "open_weights": true,
      "license": null,
      "description": "Von Z.ai stammendes, unverändert von Mistral gehostetes Open-Source-Textmodell für Coding und agentische Long-Context-Workflows.",
      "source_urls": [
        "https://docs.mistral.ai/models/zai-glm-5-2",
        "https://docs.mistral.ai/inference/pricing",
        "https://mistral.ai/pricing/api/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Leanstral 1.5",
      "api_id": "labs-leanstral-1-5",
      "tier": "coding",
      "status": "preview",
      "release_date": "2026-06-30",
      "context_window": 256000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": null,
      "input_price": 0,
      "cached_input_price": null,
      "output_price": 0,
      "price_notes": "Der öffentliche Preview-Endpunkt ist derzeit kostenlos. Der Anbieter kündigt für dieses Modell die Einstellung zum 2026-09-30 an; es ist daher nicht für Produktionsbetrieb vorgesehen.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Preview-Modell für Lean-4-Proof-Engineering, automatisiertes Theorem Proving und Autoformalization.",
      "source_urls": [
        "https://docs.mistral.ai/models/leanstral-1-5",
        "https://docs.mistral.ai/resources/changelogs",
        "https://mistral.ai/pricing/api/"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Mistral Medium 3.1",
      "api_id": "mistral-medium-3-1",
      "tier": "flagship",
      "status": "legacy",
      "release_date": "2025-08-12",
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Abgekündigt am 2026-05-22 und durch Mistral Medium 3.5 ersetzt. Die dokumentierte Frist für GA-Modelle beträgt sechs Monate; am Stichtag 2026-09-04 ist das Modell daher noch erreichbar, ein aktueller Preis ist jedoch nicht ausgewiesen.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch"
      ],
      "open_weights": false,
      "license": null,
      "description": "Abgekündigtes multimodales Frontier-Modell der vorherigen Mistral-Medium-Generation.",
      "source_urls": [
        "https://docs.mistral.ai/models/mistral-medium-3-1-25-08",
        "https://docs.mistral.ai/inference/model-lifecycle"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Magistral Medium 1.2",
      "api_id": "magistral-medium-1-2",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": "2025-09-18",
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Abgekündigt am 2026-05-22 und für neue Integrationen durch Mistral Medium 3.5 ersetzt. Die dokumentierte Frist für GA-Modelle beträgt sechs Monate; am Stichtag 2026-09-04 ist das Modell daher noch erreichbar, ein aktueller Preis ist jedoch nicht ausgewiesen.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "batch"
      ],
      "open_weights": false,
      "license": null,
      "description": "Abgekündigtes Frontier-Multimodalmodell der Magistral-Reihe für Reasoning.",
      "source_urls": [
        "https://docs.mistral.ai/models/magistral-medium-1-2-25-09",
        "https://docs.mistral.ai/inference/model-lifecycle"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Magistral Small 1.2",
      "api_id": "magistral-small-1-2",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": "2025-09-18",
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Abgekündigt am 2026-04-30 und für neue Integrationen durch Mistral Small 4 ersetzt. Die dokumentierte Frist für GA-Modelle beträgt sechs Monate; am Stichtag 2026-09-04 ist das Modell daher noch erreichbar, ein aktueller Preis ist jedoch nicht ausgewiesen.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "batch"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Abgekündigtes kleines multimodales Reasoning-Modell der Magistral-Reihe.",
      "source_urls": [
        "https://docs.mistral.ai/models/magistral-small-1-2-25-09",
        "https://docs.mistral.ai/inference/model-lifecycle"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Devstral 2",
      "api_id": "devstral-2",
      "tier": "coding",
      "status": "legacy",
      "release_date": "2025-12-09",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Abgekündigt am 2026-05-22 und für neue Integrationen durch Mistral Medium 3.5 ersetzt. Die dokumentierte Frist für GA-Modelle beträgt sechs Monate; am Stichtag 2026-09-04 ist das Modell daher noch erreichbar, ein aktueller Preis ist jedoch nicht ausgewiesen.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "batch"
      ],
      "open_weights": true,
      "license": "Modified MIT",
      "description": "Abgekündigtes Open-Weights-Modell für agentische Softwareentwicklung mit Werkzeugnutzung und Bearbeitung mehrerer Dateien.",
      "source_urls": [
        "https://docs.mistral.ai/models/devstral-2-25-12",
        "https://docs.mistral.ai/inference/model-lifecycle"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "mistral",
      "provider": "Mistral AI",
      "name": "Mistral Small 3.2",
      "api_id": "mistral-small-3-2",
      "tier": "small",
      "status": "legacy",
      "release_date": "2025-06-20",
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Abgekündigt am 2026-04-30 und durch Mistral Small 4 ersetzt. Die dokumentierte Frist für GA-Modelle beträgt sechs Monate; am Stichtag 2026-09-04 ist das Modell daher noch erreichbar, ein aktueller Preis ist jedoch nicht ausgewiesen.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch"
      ],
      "open_weights": true,
      "license": "Apache 2.0",
      "description": "Abgekündigtes kleines multimodales Allzweckmodell der vorherigen Mistral-Small-Generation.",
      "source_urls": [
        "https://docs.mistral.ai/models/mistral-small-3-2-25-06",
        "https://docs.mistral.ai/inference/model-lifecycle"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "moonshot",
      "provider": "Moonshot AI (Kimi)",
      "name": "Kimi K3",
      "api_id": "kimi-k3",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-07-16",
      "context_window": 1048576,
      "max_output_tokens": 1048576,
      "knowledge_cutoff": null,
      "input_price": 3,
      "cached_input_price": 0.3,
      "output_price": 15,
      "price_notes": "Standardtarif bei Cache-Miss; Cache-Hits kosten 0,30 USD pro 1 Mio. Input-Tokens. Einheitlicher Tarif bis zum 1M-Kontextfenster, ohne Long-Context-Staffel; Preise zuzüglich ggf. anfallender Steuern.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "Kimi K3 License",
      "description": "Flaggschiff für langlaufende Coding- und Wissensarbeit sowie tiefes Reasoning mit nativem Bild- und Videoverständnis.",
      "source_urls": [
        "https://platform.kimi.ai/docs/models",
        "https://platform.kimi.ai/docs/guide/kimi-k3-quickstart",
        "https://platform.kimi.ai/docs/pricing/chat-k3",
        "https://platform.kimi.ai/",
        "https://www.moonshot.ai/",
        "https://github.com/MoonshotAI/Kimi-K3/blob/main/LICENSE"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "moonshot",
      "provider": "Moonshot AI (Kimi)",
      "name": "Kimi K2.6",
      "api_id": "kimi-k2.6",
      "tier": "standard",
      "status": "ga",
      "release_date": "2026-04-20",
      "context_window": 262144,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.95,
      "cached_input_price": 0.16,
      "output_price": 4,
      "price_notes": "Standardtarif bei Cache-Miss; Cache-Hits kosten 0,16 USD pro 1 Mio. Input-Tokens. Im Batch-Modus beträgt der Tarif 60 % des Echtzeit-Tarifs; Preise zuzüglich ggf. anfallender Steuern.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "Modified MIT License",
      "description": "Allzweckmodell für Dialog, Coding, Agentenaufgaben und multimodales Verstehen, dessen Thinking-Modus pro Anfrage deaktiviert werden kann.",
      "source_urls": [
        "https://platform.kimi.ai/docs/models",
        "https://platform.kimi.ai/docs/guide/kimi-k2-6-quickstart",
        "https://platform.kimi.ai/docs/pricing/chat-k26",
        "https://www.kimi.ai/resources/kimi-k2-6-pricing",
        "https://huggingface.co/moonshotai/Kimi-K2.6/blob/main/README.md",
        "https://github.com/MoonshotAI/kimi-help-center/blob/master/en-US/agent/overview.md"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "moonshot",
      "provider": "Moonshot AI (Kimi)",
      "name": "Kimi K2.7 Code HighSpeed",
      "api_id": "kimi-k2.7-code-highspeed",
      "tier": "coding",
      "status": "ga",
      "release_date": null,
      "context_window": 262144,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.9,
      "cached_input_price": 0.38,
      "output_price": 8,
      "price_notes": "Eigenständige HighSpeed-API-ID mit etwa 180 Tokens/s Ausgabe, in kurzen Kontexten bis zu 260 Tokens/s. Die Preise sind der Echtzeit-Standardtarif bei Cache-Miss; Cache-Hits kosten 0,38 USD pro 1 Mio. Input-Tokens; zuzüglich ggf. anfallender Steuern.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "Modified MIT License",
      "description": "Beschleunigte API-Variante von Kimi K2.7 Code für latenzkritische Coding-Workflows; Modellverhalten und Thinking-Modus entsprechen K2.7 Code.",
      "source_urls": [
        "https://platform.kimi.ai/docs/models",
        "https://platform.kimi.ai/docs/guide/kimi-k2-7-code-quickstart",
        "https://platform.kimi.ai/docs/pricing/chat-k27-code",
        "https://www.kimi.ai/resources/kimi-k2-7-code-pricing",
        "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "moonshot",
      "provider": "Moonshot AI (Kimi)",
      "name": "Kimi K2.7 Code",
      "api_id": "kimi-k2.7-code",
      "tier": "coding",
      "status": "ga",
      "release_date": null,
      "context_window": 262144,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 0.95,
      "cached_input_price": 0.19,
      "output_price": 4,
      "price_notes": "Standardtarif bei Cache-Miss; Cache-Hits kosten 0,19 USD pro 1 Mio. Input-Tokens. Im Batch-Modus beträgt der Tarif 60 % des Echtzeit-Tarifs; Preise zuzüglich ggf. anfallender Steuern.",
      "modalities_input": [
        "text",
        "image",
        "video"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "Modified MIT License",
      "description": "Dediziertes Coding- und Agentenmodell für langlaufende Software-Engineering-Aufgaben mit dauerhaft aktiviertem Thinking.",
      "source_urls": [
        "https://platform.kimi.ai/docs/models",
        "https://platform.kimi.ai/docs/guide/kimi-k2-7-code-quickstart",
        "https://platform.kimi.ai/docs/pricing/chat-k27-code",
        "https://www.kimi.ai/resources/kimi-k2-7-code-pricing",
        "https://huggingface.co/moonshotai/Kimi-K2.7-Code"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.6 Sol",
      "api_id": "gpt-5.6-sol",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-07-09",
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-02-16",
      "input_price": 4,
      "cached_input_price": 0.4,
      "output_price": 20,
      "price_notes": "Standardtarif einschließlich der bis mindestens 2026-11-21 angekündigten Aktionspreise. Ab mehr als 272.000 Input-Tokens: 2× Input und 1,5× Output; Cache-Writes kosten 1,25× Input. Batch und Flex kosten 50 % des Standardtarifs, Fast Mode 2×; Regional Processing ggf. +10 %.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Flaggschiff der GPT-5.6-Familie für komplexe professionelle Aufgaben, Coding und agentische Workflows.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
        "https://platform.openai.com/docs/pricing",
        "https://openai.com/index/gpt-5-6/"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.6 Terra",
      "api_id": "gpt-5.6-terra",
      "tier": "standard",
      "status": "ga",
      "release_date": "2026-07-09",
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-02-16",
      "input_price": 2,
      "cached_input_price": 0.2,
      "output_price": 12,
      "price_notes": "Standardtarif. Ab mehr als 272.000 Input-Tokens: 2× Input und 1,5× Output; Cache-Writes kosten 1,25× Input. Batch und Flex kosten 50 % des Standardtarifs, Fast Mode 2×; Regional Processing ggf. +10 %.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Ausgewogenes GPT-5.6-Modell für Workloads, bei denen Intelligenz und Kosten im Gleichgewicht stehen sollen.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
        "https://platform.openai.com/docs/pricing",
        "https://openai.com/index/gpt-5-6/"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.3-Codex",
      "api_id": "gpt-5.3-codex",
      "tier": "coding",
      "status": "ga",
      "release_date": "2026-02-05",
      "context_window": 400000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2025-08-31",
      "input_price": 1.75,
      "cached_input_price": 0.175,
      "output_price": 14,
      "price_notes": "Standardtarif. Fast Mode kostet 3,50 USD Input, 0,35 USD Cached Input und 28 USD Output pro 1 Mio. Tokens. Weitere Tool-Kosten, etwa für Web Search oder Container, werden separat abgerechnet.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Spezialisiertes Modell für agentisches Coding in Codex oder vergleichbaren Entwicklungsumgebungen.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.3-codex",
        "https://platform.openai.com/docs/pricing",
        "https://openai.com/index/introducing-gpt-5-3-codex/"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.6 Luna",
      "api_id": "gpt-5.6-luna",
      "tier": "small",
      "status": "ga",
      "release_date": "2026-07-09",
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-02-16",
      "input_price": 0.2,
      "cached_input_price": 0.02,
      "output_price": 1.2,
      "price_notes": "Standardtarif. Ab mehr als 272.000 Input-Tokens: 2× Input und 1,5× Output; Cache-Writes kosten 1,25× Input. Batch und Flex kosten 50 % des Standardtarifs, Fast Mode 2×; Regional Processing ggf. +10 %.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Das kostenoptimierte GPT-5.6-Modell für schnelle Anwendungen mit hohem Volumen.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
        "https://platform.openai.com/docs/pricing",
        "https://openai.com/index/gpt-5-6/"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.6 Cyber",
      "api_id": "gpt-5.6-cyber",
      "tier": "specialized",
      "status": "ga",
      "release_date": "2026-08-10",
      "context_window": 400000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-02-16",
      "input_price": 12.5,
      "cached_input_price": 1.25,
      "output_price": 75,
      "price_notes": "Standardtarif; Zugang setzt separate Genehmigung und Provisionierung im Daybreak-Programm voraus. Laut Modellspezifikation greifen oberhalb von 272.000 Input-Tokens 2× Input und 1,5× Output; Cache-Writes kosten 1,25× Input. Regional Processing ggf. +10 %.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Spezialmodell für autorisierte Schwachstellenforschung, Exploit-Validierung und Security-Tests durch zugelassene Verteidiger.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.6-cyber",
        "https://platform.openai.com/docs/pricing",
        "https://openai.com/index/expanding-daybreak-as-the-cyber-defense-window-narrows/"
      ],
      "verified_in_api": false
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "gpt-oss-120b",
      "api_id": null,
      "tier": "specialized",
      "status": "ga",
      "release_date": "2025-08-05",
      "context_window": 131072,
      "max_output_tokens": 131072,
      "knowledge_cutoff": "2024-06-01",
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Open-Weights-Modell ohne separat bepreiste, nutzbare OpenAI-Hosted-API; die Infrastrukturkosten der eigenen Bereitstellung fallen außerhalb der OpenAI-Tokenpreise an.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "fine_tuning",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Leistungsstärkstes Open-Weights-Modell von OpenAI für lokale oder spezialisierte Reasoning- und Agent-Workloads.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-oss-120b",
        "https://openai.com/index/introducing-gpt-oss/"
      ],
      "verified_in_api": false
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "gpt-oss-20b",
      "api_id": null,
      "tier": "specialized",
      "status": "ga",
      "release_date": "2025-08-05",
      "context_window": 131072,
      "max_output_tokens": 131072,
      "knowledge_cutoff": "2024-06-01",
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Open-Weights-Modell ohne separat bepreiste, nutzbare OpenAI-Hosted-API; die Infrastrukturkosten der eigenen Bereitstellung fallen außerhalb der OpenAI-Tokenpreise an.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "fine_tuning",
        "streaming"
      ],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Kompakteres Open-Weights-Modell für latenzarme lokale und spezialisierte Anwendungen.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-oss-20b",
        "https://openai.com/index/introducing-gpt-oss/"
      ],
      "verified_in_api": false
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-6 Astra",
      "api_id": "gpt-6-astra",
      "tier": "flagship",
      "status": "preview",
      "release_date": "2026-09-03",
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2026-04-30",
      "input_price": 10,
      "cached_input_price": 1,
      "output_price": 50,
      "price_notes": "Standardtarif. Ab mehr als 272.000 Input-Tokens gilt für die gesamte Anfrage der doppelte Input-/Cache-Tarif und der 1,5-fache Output-Tarif. Batch und Flex kosten 50 % des Standardtarifs; Fast Mode kostet das Doppelte. Regional Processing kostet, soweit verfügbar, 10 % Aufschlag.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Das leistungsstärkste Modell für besonders anspruchsvolle End-to-End-Arbeit, komplexes Reasoning, Coding, Recherche und Computer Use.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-6-astra",
        "https://platform.openai.com/docs/pricing",
        "https://openai.com/index/gpt-6-astra/"
      ],
      "verified_in_api": false
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.5",
      "api_id": "gpt-5.5",
      "tier": "flagship",
      "status": "legacy",
      "release_date": "2026-04-23",
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2025-12-01",
      "input_price": 5,
      "cached_input_price": 0.5,
      "output_price": 30,
      "price_notes": "Standardtarif. Ab mehr als 272.000 Input-Tokens: 2× Input und 1,5× Output für die gesamte Session. Batch und Flex kosten 50 % des damaligen Standardtarifs; Priority/Fast-Verarbeitung wurde mit einem Aufschlag angeboten. Regional Processing kostet +10 %.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Vorgänger-Flaggschiff für anspruchsvolle professionelle Arbeit, agentisches Coding, Recherche und Computer Use.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.5",
        "https://openai.com/index/introducing-gpt-5-5/"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.4",
      "api_id": "gpt-5.4",
      "tier": "flagship",
      "status": "legacy",
      "release_date": "2026-03-05",
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2025-08-31",
      "input_price": 2.5,
      "cached_input_price": 0.25,
      "output_price": 15,
      "price_notes": "Standardtarif. Ab mehr als 272.000 Input-Tokens: 2× Input und 1,5× Output für die gesamte Session. Regional Processing kostet +10 %; Batch-/Flex- und Fast-Tarife können abweichen.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Weiterhin verfügbares Vorgänger-Flaggschiff für professionelle Arbeit, Coding und Tool-gestützte agentische Aufgaben.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.4",
        "https://openai.com/index/introducing-gpt-5-4/"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.5 Pro",
      "api_id": "gpt-5.5-pro",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": "2026-04-23",
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2025-12-01",
      "input_price": 30,
      "cached_input_price": null,
      "output_price": 180,
      "price_notes": "Standardtarif; kein Rabatt für gecachten Input. Das Modell ist über die Responses API einschließlich Batch verfügbar; Regional Processing kostet +10 %.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch"
      ],
      "open_weights": false,
      "license": null,
      "description": "Rechenintensivere Pro-Variante von GPT-5.5 für besonders schwierige Aufgaben mit höherer Antwortgenauigkeit.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.5-pro",
        "https://openai.com/index/introducing-gpt-5-5/"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.4 Pro",
      "api_id": "gpt-5.4-pro",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": "2026-03-05",
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2025-08-31",
      "input_price": 30,
      "cached_input_price": null,
      "output_price": 180,
      "price_notes": "Standardtarif; kein Rabatt für gecachten Input. Ab mehr als 272.000 Input-Tokens: 2× Input und 1,5× Output für die gesamte Session. Regional Processing kostet +10 %.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "web_search",
        "computer_use",
        "batch",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Pro-Variante von GPT-5.4 mit mehr Inferenz-Rechenaufwand für schwierigste professionelle Aufgaben.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.4-pro",
        "https://openai.com/index/introducing-gpt-5-4/"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.4 Mini",
      "api_id": "gpt-5.4-mini",
      "tier": "small",
      "status": "legacy",
      "release_date": "2026-03-17",
      "context_window": 400000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2025-08-31",
      "input_price": 0.75,
      "cached_input_price": 0.075,
      "output_price": 4.5,
      "price_notes": "Standardtarif; Regional Processing kostet +10 %. Batch-, Flex- oder Fast-Varianten sind nicht als separate Modelle erfasst.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "computer_use",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Kleineres GPT-5.4-Modell für effiziente High-Volume-Workloads, Coding, Computer Use und Subagents.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.4-mini",
        "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "openai",
      "provider": "OpenAI",
      "name": "GPT-5.4 nano",
      "api_id": "gpt-5.4-nano",
      "tier": "small",
      "status": "legacy",
      "release_date": "2026-03-17",
      "context_window": 400000,
      "max_output_tokens": 128000,
      "knowledge_cutoff": "2025-08-31",
      "input_price": 0.2,
      "cached_input_price": 0.02,
      "output_price": 1.25,
      "price_notes": "Standardtarif; Regional Processing kostet +10 %. Batch-, Flex- oder Fast-Varianten sind nicht als separate Modelle erfasst.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "web_search",
        "code_execution",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Preisgünstigstes GPT-5.4-Modell für Klassifikation, Extraktion, Ranking und einfache Subagent-Aufgaben.",
      "source_urls": [
        "https://developers.openai.com/api/docs/models/gpt-5.4-nano",
        "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/"
      ],
      "verified_in_api": true
    },
    {
      "provider_id": "perplexity",
      "provider": "Perplexity (Sonar)",
      "name": "Sonar Pro",
      "api_id": "sonar-pro",
      "tier": "flagship",
      "status": "legacy",
      "release_date": "2025-01",
      "context_window": 200000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 3,
      "cached_input_price": null,
      "output_price": 15,
      "price_notes": "Zusätzlich zur Tokenabrechnung: Fast-Search-Request-Gebühr je 1.000 Anfragen von 6 USD (Low, Standard), 10 USD (Medium) oder 14 USD (High). Optionales Pro Search kostet 14 USD, 18 USD bzw. 22 USD je 1.000 Anfragen (Low/Medium/High); bei auto erfolgt die Abrechnung nach der automatischen Klassifikation. Bild-Inputs werden als Input-Tokens abgerechnet.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text",
        "image"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "web_search",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Premium-Suchmodell für komplexe Abfragen, vertiefte Informationssynthese und Pro Search mit mehrstufiger Werkzeugnutzung.",
      "source_urls": [
        "https://docs.perplexity.ai/docs/sonar/models/sonar-pro",
        "https://docs.perplexity.ai/docs/getting-started/pricing",
        "https://docs.perplexity.ai/docs/sonar/pro-search/quickstart",
        "https://docs.perplexity.ai/docs/sonar/media",
        "https://docs.perplexity.ai/docs/resources/changelog"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "perplexity",
      "provider": "Perplexity (Sonar)",
      "name": "Sonar Reasoning Pro",
      "api_id": "sonar-reasoning-pro",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": null,
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 2,
      "cached_input_price": null,
      "output_price": 8,
      "price_notes": "Zusätzlich zur Tokenabrechnung: Request-Gebühr je 1.000 Anfragen von 6 USD (Low, Standard), 10 USD (Medium) oder 14 USD (High). Das Modell gibt Reasoning im <think>-Abschnitt aus; Bild-Input in Verbindung mit Structured Outputs wird nicht unterstützt. Bild-Inputs werden ansonsten als Input-Tokens abgerechnet.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text",
        "image"
      ],
      "capabilities": [
        "reasoning",
        "structured_output",
        "web_search",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Reasoning-Modell für anspruchsvolle mehrstufige Analysen und Problemlösung mit Chain-of-Thought sowie Websuche.",
      "source_urls": [
        "https://docs.perplexity.ai/docs/sonar/models/sonar-reasoning-pro",
        "https://docs.perplexity.ai/docs/getting-started/pricing",
        "https://docs.perplexity.ai/docs/sonar/media",
        "https://docs.perplexity.ai/docs/resources/changelog"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "perplexity",
      "provider": "Perplexity (Sonar)",
      "name": "Sonar Deep Research",
      "api_id": "sonar-deep-research",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": null,
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 2,
      "cached_input_price": null,
      "output_price": 8,
      "price_notes": "Neben 2 USD Input und 8 USD Output pro 1 Mio. Tokens werden Citation Tokens mit 2 USD pro 1 Mio. Tokens, Reasoning Tokens mit 3 USD pro 1 Mio. Tokens und Suchanfragen mit 5 USD pro 1.000 Suchanfragen berechnet. Der reasoning_effort kann low, medium oder high sein und beeinflusst insbesondere die verbrauchten Reasoning Tokens; asynchrone Ausführung ist verfügbar. Bild-Inputs werden als Input-Tokens abgerechnet.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text",
        "image"
      ],
      "capabilities": [
        "reasoning",
        "structured_output",
        "web_search",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Deep-Research-Modell für umfangreiche Webrecherchen über viele Quellen und detaillierte, fachlich fundierte Berichte.",
      "source_urls": [
        "https://docs.perplexity.ai/docs/sonar/models/sonar-deep-research",
        "https://docs.perplexity.ai/docs/getting-started/pricing",
        "https://docs.perplexity.ai/docs/sonar/media",
        "https://docs.perplexity.ai/docs/resources/changelog"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "perplexity",
      "provider": "Perplexity (Sonar)",
      "name": "Sonar",
      "api_id": "sonar",
      "tier": "small",
      "status": "legacy",
      "release_date": "2025-01",
      "context_window": 128000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1,
      "cached_input_price": null,
      "output_price": 1,
      "price_notes": "Zusätzlich zur Tokenabrechnung: Request-Gebühr je 1.000 Anfragen von 5 USD (Low, Standard), 8 USD (Medium) oder 12 USD (High) abhängig von der Search-Context-Größe. Bild-Inputs werden als Input-Tokens abgerechnet.",
      "modalities_input": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_output": [
        "text",
        "image"
      ],
      "capabilities": [
        "structured_output",
        "web_search",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Kleines, schnelles und kostengünstiges Suchmodell für web-gestützte Antworten sowie einfache Frage-Antwort-Aufgaben.",
      "source_urls": [
        "https://docs.perplexity.ai/docs/sonar/models/sonar",
        "https://docs.perplexity.ai/docs/getting-started/pricing",
        "https://docs.perplexity.ai/docs/sonar/media",
        "https://docs.perplexity.ai/docs/resources/changelog"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-5.3",
      "api_id": "glm-5.3",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-08-18",
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 1.4,
      "cached_input_price": 0.26,
      "output_price": 4.4,
      "price_notes": "Standardtarif; Context-Cache-Speicherung laut Preisseite zeitlich begrenzt kostenlos.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "GLM-5.3 License",
      "description": "Das aktuelle Flaggschiff für anspruchsvolles Coding, lange agentische Aufgaben und steuerbare Reasoning-Tiefe.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/llm/glm-5.3",
        "https://docs.z.ai/api-reference/llm/chat-completion",
        "https://huggingface.co/zai-org/GLM-5.3",
        "https://github.com/zai-org/GLM-5"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-5.2",
      "api_id": "glm-5.2",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-06-16",
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 1.4,
      "cached_input_price": 0.26,
      "output_price": 4.4,
      "price_notes": "Standardtarif; Context-Cache-Speicherung laut Preisseite zeitlich begrenzt kostenlos.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT",
      "description": "Ein Flaggschiff für langlaufende Coding-Agenten mit 1-Mio.-Token-Kontext und wählbarem Reasoning-Aufwand.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/llm/glm-5.2",
        "https://docs.z.ai/api-reference/llm/chat-completion",
        "https://huggingface.co/zai-org/GLM-5.2",
        "https://github.com/zai-org/GLM-5"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-5.1",
      "api_id": "glm-5.1",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-04-07",
      "context_window": 200000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 1.4,
      "cached_input_price": 0.26,
      "output_price": 4.4,
      "price_notes": "Standardtarif; Context-Cache-Speicherung laut Preisseite zeitlich begrenzt kostenlos.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT",
      "description": "Ein weiterhin verfügbares GLM-5-Modell für langfristige agentische Engineering- und Coding-Aufgaben.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/overview/overview",
        "https://docs.z.ai/api-reference/llm/chat-completion",
        "https://huggingface.co/zai-org/GLM-5.1",
        "https://github.com/zai-org/GLM-5"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-5",
      "api_id": "glm-5",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-02-12",
      "context_window": 200000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 1,
      "cached_input_price": 0.2,
      "output_price": 3.2,
      "price_notes": "Standardtarif; Context-Cache-Speicherung laut Preisseite zeitlich begrenzt kostenlos.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT",
      "description": "Ein Open-Weights-Modell für komplexes System-Engineering, tiefes Debugging und langfristige Agentenaufgaben.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/overview/overview",
        "https://docs.z.ai/api-reference/llm/chat-completion",
        "https://huggingface.co/zai-org/GLM-5",
        "https://github.com/zai-org/GLM-5"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-4.7",
      "api_id": "glm-4.7",
      "tier": "coding",
      "status": "ga",
      "release_date": "2025-12-22",
      "context_window": 200000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 0.6,
      "cached_input_price": 0.11,
      "output_price": 2.2,
      "price_notes": "Standardtarif; Context-Cache-Speicherung laut Preisseite zeitlich begrenzt kostenlos.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT",
      "description": "Ein Open-Weights-Coding- und Agentenmodell mit verbessertem mehrstufigem Reasoning und Tool-Use.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/llm/glm-4.7",
        "https://docs.z.ai/api-reference/llm/chat-completion",
        "https://huggingface.co/zai-org/GLM-4.7",
        "https://github.com/zai-org/GLM-4.5"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-4.7-FlashX",
      "api_id": "glm-4.7-flashx",
      "tier": "small",
      "status": "ga",
      "release_date": null,
      "context_window": 200000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 0.07,
      "cached_input_price": 0.01,
      "output_price": 0.4,
      "price_notes": "Standardtarif; Context-Cache-Speicherung laut Preisseite zeitlich begrenzt kostenlos.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Die schnelle und günstige GLM-4.7-Variante für allgemeine Agentic-Coding-Aufgaben mit hohem Durchsatz.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/llm/glm-4.7",
        "https://docs.z.ai/api-reference/llm/chat-completion"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-4.7-Flash",
      "api_id": "glm-4.7-flash",
      "tier": "small",
      "status": "ga",
      "release_date": "2026-01-19",
      "context_window": 200000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 0,
      "cached_input_price": 0,
      "output_price": 0,
      "price_notes": "Kostenloses Modell laut Preisseite.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT",
      "description": "Ein kostenloses, kompaktes Open-Weights-Modell für latenzsensible Coding-, Reasoning- und Textaufgaben.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/llm/glm-4.7",
        "https://docs.z.ai/api-reference/llm/chat-completion",
        "https://huggingface.co/zai-org/GLM-4.7-Flash",
        "https://github.com/zai-org/GLM-4.5"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-5.3-Flash",
      "api_id": "glm-5.3-flash",
      "tier": "specialized",
      "status": "ga",
      "release_date": "2026-08-26",
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 0.075,
      "cached_input_price": 0.015,
      "output_price": 0.25,
      "price_notes": "Aktueller, bis 2026-09-09 um 24:00 Uhr UTC+8 befristet um 50 % reduzierter Tarif; reguläre Listenpreise sind 0,15 USD Input, 0,03 USD Cached Input und 0,50 USD Output pro 1 Mio. Tokens. Context-Cache-Speicherung ist zeitlich begrenzt kostenlos.",
      "modalities_input": [
        "text",
        "image",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT",
      "description": "Ein natives multimodales und besonders kosteneffizientes Modell für visuelles Coding, GUI-Aufgaben und lange Kontexte.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/vlm/glm-5.3-flash",
        "https://docs.z.ai/api-reference/llm/chat-completion",
        "https://huggingface.co/zai-org/GLM-5.3-Flash",
        "https://github.com/zai-org/GLM-5"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-4.6",
      "api_id": "glm-4.6",
      "tier": "coding",
      "status": "legacy",
      "release_date": "2025-09-30",
      "context_window": 200000,
      "max_output_tokens": 131072,
      "knowledge_cutoff": null,
      "input_price": 0.6,
      "cached_input_price": 0.11,
      "output_price": 2.2,
      "price_notes": "Standardtarif; Context-Cache-Speicherung laut Preisseite zeitlich begrenzt kostenlos.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT",
      "description": "Die weiterhin buchbare, letzte Vorgängergeneration für Coding, Reasoning und Tool-gestützte Agenten.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/overview/overview",
        "https://docs.z.ai/api-reference/llm/chat-completion",
        "https://huggingface.co/zai-org/GLM-4.6",
        "https://github.com/zai-org/GLM-4.5"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-4.6V",
      "api_id": "glm-4.6v",
      "tier": "specialized",
      "status": "legacy",
      "release_date": "2025-12-08",
      "context_window": 128000,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0.3,
      "cached_input_price": 0.05,
      "output_price": 0.9,
      "price_notes": "Standardtarif; Context-Cache-Speicherung laut Preisseite zeitlich begrenzt kostenlos.",
      "modalities_input": [
        "text",
        "image",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT",
      "description": "Ein Open-Weights-VLM für Bild-, Video-, Dokument- und GUI-Verständnis mit nativem multimodalem Function Calling.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/vlm/glm-4.6v",
        "https://docs.z.ai/api-reference/llm/chat-completion",
        "https://huggingface.co/zai-org/GLM-4.6V",
        "https://github.com/zai-org/GLM-V"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-4.6V-FlashX",
      "api_id": "glm-4.6v-flashx",
      "tier": "specialized",
      "status": "legacy",
      "release_date": "2025-12-08",
      "context_window": 128000,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0.04,
      "cached_input_price": 0.004,
      "output_price": 0.4,
      "price_notes": "Standardtarif; Context-Cache-Speicherung laut Preisseite zeitlich begrenzt kostenlos.",
      "modalities_input": [
        "text",
        "image",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Die schnelle, preisgünstige multimodale GLM-4.6V-Endpunktvariante für visuelle Analyse und Tool-Aufgaben.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/vlm/glm-4.6v",
        "https://docs.z.ai/api-reference/llm/chat-completion"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "zai",
      "provider": "Z.ai / Zhipu (GLM)",
      "name": "GLM-4.6V-Flash",
      "api_id": "glm-4.6v-flash",
      "tier": "specialized",
      "status": "legacy",
      "release_date": "2025-12-08",
      "context_window": 128000,
      "max_output_tokens": 32768,
      "knowledge_cutoff": null,
      "input_price": 0,
      "cached_input_price": 0,
      "output_price": 0,
      "price_notes": "Kostenloses Modell laut Preisseite.",
      "modalities_input": [
        "text",
        "image",
        "video",
        "pdf"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": true,
      "license": "MIT",
      "description": "Ein kostenloses, kompaktes Open-Weights-VLM für multimodales Verständnis und visuell gestützte Tool-Use-Szenarien.",
      "source_urls": [
        "https://docs.z.ai/guides/overview/pricing",
        "https://docs.z.ai/guides/vlm/glm-4.6v",
        "https://docs.z.ai/api-reference/llm/chat-completion",
        "https://huggingface.co/zai-org/GLM-4.6V-Flash",
        "https://github.com/zai-org/GLM-V"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "xai",
      "provider": "xAI (Grok)",
      "name": "Grok 4.6",
      "api_id": "grok-4.6",
      "tier": "flagship",
      "status": "ga",
      "release_date": "2026-08-12",
      "context_window": 500000,
      "max_output_tokens": null,
      "knowledge_cutoff": "2026-02-01",
      "input_price": 2.0,
      "cached_input_price": 0.5,
      "output_price": 6.0,
      "price_notes": "Ab 200.000 Prompt-Tokens gelten für alle Tokens des Requests 4,00 $ Input, 1,00 $ Cached Input und 12,00 $ Output pro 1 Mio. Tokens. Priority Processing kostet das Doppelte der Standardpreise; Batch wird nicht unterstützt.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Aktuelles Flaggschiff für Coding, agentische Aufgaben und Wissensarbeit mit konfigurierbarer Reasoning-Tiefe.",
      "source_urls": [
        "https://docs.x.ai/developers/models/grok-4.6",
        "https://docs.x.ai/developers/pricing",
        "https://docs.x.ai/developers/release-notes",
        "https://x.ai/news/grok-4-6"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "xai",
      "provider": "xAI (Grok)",
      "name": "Grok 4.3",
      "api_id": "grok-4.3",
      "tier": "standard",
      "status": "ga",
      "release_date": null,
      "context_window": 1000000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.25,
      "cached_input_price": 0.2,
      "output_price": 2.5,
      "price_notes": "Ab 200.000 Prompt-Tokens gelten 2,50 $ Input, 0,40 $ Cached Input und 5,00 $ Output pro 1 Mio. Tokens. Priority Processing kostet das Doppelte; die Batch API wird unterstützt und gewährt 20 % Rabatt.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Schnelles Standardmodell mit starkem Tool Calling, Instruction Following und abschalt- beziehungsweise konfigurierbarem Reasoning.",
      "source_urls": [
        "https://docs.x.ai/developers/models/grok-4.3",
        "https://docs.x.ai/developers/pricing",
        "https://docs.x.ai/developers/migration/may-15-retirement"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "xai",
      "provider": "xAI (Grok)",
      "name": "Grok Build 0.1",
      "api_id": "grok-build-0.1",
      "tier": "coding",
      "status": "preview",
      "release_date": "2026-05-19",
      "context_window": 256000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.0,
      "cached_input_price": 0.2,
      "output_price": 2.0,
      "price_notes": "Der Coding-Modellzugang befindet sich laut Release Notes weiterhin im Early Access. Ab 200.000 Prompt-Tokens gelten 2,00 $ Input, 0,40 $ Cached Input und 4,00 $ Output pro 1 Mio. Tokens; Priority Processing kostet das Doppelte, Batch wird nicht unterstützt.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Early-Access-Codingmodell für agentische Softwareentwicklung, Engineering- und Workflow-Aufgaben.",
      "source_urls": [
        "https://docs.x.ai/developers/models/grok-build-0.1",
        "https://docs.x.ai/developers/pricing",
        "https://docs.x.ai/developers/release-notes"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "xai",
      "provider": "xAI (Grok)",
      "name": "Grok 4.20 Multi-Agent Beta",
      "api_id": "grok-4.20-multi-agent",
      "tier": "specialized",
      "status": "preview",
      "release_date": "2026-03-10",
      "context_window": 1000000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.25,
      "cached_input_price": 0.2,
      "output_price": 2.5,
      "price_notes": "Die undatierte API-ID ist ein Alias des dokumentierten Snapshots `grok-4.20-multi-agent-0309`. Ab 200.000 Prompt-Tokens gelten 2,50 $ Input, 0,40 $ Cached Input und 5,00 $ Output pro 1 Mio. Tokens; Priority Processing kostet das Doppelte und Batch bietet 20 % Rabatt.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Beta-Spezialmodell, bei dem mehrere Agenten parallel zusammenarbeiten und ein Leader-Agent das Ergebnis synthetisiert.",
      "source_urls": [
        "https://docs.x.ai/developers/models/grok-4.20-multi-agent-0309",
        "https://docs.x.ai/developers/pricing",
        "https://docs.x.ai/developers/release-notes",
        "https://docs.x.ai/developers/model-capabilities/text/multi-agent"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "xai",
      "provider": "xAI (Grok)",
      "name": "Grok 4.5",
      "api_id": "grok-4.5",
      "tier": "flagship",
      "status": "legacy",
      "release_date": "2026-07-08",
      "context_window": 500000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 2.0,
      "cached_input_price": 0.3,
      "output_price": 6.0,
      "price_notes": "Weiterhin direkt buchbarer Vorgänger des Flaggschiffs. Ab 200.000 Prompt-Tokens gelten 4,00 $ Input, 0,60 $ Cached Input und 12,00 $ Output pro 1 Mio. Tokens; Priority Processing kostet das Doppelte, Batch wird nicht unterstützt.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Vorgänger-Flaggschiff für Coding, agentische Aufgaben und Wissensarbeit mit konfigurierbarem Reasoning.",
      "source_urls": [
        "https://docs.x.ai/developers/models/grok-4.5",
        "https://docs.x.ai/developers/pricing",
        "https://docs.x.ai/developers/release-notes"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "xai",
      "provider": "xAI (Grok)",
      "name": "Grok-1",
      "api_id": null,
      "tier": "flagship",
      "status": "legacy",
      "release_date": "2024-03-17",
      "context_window": 8192,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": null,
      "cached_input_price": null,
      "output_price": null,
      "price_notes": "Keine eigene gehostete xAI-API und daher keine tokenbasierten API-Preise; die Gewichte und Architektur wurden zur lokalen beziehungsweise selbst betriebenen Nutzung veröffentlicht.",
      "modalities_input": [
        "text"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [],
      "open_weights": true,
      "license": "Apache-2.0",
      "description": "Offen veröffentlichter 314B-MoE-Basiskontrollpunkt, der nicht für Dialoganwendungen nachtrainiert wurde.",
      "source_urls": [
        "https://x.ai/news/grok-os",
        "https://github.com/xai-org/grok-1"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "xai",
      "provider": "xAI (Grok)",
      "name": "Grok 4.20 (Reasoning)",
      "api_id": "grok-4.20-reasoning",
      "tier": "reasoning",
      "status": "legacy",
      "release_date": "2026-03-10",
      "context_window": 1000000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.25,
      "cached_input_price": 0.2,
      "output_price": 2.5,
      "price_notes": "Die undatierte API-ID ist ein Alias des aktuell dokumentierten Snapshots `grok-4.20-0309-reasoning`. Ab 200.000 Prompt-Tokens gelten 2,50 $ Input, 0,40 $ Cached Input und 5,00 $ Output pro 1 Mio. Tokens; Priority Processing kostet das Doppelte und Batch bietet 20 % Rabatt.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool_use",
        "structured_output",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Weiter buchbare Reasoning-Variante der Grok-4.20-Generation für präzise agentische und Tool-Calling-Aufgaben.",
      "source_urls": [
        "https://docs.x.ai/developers/models/grok-4.20",
        "https://docs.x.ai/developers/pricing",
        "https://docs.x.ai/developers/release-notes"
      ],
      "verified_in_api": null
    },
    {
      "provider_id": "xai",
      "provider": "xAI (Grok)",
      "name": "Grok 4.20 (Non-Reasoning)",
      "api_id": "grok-4.20-non-reasoning",
      "tier": "standard",
      "status": "legacy",
      "release_date": "2026-03-10",
      "context_window": 1000000,
      "max_output_tokens": null,
      "knowledge_cutoff": null,
      "input_price": 1.25,
      "cached_input_price": 0.2,
      "output_price": 2.5,
      "price_notes": "Die undatierte API-ID ist ein Alias des aktuell dokumentierten Snapshots `grok-4.20-0309-non-reasoning`. Ab 200.000 Prompt-Tokens gelten 2,50 $ Input, 0,40 $ Cached Input und 5,00 $ Output pro 1 Mio. Tokens; Priority Processing kostet das Doppelte und Batch bietet 20 % Rabatt.",
      "modalities_input": [
        "text",
        "image"
      ],
      "modalities_output": [
        "text"
      ],
      "capabilities": [
        "tool_use",
        "structured_output",
        "batch",
        "prompt_caching",
        "streaming"
      ],
      "open_weights": false,
      "license": null,
      "description": "Weiter buchbare Grok-4.20-Variante für Antworten ohne aktiviertes Reasoning.",
      "source_urls": [
        "https://docs.x.ai/developers/models/grok-4.20-beta-0309-non-reasoning",
        "https://docs.x.ai/developers/pricing",
        "https://docs.x.ai/developers/release-notes"
      ],
      "verified_in_api": null
    }
  ],
  "live_api_lists": {
    "openai": [
      "babbage-002",
      "chat-latest",
      "chatgpt-image-latest",
      "davinci-002",
      "gpt-3.5-turbo",
      "gpt-3.5-turbo-0125",
      "gpt-3.5-turbo-1106",
      "gpt-3.5-turbo-16k",
      "gpt-3.5-turbo-instruct",
      "gpt-3.5-turbo-instruct-0914",
      "gpt-4",
      "gpt-4-0613",
      "gpt-4-turbo",
      "gpt-4-turbo-2024-04-09",
      "gpt-4.1",
      "gpt-4.1-2025-04-14",
      "gpt-4.1-mini",
      "gpt-4.1-mini-2025-04-14",
      "gpt-4.1-nano",
      "gpt-4.1-nano-2025-04-14",
      "gpt-4o",
      "gpt-4o-2024-05-13",
      "gpt-4o-2024-08-06",
      "gpt-4o-2024-11-20",
      "gpt-4o-mini",
      "gpt-4o-mini-2024-07-18",
      "gpt-4o-mini-search-preview",
      "gpt-4o-mini-search-preview-2025-03-11",
      "gpt-4o-mini-transcribe",
      "gpt-4o-mini-transcribe-2025-03-20",
      "gpt-4o-mini-transcribe-2025-12-15",
      "gpt-4o-mini-tts",
      "gpt-4o-mini-tts-2025-03-20",
      "gpt-4o-mini-tts-2025-12-15",
      "gpt-4o-search-preview",
      "gpt-4o-search-preview-2025-03-11",
      "gpt-4o-transcribe",
      "gpt-4o-transcribe-diarize",
      "gpt-5",
      "gpt-5-2025-08-07",
      "gpt-5-chat-latest",
      "gpt-5-codex",
      "gpt-5-mini",
      "gpt-5-mini-2025-08-07",
      "gpt-5-nano",
      "gpt-5-nano-2025-08-07",
      "gpt-5-pro",
      "gpt-5-pro-2025-10-06",
      "gpt-5-search-api",
      "gpt-5-search-api-2025-10-14",
      "gpt-5.1",
      "gpt-5.1-2025-11-13",
      "gpt-5.1-chat-latest",
      "gpt-5.1-codex",
      "gpt-5.1-codex-max",
      "gpt-5.1-codex-mini",
      "gpt-5.2",
      "gpt-5.2-2025-12-11",
      "gpt-5.2-chat-latest",
      "gpt-5.2-codex",
      "gpt-5.2-pro",
      "gpt-5.2-pro-2025-12-11",
      "gpt-5.3-chat-latest",
      "gpt-5.3-codex",
      "gpt-5.4",
      "gpt-5.4-2026-03-05",
      "gpt-5.4-mini",
      "gpt-5.4-mini-2026-03-17",
      "gpt-5.4-nano",
      "gpt-5.4-nano-2026-03-17",
      "gpt-5.4-pro",
      "gpt-5.4-pro-2026-03-05",
      "gpt-5.5",
      "gpt-5.5-2026-04-23",
      "gpt-5.5-pro",
      "gpt-5.5-pro-2026-04-23",
      "gpt-5.6-luna",
      "gpt-5.6-sol",
      "gpt-5.6-terra",
      "gpt-audio",
      "gpt-audio-1.5",
      "gpt-audio-2025-08-28",
      "gpt-audio-mini",
      "gpt-audio-mini-2025-10-06",
      "gpt-audio-mini-2025-12-15",
      "gpt-image-1",
      "gpt-image-1-mini",
      "gpt-image-1.5",
      "gpt-image-2",
      "gpt-image-2-2026-04-21",
      "gpt-live-transcribe",
      "gpt-realtime",
      "gpt-realtime-1.5",
      "gpt-realtime-2",
      "gpt-realtime-2.1",
      "gpt-realtime-2.1-mini",
      "gpt-realtime-2025-08-28",
      "gpt-realtime-mini",
      "gpt-realtime-mini-2025-12-15",
      "gpt-realtime-translate",
      "gpt-realtime-whisper",
      "gpt-transcribe",
      "o1",
      "o1-2024-12-17",
      "o1-pro",
      "o1-pro-2025-03-19",
      "o3",
      "o3-2025-04-16",
      "o3-mini",
      "o3-mini-2025-01-31",
      "o4-mini",
      "o4-mini-2025-04-16",
      "omni-moderation-2024-09-26",
      "omni-moderation-latest",
      "sora-2",
      "sora-2-pro",
      "text-embedding-3-large",
      "text-embedding-3-small",
      "text-embedding-ada-002",
      "tts-1",
      "tts-1-1106",
      "tts-1-hd",
      "tts-1-hd-1106",
      "whisper-1"
    ],
    "anthropic": [
      "claude-fable-5",
      "claude-fable-5-1",
      "claude-haiku-4-5-20251001",
      "claude-opus-4-5-20251101",
      "claude-opus-4-6",
      "claude-opus-4-7",
      "claude-opus-4-8",
      "claude-opus-5",
      "claude-sonnet-4-5-20250929",
      "claude-sonnet-4-6",
      "claude-sonnet-5"
    ],
    "gemini": [
      "antigravity-preview-05-2026",
      "aqa",
      "deep-research-max-preview-04-2026",
      "deep-research-preview-04-2026",
      "deep-research-pro-preview-12-2025",
      "gemini-2.5-computer-use-preview-10-2025",
      "gemini-2.5-flash",
      "gemini-2.5-flash-image",
      "gemini-2.5-flash-lite",
      "gemini-2.5-flash-native-audio-latest",
      "gemini-2.5-flash-native-audio-preview-09-2025",
      "gemini-2.5-flash-native-audio-preview-12-2025",
      "gemini-2.5-flash-preview-tts",
      "gemini-2.5-pro",
      "gemini-2.5-pro-preview-tts",
      "gemini-3-flash-preview",
      "gemini-3-pro-image",
      "gemini-3-pro-image-preview",
      "gemini-3.1-flash-image",
      "gemini-3.1-flash-image-preview",
      "gemini-3.1-flash-lite",
      "gemini-3.1-flash-lite-image",
      "gemini-3.1-flash-lite-preview",
      "gemini-3.1-flash-live-preview",
      "gemini-3.1-flash-tts-preview",
      "gemini-3.1-pro-preview",
      "gemini-3.1-pro-preview-customtools",
      "gemini-3.5-flash",
      "gemini-3.5-flash-lite",
      "gemini-3.5-live-translate-preview",
      "gemini-3.5-transcribe",
      "gemini-3.5-transcribe-live",
      "gemini-3.6-flash",
      "gemini-3.7-flash",
      "gemini-3.8-flash",
      "gemini-embedding-001",
      "gemini-embedding-2",
      "gemini-embedding-2-preview",
      "gemini-flash-latest",
      "gemini-flash-lite-latest",
      "gemini-omni-1.1-flash",
      "gemini-omni-flash-preview",
      "gemini-pro-latest",
      "gemini-robotics-er-2-preview",
      "gemini-robotics-er-2-streaming-preview",
      "gemma-4-26b-a4b-it",
      "gemma-4-31b-it",
      "lyria-3-clip-preview",
      "lyria-3-pro-preview",
      "lyria-3.5",
      "nano-banana-pro-preview",
      "veo-3.1-fast-generate-preview",
      "veo-3.1-generate-preview",
      "veo-3.1-lite-generate-preview"
    ],
    "deepseek": [
      "deepseek-v4-flash",
      "deepseek-v4-flash-vision-exp",
      "deepseek-v4-pro"
    ],
    "moonshot": [
      "kimi-k2.6",
      "kimi-k2.7-code",
      "kimi-k2.7-code-highspeed",
      "kimi-k3"
    ]
  }
}