{
  "timestamp": 1735150222098,
  "models": [
    {
      "id": "google/gemini-pro-1.5-exp",
      "name": "Google: Gemini Pro 1.5 Experimental",
      "created": 1722470400,
      "description": "Gemini 1.5 Pro Experimental is a bleeding-edge version of the [Gemini 1.5 Pro](/models/google/gemini-pro-1.5) model. Because it's currently experimental, it will be **heavily rate-limited** by Google.\n\nUsage of Gemini is subject to Google's [Gemini Terms of Use](https://ai.google.dev/terms).\n\n#multimodal",
      "context_length": 1000000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Gemini",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0",
        "completion": "0",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 1000000,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "meta-llama/llama-3.2-11b-vision-instruct:free",
      "name": "Meta: Llama 3.2 11B Vision Instruct (free)",
      "created": 1727222400,
      "description": "Llama 3.2 11B Vision is a multimodal model with 11 billion parameters, designed to handle tasks combining visual and textual data. It excels in tasks such as image captioning and visual question answering, bridging the gap between language generation and visual reasoning. Pre-trained on a massive dataset of image-text pairs, it performs well in complex, high-accuracy image analysis.\n\nIts ability to integrate visual understanding with language processing makes it an ideal solution for industries requiring comprehensive visual-linguistic AI applications, such as content creation, AI-driven customer service, and research.\n\nClick here for the [original model card](https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/MODEL_CARD_VISION.md).\n\nUsage of this model is subject to [Meta's Acceptable Use Policy](https://www.llama.com/llama3/use-policy/).",
      "context_length": 131072,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0",
        "completion": "0",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 8192,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "google/gemini-flash-1.5-exp",
      "name": "Google: Gemini Flash 1.5 Experimental",
      "created": 1724803200,
      "description": "Gemini 1.5 Flash Experimental is an experimental version of the [Gemini 1.5 Flash](/models/google/gemini-flash-1.5) model.\n\nUsage of Gemini is subject to Google's [Gemini Terms of Use](https://ai.google.dev/terms).\n\n#multimodal\n\nNote: This model is experimental and not suited for production use-cases. It may be removed or redirected to another model in the future.",
      "context_length": 1000000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Gemini",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0",
        "completion": "0",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 1000000,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "google/gemini-flash-1.5-8b-exp",
      "name": "Google: Gemini Flash 1.5 8B Experimental",
      "created": 1724803200,
      "description": "Gemini Flash 1.5 8B Experimental is an experimental, 8B parameter version of the [Gemini Flash 1.5](/models/google/gemini-flash-1.5) model.\n\nUsage of Gemini is subject to Google's [Gemini Terms of Use](https://ai.google.dev/terms).\n\n#multimodal\n\nNote: This model is currently experimental and not suitable for production use-cases, and may be heavily rate-limited.",
      "context_length": 1000000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Gemini",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0",
        "completion": "0",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 1000000,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "microsoft/phi-3-medium-128k-instruct:free",
      "name": "Microsoft: Phi-3 Medium 128K Instruct (free)",
      "created": 1716508800,
      "description": "Phi-3 128K Medium is a powerful 14-billion parameter model designed for advanced language understanding, reasoning, and instruction following. Optimized through supervised fine-tuning and preference adjustments, it excels in tasks involving common sense, mathematics, logical reasoning, and code processing.\n\nAt time of release, Phi-3 Medium demonstrated state-of-the-art performance among lightweight models. In the MMLU-Pro eval, the model even comes close to a Llama3 70B level of performance.\n\nFor 4k context length, try [Phi-3 Medium 4K](/models/microsoft/phi-3-medium-4k-instruct).",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Other",
        "instruct_type": "phi3"
      },
      "pricing": {
        "prompt": "0",
        "completion": "0",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 8192,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "microsoft/phi-3-mini-128k-instruct:free",
      "name": "Microsoft: Phi-3 Mini 128K Instruct (free)",
      "created": 1716681600,
      "description": "Phi-3 Mini is a powerful 3.8B parameter model designed for advanced language understanding, reasoning, and instruction following. Optimized through supervised fine-tuning and preference adjustments, it excels in tasks involving common sense, mathematics, logical reasoning, and code processing.\n\nAt time of release, Phi-3 Medium demonstrated state-of-the-art performance among lightweight models. This model is static, trained on an offline dataset with an October 2023 cutoff date.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Other",
        "instruct_type": "phi3"
      },
      "pricing": {
        "prompt": "0",
        "completion": "0",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 8192,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "google/gemini-2.0-flash-thinking-exp:free",
      "name": "Google: Gemini 2.0 Flash Thinking Experimental (free)",
      "created": 1734650026,
      "description": "Gemini 2.0 Flash Thinking Mode is an experimental model that's trained to generate the \"thinking process\" the model goes through as part of its response. As a result, Thinking Mode is capable of stronger reasoning capabilities in its responses than the [base Gemini 2.0 Flash model](/google/gemini-2.0-flash-exp).",
      "context_length": 40000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Gemini",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0",
        "completion": "0",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 40000,
        "max_completion_tokens": 8000,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "google/gemini-2.0-flash-exp:free",
      "name": "Google: Gemini Flash 2.0 Experimental (free)",
      "created": 1733937523,
      "description": "Gemini Flash 2.0 offers a significantly faster time to first token (TTFT) compared to [Gemini Flash 1.5](google/gemini-flash-1.5), while maintaining quality on par with larger models like [Gemini Pro 1.5](google/gemini-pro-1.5). It introduces notable enhancements in multimodal understanding, coding capabilities, complex instruction following, and function calling. These advancements come together to deliver more seamless and robust agentic experiences.",
      "context_length": 1048576,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Gemini",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0",
        "completion": "0",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 1048576,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "meta-llama/llama-3.2-1b-instruct",
      "name": "Meta: Llama 3.2 1B Instruct",
      "created": 1727222400,
      "description": "Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate efficiently in low-resource environments while maintaining strong task performance.\n\nSupporting eight core languages and fine-tunable for more, Llama 1.3B is ideal for businesses or developers seeking lightweight yet powerful AI solutions that can operate in diverse multilingual settings without the high computational demand of larger models.\n\nClick here for the [original model card](https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/MODEL_CARD.md).\n\nUsage of this model is subject to [Meta's Acceptable Use Policy](https://www.llama.com/llama3/use-policy/).",
      "context_length": 131072,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0.00000001",
        "completion": "0.00000002",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 131072,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "meta-llama/llama-3.2-3b-instruct",
      "name": "Meta: Llama 3.2 3B Instruct",
      "created": 1727222400,
      "description": "Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it supports eight languages, including English, Spanish, and Hindi, and is adaptable for additional languages.\n\nTrained on 9 trillion tokens, the Llama 3.2 3B model excels in instruction-following, complex reasoning, and tool use. Its balanced performance makes it ideal for applications needing accuracy and efficiency in text generation across multilingual settings.\n\nClick here for the [original model card](https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/MODEL_CARD.md).\n\nUsage of this model is subject to [Meta's Acceptable Use Policy](https://www.llama.com/llama3/use-policy/).",
      "context_length": 131072,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0.000000018",
        "completion": "0.00000003",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 131072,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "meta-llama/llama-3.1-8b-instruct",
      "name": "Meta: Llama 3.1 8B Instruct",
      "created": 1721692800,
      "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient.\n\nIt has demonstrated strong performance compared to leading closed-source models in human evaluations.\n\nTo read more about the model release, [click here](https://ai.meta.com/blog/meta-llama-3-1/). Usage of this model is subject to [Meta's Acceptable Use Policy](https://llama.meta.com/llama3/use-policy/).",
      "context_length": 131072,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0.00000002",
        "completion": "0.00000005",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 131072,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mistral-7b-instruct",
      "name": "Mistral: Mistral 7B Instruct",
      "created": 1716768000,
      "description": "A high-performing, industry-standard 7.3B parameter model, with optimizations for speed and context length.\n\n*Mistral 7B Instruct has multiple version variants, and this is intended to be the latest version.*",
      "context_length": 32768,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": "mistral"
      },
      "pricing": {
        "prompt": "0.00000003",
        "completion": "0.000000055",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32768,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mistral-7b-instruct-v0.3",
      "name": "Mistral: Mistral 7B Instruct v0.3",
      "created": 1716768000,
      "description": "A high-performing, industry-standard 7.3B parameter model, with optimizations for speed and context length.\n\nAn improved version of [Mistral 7B Instruct v0.2](/models/mistralai/mistral-7b-instruct-v0.2), with the following changes:\n\n- Extended vocabulary to 32768\n- Supports v3 Tokenizer\n- Supports function calling\n\nNOTE: Support for function calling depends on the provider.",
      "context_length": 32768,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": "mistral"
      },
      "pricing": {
        "prompt": "0.00000003",
        "completion": "0.000000055",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32768,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "meta-llama/llama-3-8b-instruct",
      "name": "Meta: Llama 3 8B Instruct",
      "created": 1713398400,
      "description": "Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 8B instruct-tuned version was optimized for high quality dialogue usecases.\n\nIt has demonstrated strong performance compared to leading closed-source models in human evaluations.\n\nTo read more about the model release, [click here](https://ai.meta.com/blog/meta-llama-3/). Usage of this model is subject to [Meta's Acceptable Use Policy](https://llama.meta.com/llama3/use-policy/).",
      "context_length": 8192,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0.00000003",
        "completion": "0.00000006",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 8192,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "amazon/nova-micro-v1",
      "name": "Amazon: Nova Micro 1.0",
      "created": 1733437237,
      "description": "Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length of 128K tokens and optimized for speed and cost, Amazon Nova Micro excels at tasks such as text summarization, translation, content classification, interactive chat, and brainstorming. It has  simple mathematical reasoning and coding abilities.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Nova",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000000035",
        "completion": "0.00000014",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 5120,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "google/gemini-flash-1.5-8b",
      "name": "Google: Gemini Flash 1.5 8B",
      "created": 1727913600,
      "description": "Gemini Flash 1.5 8B is optimized for speed and efficiency, offering enhanced performance in small prompt tasks like chat, transcription, and translation. With reduced latency, it is highly effective for real-time and large-scale operations. This model focuses on cost-effective solutions while maintaining high-quality results.\n\n[Click here to learn more about this model](https://developers.googleblog.com/en/gemini-15-flash-8b-is-now-generally-available-for-use/).\n\nUsage of Gemini is subject to Google's [Gemini Terms of Use](https://ai.google.dev/terms).",
      "context_length": 1000000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Gemini",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000000375",
        "completion": "0.00000015",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 1000000,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/ministral-3b",
      "name": "Mistral: Ministral 3B",
      "created": 1729123200,
      "description": "Ministral 3B is a 3B parameter model optimized for on-device and edge computing. It excels in knowledge, commonsense reasoning, and function-calling, outperforming larger models like Mistral 7B on most benchmarks. Supporting up to 128k context length, it’s ideal for orchestrating agentic workflows and specialist tasks with efficient inference.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000004",
        "completion": "0.00000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "meta-llama/llama-3.2-11b-vision-instruct",
      "name": "Meta: Llama 3.2 11B Vision Instruct",
      "created": 1727222400,
      "description": "Llama 3.2 11B Vision is a multimodal model with 11 billion parameters, designed to handle tasks combining visual and textual data. It excels in tasks such as image captioning and visual question answering, bridging the gap between language generation and visual reasoning. Pre-trained on a massive dataset of image-text pairs, it performs well in complex, high-accuracy image analysis.\n\nIts ability to integrate visual understanding with language processing makes it an ideal solution for industries requiring comprehensive visual-linguistic AI applications, such as content creation, AI-driven customer service, and research.\n\nClick here for the [original model card](https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/MODEL_CARD_VISION.md).\n\nUsage of this model is subject to [Meta's Acceptable Use Policy](https://www.llama.com/llama3/use-policy/).",
      "context_length": 131072,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0.000000055",
        "completion": "0.000000055",
        "image": "0.00007948",
        "request": "0"
      },
      "top_provider": {
        "context_length": 131072,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "amazon/nova-lite-v1",
      "name": "Amazon: Nova Lite 1.0",
      "created": 1733437363,
      "description": "Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite can handle real-time customer interactions, document analysis, and visual question-answering tasks with high accuracy.\n\nWith an input context of 300K tokens, it can analyze multiple images or up to 30 minutes of video in a single input.",
      "context_length": 300000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Nova",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000006",
        "completion": "0.00000024",
        "image": "0.00009",
        "request": "0"
      },
      "top_provider": {
        "context_length": 300000,
        "max_completion_tokens": 5120,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "google/gemini-flash-1.5",
      "name": "Google: Gemini Flash 1.5",
      "created": 1715644800,
      "description": "Gemini 1.5 Flash is a foundation model that performs well at a variety of multimodal tasks such as visual understanding, classification, summarization, and creating content from image, audio and video. It's adept at processing visual and text inputs such as photographs, documents, infographics, and screenshots.\n\nGemini 1.5 Flash is designed for high-volume, high-frequency tasks where cost and latency matter. On most common tasks, Flash achieves comparable quality to other Gemini Pro models at a significantly reduced cost. Flash is well-suited for applications like chat assistants and on-demand content generation where speed and scale matter.\n\nUsage of Gemini is subject to Google's [Gemini Terms of Use](https://ai.google.dev/terms).\n\n#multimodal",
      "context_length": 1000000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Gemini",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000000075",
        "completion": "0.0000003",
        "image": "0.00004",
        "request": "0"
      },
      "top_provider": {
        "context_length": 1000000,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/ministral-8b",
      "name": "Mistral: Ministral 8B",
      "created": 1729123200,
      "description": "Ministral 8B is an 8B parameter model featuring a unique interleaved sliding-window attention pattern for faster, memory-efficient inference. Designed for edge use cases, it supports up to 128k context length and excels in knowledge and reasoning tasks. It outperforms peers in the sub-10B category, making it perfect for low-latency, privacy-first applications.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000001",
        "completion": "0.0000001",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "microsoft/phi-3-mini-128k-instruct",
      "name": "Microsoft: Phi-3 Mini 128K Instruct",
      "created": 1716681600,
      "description": "Phi-3 Mini is a powerful 3.8B parameter model designed for advanced language understanding, reasoning, and instruction following. Optimized through supervised fine-tuning and preference adjustments, it excels in tasks involving common sense, mathematics, logical reasoning, and code processing.\n\nAt time of release, Phi-3 Medium demonstrated state-of-the-art performance among lightweight models. This model is static, trained on an offline dataset with an October 2023 cutoff date.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Other",
        "instruct_type": "phi3"
      },
      "pricing": {
        "prompt": "0.0000001",
        "completion": "0.0000001",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "microsoft/phi-3.5-mini-128k-instruct",
      "name": "Microsoft: Phi-3.5 Mini 128K Instruct",
      "created": 1724198400,
      "description": "Phi-3.5 models are lightweight, state-of-the-art open models. These models were trained with Phi-3 datasets that include both synthetic data and the filtered, publicly available websites data, with a focus on high quality and reasoning-dense properties. Phi-3.5 Mini uses 3.8B parameters, and is a dense decoder-only transformer model using the same tokenizer as [Phi-3 Mini](/models/microsoft/phi-3-mini-128k-instruct).\n\nThe models underwent a rigorous enhancement process, incorporating both supervised fine-tuning, proximal policy optimization, and direct preference optimization to ensure precise instruction adherence and robust safety measures. When assessed against benchmarks that test common sense, language understanding, math, code, long context and logical reasoning, Phi-3.5 models showcased robust and state-of-the-art performance among models with less than 13 billion parameters.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Other",
        "instruct_type": "phi3"
      },
      "pricing": {
        "prompt": "0.0000001",
        "completion": "0.0000001",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "meta-llama/llama-3.1-70b-instruct",
      "name": "Meta: Llama 3.1 70B Instruct",
      "created": 1721692800,
      "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases.\n\nIt has demonstrated strong performance compared to leading closed-source models in human evaluations.\n\nTo read more about the model release, [click here](https://ai.meta.com/blog/meta-llama-3-1/). Usage of this model is subject to [Meta's Acceptable Use Policy](https://llama.meta.com/llama3/use-policy/).",
      "context_length": 131072,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0.00000013",
        "completion": "0.0000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 131072,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "deepseek/deepseek-chat",
      "name": "DeepSeek V2.5",
      "created": 1715644800,
      "description": "DeepSeek-V2.5 is an upgraded version that combines DeepSeek-V2-Chat and DeepSeek-Coder-V2-Instruct. The new model integrates the general and coding abilities of the two previous versions. For model details, please visit [DeepSeek-V2 page](https://github.com/deepseek-ai/DeepSeek-V2) for more information.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Other",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000014",
        "completion": "0.00000028",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 65536,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "cohere/command-r-08-2024",
      "name": "Cohere: Command R (08-2024)",
      "created": 1724976000,
      "description": "command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and is competitive with the previous version of the larger Command R+ model.\n\nRead the launch post [here](https://docs.cohere.com/changelog/command-gets-refreshed).\n\nUse of this model is subject to Cohere's [Acceptable Use Policy](https://docs.cohere.com/docs/c4ai-acceptable-use-policy).",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Cohere",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000001425",
        "completion": "0.00000057",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 4000,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mistral-nemo",
      "name": "Mistral: Mistral Nemo",
      "created": 1721347200,
      "description": "A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA.\n\nThe model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese, Korean, Arabic, and Hindi.\n\nIt supports function calling and is released under the Apache 2.0 license.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": "mistral"
      },
      "pricing": {
        "prompt": "0.00000015",
        "completion": "0.00000015",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/pixtral-12b",
      "name": "Mistral: Pixtral 12B",
      "created": 1725926400,
      "description": "The first multi-modal, text+image-to-text model from Mistral AI. Its weights were launched via torrent: https://x.com/mistralai/status/1833758285167722836.",
      "context_length": 4096,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000015",
        "completion": "0.00000015",
        "image": "0.0002168",
        "request": "0"
      },
      "top_provider": {
        "context_length": 4096,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4o-mini",
      "name": "OpenAI: GPT-4o-mini",
      "created": 1721260800,
      "description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs.\n\nAs their most advanced small model, it is many multiples more affordable than other recent frontier models, and more than 60% cheaper than [GPT-3.5 Turbo](/models/openai/gpt-3.5-turbo). It maintains SOTA intelligence, while being significantly more cost-effective.\n\nGPT-4o mini achieves an 82% score on MMLU and presently ranks higher than GPT-4 on chat preferences [common leaderboards](https://arena.lmsys.org/).\n\nCheck out the [launch announcement](https://openai.com/index/gpt-4o-mini-advancing-cost-efficient-intelligence/) to learn more.\n\n#multimodal",
      "context_length": 128000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000015",
        "completion": "0.0000006",
        "image": "0.007225",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 16384,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4o-mini-2024-07-18",
      "name": "OpenAI: GPT-4o-mini (2024-07-18)",
      "created": 1721260800,
      "description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs.\n\nAs their most advanced small model, it is many multiples more affordable than other recent frontier models, and more than 60% cheaper than [GPT-3.5 Turbo](/models/openai/gpt-3.5-turbo). It maintains SOTA intelligence, while being significantly more cost-effective.\n\nGPT-4o mini achieves an 82% score on MMLU and presently ranks higher than GPT-4 on chat preferences [common leaderboards](https://arena.lmsys.org/).\n\nCheck out the [launch announcement](https://openai.com/index/gpt-4o-mini-advancing-cost-efficient-intelligence/) to learn more.\n\n#multimodal",
      "context_length": 128000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000015",
        "completion": "0.0000006",
        "image": "0.007225",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 16384,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mistral-7b-instruct-v0.1",
      "name": "Mistral: Mistral 7B Instruct v0.1",
      "created": 1695859200,
      "description": "A 7.3B parameter model that outperforms Llama 2 13B on all benchmarks, with optimizations for speed and context length.",
      "context_length": 4096,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": "mistral"
      },
      "pricing": {
        "prompt": "0.00000018",
        "completion": "0.00000018",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 4096,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "ai21/jamba-1-5-mini",
      "name": "AI21: Jamba 1.5 Mini",
      "created": 1724371200,
      "description": "Jamba 1.5 Mini is the world's first production-grade Mamba-based model, combining SSM and Transformer architectures for a 256K context window and high efficiency.\n\nIt works with 9 languages and can handle various writing and analysis tasks as well as or better than similar small models.\n\nThis model uses less computer memory and works faster with longer texts than previous designs.\n\nRead their [announcement](https://www.ai21.com/blog/announcing-jamba-model-family) to learn more.",
      "context_length": 256000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Other",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000002",
        "completion": "0.0000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 256000,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mistral-small",
      "name": "Mistral Small",
      "created": 1704844800,
      "description": "With 22 billion parameters, Mistral Small v24.09 offers a convenient mid-point between (Mistral NeMo 12B)[/mistralai/mistral-nemo] and (Mistral Large 2)[/mistralai/mistral-large], providing a cost-effective solution that can be deployed across various platforms and environments. It has better reasoning, exhibits more capabilities, can produce and reason about code, and is multiligual, supporting English, French, German, Italian, and Spanish.",
      "context_length": 32000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000002",
        "completion": "0.0000006",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "qwen/qwen-2.5-72b-instruct",
      "name": "Qwen2.5 72B Instruct",
      "created": 1726704000,
      "description": "Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2:\n\n- Significantly more knowledge and has greatly improved capabilities in coding and mathematics, thanks to our specialized expert models in these domains.\n\n- Significant improvements in instruction following, generating long texts (over 8K tokens), understanding structured data (e.g, tables), and generating structured outputs especially JSON. More resilient to the diversity of system prompts, enhancing role-play implementation and condition-setting for chatbots.\n\n- Long-context Support up to 128K tokens and can generate up to 8K tokens.\n\n- Multilingual support for over 29 languages, including Chinese, English, French, Spanish, Portuguese, German, Italian, Russian, Japanese, Korean, Vietnamese, Thai, Arabic, and more.\n\nUsage of this model is subject to [Tongyi Qianwen LICENSE AGREEMENT](https://huggingface.co/Qwen/Qwen1.5-110B-Chat/blob/main/LICENSE).",
      "context_length": 131072,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Qwen",
        "instruct_type": "chatml"
      },
      "pricing": {
        "prompt": "0.00000023",
        "completion": "0.0000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32000,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "nvidia/llama-3.1-nemotron-70b-instruct",
      "name": "NVIDIA: Llama 3.1 Nemotron 70B Instruct",
      "created": 1728950400,
      "description": "NVIDIA's Llama 3.1 Nemotron 70B is a language model designed for generating precise and useful responses. Leveraging [Llama 3.1 70B](/models/meta-llama/llama-3.1-70b-instruct) architecture and Reinforcement Learning from Human Feedback (RLHF), it excels in automatic alignment benchmarks. This model is tailored for applications requiring high accuracy in helpfulness and response generation, suitable for diverse user queries across multiple domains.\n\nUsage of this model is subject to [Meta's Acceptable Use Policy](https://www.llama.com/llama3/use-policy/).",
      "context_length": 131072,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0.00000023",
        "completion": "0.0000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 131072,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "meta-llama/llama-3-70b-instruct",
      "name": "Meta: Llama 3 70B Instruct",
      "created": 1713398400,
      "description": "Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 70B instruct-tuned version was optimized for high quality dialogue usecases.\n\nIt has demonstrated strong performance compared to leading closed-source models in human evaluations.\n\nTo read more about the model release, [click here](https://ai.meta.com/blog/meta-llama-3/). Usage of this model is subject to [Meta's Acceptable Use Policy](https://llama.meta.com/llama3/use-policy/).",
      "context_length": 8192,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0.00000023",
        "completion": "0.0000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 8192,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mixtral-8x7b-instruct",
      "name": "Mistral: Mixtral 8x7B Instruct",
      "created": 1702166400,
      "description": "Mixtral 8x7B Instruct is a pretrained generative Sparse Mixture of Experts, by Mistral AI, for chat and instruction use. Incorporates 8 experts (feed-forward networks) for a total of 47 billion parameters.\n\nInstruct model fine-tuned by Mistral. #moe",
      "context_length": 32768,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": "mistral"
      },
      "pricing": {
        "prompt": "0.00000024",
        "completion": "0.00000024",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32768,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mistral-tiny",
      "name": "Mistral Tiny",
      "created": 1704844800,
      "description": "This model is currently powered by Mistral-7B-v0.2, and incorporates a \"better\" fine-tuning than [Mistral 7B](/models/mistralai/mistral-7b-instruct-v0.1), inspired by community work. It's best used for large batch processing tasks where cost is a significant factor but reasoning capabilities are not crucial.",
      "context_length": 32000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000025",
        "completion": "0.00000025",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/codestral-mamba",
      "name": "Mistral: Codestral Mamba",
      "created": 1721347200,
      "description": "A 7.3B parameter Mamba-based model designed for code and reasoning tasks.\n\n- Linear time inference, allowing for theoretically infinite sequence lengths\n- 256k token context window\n- Optimized for quick responses, especially beneficial for code productivity\n- Performs comparably to state-of-the-art transformer models in code and reasoning tasks\n- Available under the Apache 2.0 license for free use, modification, and distribution",
      "context_length": 256000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000025",
        "completion": "0.00000025",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 256000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3-haiku:beta",
      "name": "Anthropic: Claude 3 Haiku (self-moderated)",
      "created": 1710288000,
      "description": "Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000025",
        "completion": "0.00000125",
        "image": "0.0004",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3-haiku",
      "name": "Anthropic: Claude 3 Haiku",
      "created": 1710288000,
      "description": "Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000025",
        "completion": "0.00000125",
        "image": "0.0004",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "nousresearch/hermes-3-llama-3.1-70b",
      "name": "Nous: Hermes 3 70B Instruct",
      "created": 1723939200,
      "description": "Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the board.\n\nHermes 3 70B is a competitive, if not superior finetune of the [Llama-3.1 70B foundation model](/models/meta-llama/llama-3.1-70b-instruct), focused on aligning LLMs to the user, with powerful steering capabilities and control given to the end user.\n\nThe Hermes 3 series builds and expands on the Hermes 2 set of capabilities, including more powerful and reliable function calling and structured output capabilities, generalist assistant capabilities, and improved code generation skills.",
      "context_length": 131072,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Llama3",
        "instruct_type": "chatml"
      },
      "pricing": {
        "prompt": "0.0000004",
        "completion": "0.0000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 12288,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "cohere/command-r-03-2024",
      "name": "Cohere: Command R (03-2024)",
      "created": 1709341200,
      "description": "Command-R is a 35B parameter model that performs conversational language tasks at a higher quality, more reliably, and with a longer context than previous models. It can be used for complex workflows like code generation, retrieval augmented generation (RAG), tool use, and agents.\n\nRead the launch post [here](https://txt.cohere.com/command-r/).\n\nUse of this model is subject to Cohere's [Acceptable Use Policy](https://docs.cohere.com/docs/c4ai-acceptable-use-policy).",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Cohere",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000000475",
        "completion": "0.000001425",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 4000,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "cohere/command-r",
      "name": "Cohere: Command R",
      "created": 1710374400,
      "description": "Command-R is a 35B parameter model that performs conversational language tasks at a higher quality, more reliably, and with a longer context than previous models. It can be used for complex workflows like code generation, retrieval augmented generation (RAG), tool use, and agents.\n\nRead the launch post [here](https://txt.cohere.com/command-r/).\n\nUse of this model is subject to Cohere's [Acceptable Use Policy](https://docs.cohere.com/docs/c4ai-acceptable-use-policy).",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Cohere",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000000475",
        "completion": "0.000001425",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 4000,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-3.5-turbo-0125",
      "name": "OpenAI: GPT-3.5 Turbo 16k",
      "created": 1685232000,
      "description": "The latest GPT-3.5 Turbo model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Sep 2021.\n\nThis version has a higher accuracy at responding in requested formats and a fix for a bug which caused a text encoding issue for non-English language function calls.",
      "context_length": 16385,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000005",
        "completion": "0.0000015",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 16385,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "google/gemini-pro",
      "name": "Google: Gemini Pro 1.0",
      "created": 1702425600,
      "description": "Google's flagship text generation model. Designed to handle natural language tasks, multiturn text and code chat, and code generation.\n\nSee the benchmarks and prompting guidelines from [Deepmind](https://deepmind.google/technologies/gemini/).\n\nUsage of Gemini is subject to Google's [Gemini Terms of Use](https://ai.google.dev/terms).",
      "context_length": 32760,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Gemini",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000005",
        "completion": "0.0000015",
        "image": "0.0025",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32760,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-3.5-turbo",
      "name": "OpenAI: GPT-3.5 Turbo",
      "created": 1685232000,
      "description": "GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.",
      "context_length": 16385,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000005",
        "completion": "0.0000015",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 16385,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mixtral-8x7b-instruct:nitro",
      "name": "Mistral: Mixtral 8x7B Instruct (nitro)",
      "created": 1702166400,
      "description": "Mixtral 8x7B Instruct is a pretrained generative Sparse Mixture of Experts, by Mistral AI, for chat and instruction use. Incorporates 8 experts (feed-forward networks) for a total of 47 billion parameters.\n\nInstruct model fine-tuned by Mistral. #moe",
      "context_length": 32768,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": "mistral"
      },
      "pricing": {
        "prompt": "0.00000054",
        "completion": "0.00000054",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32768,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "amazon/nova-pro-v1",
      "name": "Amazon: Nova Pro 1.0",
      "created": 1733436303,
      "description": "Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December 2024, it achieves state-of-the-art performance on key benchmarks including visual question answering (TextVQA) and video understanding (VATEX).\n\nAmazon Nova Pro demonstrates strong capabilities in processing both visual and textual information and at analyzing financial documents.\n\n**NOTE**: Video input is not supported at this time.",
      "context_length": 300000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Nova",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000008",
        "completion": "0.0000032",
        "image": "0.0012",
        "request": "0"
      },
      "top_provider": {
        "context_length": 300000,
        "max_completion_tokens": 5120,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3.5-haiku:beta",
      "name": "Anthropic: Claude 3.5 Haiku (self-moderated)",
      "created": 1730678400,
      "description": "Claude 3.5 Haiku features offers enhanced capabilities in speed, coding accuracy, and tool use. Engineered to excel in real-time applications, it delivers quick response times that are essential for dynamic tasks such as chat interactions and immediate coding suggestions.\n\nThis makes it highly suitable for environments that demand both speed and precision, such as software development, customer service bots, and data management systems.\n\nThis model is currently pointing to [Claude 3.5 Haiku (2024-10-22)](/anthropic/claude-3-5-haiku-20241022).",
      "context_length": 200000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000008",
        "completion": "0.000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3.5-haiku",
      "name": "Anthropic: Claude 3.5 Haiku",
      "created": 1730678400,
      "description": "Claude 3.5 Haiku features offers enhanced capabilities in speed, coding accuracy, and tool use. Engineered to excel in real-time applications, it delivers quick response times that are essential for dynamic tasks such as chat interactions and immediate coding suggestions.\n\nThis makes it highly suitable for environments that demand both speed and precision, such as software development, customer service bots, and data management systems.\n\nThis model is currently pointing to [Claude 3.5 Haiku (2024-10-22)](/anthropic/claude-3-5-haiku-20241022).",
      "context_length": 200000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000008",
        "completion": "0.000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 8192,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3.5-haiku-20241022:beta",
      "name": "Anthropic: Claude 3.5 Haiku (2024-10-22) (self-moderated)",
      "created": 1730678400,
      "description": "Claude 3.5 Haiku features enhancements across all skill sets including coding, tool use, and reasoning. As the fastest model in the Anthropic lineup, it offers rapid response times suitable for applications that require high interactivity and low latency, such as user-facing chatbots and on-the-fly code completions. It also excels in specialized tasks like data extraction and real-time content moderation, making it a versatile tool for a broad range of industries.\n\nIt does not support image inputs.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/3-5-models-and-computer-use)",
      "context_length": 200000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000008",
        "completion": "0.000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3.5-haiku-20241022",
      "name": "Anthropic: Claude 3.5 Haiku (2024-10-22)",
      "created": 1730678400,
      "description": "Claude 3.5 Haiku features enhancements across all skill sets including coding, tool use, and reasoning. As the fastest model in the Anthropic lineup, it offers rapid response times suitable for applications that require high interactivity and low latency, such as user-facing chatbots and on-the-fly code completions. It also excels in specialized tasks like data extraction and real-time content moderation, making it a versatile tool for a broad range of industries.\n\nIt does not support image inputs.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/3-5-models-and-computer-use)",
      "context_length": 200000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000008",
        "completion": "0.000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 8192,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "meta-llama/llama-3.1-405b-instruct",
      "name": "Meta: Llama 3.1 405B Instruct",
      "created": 1721692800,
      "description": "The highly anticipated 400B class of Llama3 is here! Clocking in at 128k context with impressive eval scores, the Meta AI team continues to push the frontier of open-source LLMs.\n\nMeta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 405B instruct-tuned version is optimized for high quality dialogue usecases.\n\nIt has demonstrated strong performance compared to leading closed-source models including GPT-4o and Claude 3.5 Sonnet in evaluations.\n\nTo read more about the model release, [click here](https://ai.meta.com/blog/meta-llama-3-1/). Usage of this model is subject to [Meta's Acceptable Use Policy](https://llama.meta.com/llama3/use-policy/).",
      "context_length": 131072,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0.0000009",
        "completion": "0.0000009",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32000,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "microsoft/phi-3-medium-128k-instruct",
      "name": "Microsoft: Phi-3 Medium 128K Instruct",
      "created": 1716508800,
      "description": "Phi-3 128K Medium is a powerful 14-billion parameter model designed for advanced language understanding, reasoning, and instruction following. Optimized through supervised fine-tuning and preference adjustments, it excels in tasks involving common sense, mathematics, logical reasoning, and code processing.\n\nAt time of release, Phi-3 Medium demonstrated state-of-the-art performance among lightweight models. In the MMLU-Pro eval, the model even comes close to a Llama3 70B level of performance.\n\nFor 4k context length, try [Phi-3 Medium 4K](/models/microsoft/phi-3-medium-4k-instruct).",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Other",
        "instruct_type": "phi3"
      },
      "pricing": {
        "prompt": "0.000001",
        "completion": "0.000001",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-3.5-turbo-1106",
      "name": "OpenAI: GPT-3.5 Turbo 16k (older v1106)",
      "created": 1699228800,
      "description": "An older GPT-3.5 Turbo model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Sep 2021.",
      "context_length": 16385,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000001",
        "completion": "0.000002",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 16385,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-3.5-turbo-0613",
      "name": "OpenAI: GPT-3.5 Turbo (older v0613)",
      "created": 1706140800,
      "description": "GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.",
      "context_length": 4095,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000001",
        "completion": "0.000002",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 4095,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "meta-llama/llama-3.2-90b-vision-instruct",
      "name": "Meta: Llama 3.2 90B Vision Instruct",
      "created": 1727222400,
      "description": "The Llama 90B Vision model is a top-tier, 90-billion-parameter multimodal model designed for the most challenging visual reasoning and language tasks. It offers unparalleled accuracy in image captioning, visual question answering, and advanced image-text comprehension. Pre-trained on vast multimodal datasets and fine-tuned with human feedback, the Llama 90B Vision is engineered to handle the most demanding image-based AI tasks.\n\nThis model is perfect for industries requiring cutting-edge multimodal AI capabilities, particularly those dealing with complex, real-time visual and textual analysis.\n\nClick here for the [original model card](https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/MODEL_CARD_VISION.md).\n\nUsage of this model is subject to [Meta's Acceptable Use Policy](https://www.llama.com/llama3/use-policy/).",
      "context_length": 131072,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Llama3",
        "instruct_type": "llama3"
      },
      "pricing": {
        "prompt": "0.00000108",
        "completion": "0.00000108",
        "image": "0.0015606",
        "request": "0"
      },
      "top_provider": {
        "context_length": 131072,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "google/gemini-pro-1.5",
      "name": "Google: Gemini Pro 1.5",
      "created": 1712620800,
      "description": "Google's latest multimodal model, supports image and video[0] in text or chat prompts.\n\nOptimized for language tasks including:\n\n- Code generation\n- Text generation\n- Text editing\n- Problem solving\n- Recommendations\n- Information extraction\n- Data extraction or generation\n- AI agents\n\nUsage of Gemini is subject to Google's [Gemini Terms of Use](https://ai.google.dev/terms).\n\n* [0]: Video input is not available through OpenRouter at this time.",
      "context_length": 2000000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Gemini",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000125",
        "completion": "0.000005",
        "image": "0.0006575",
        "request": "0"
      },
      "top_provider": {
        "context_length": 2000000,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mixtral-8x22b-instruct",
      "name": "Mistral: Mixtral 8x22B Instruct",
      "created": 1713312000,
      "description": "Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include:\n- strong math, coding, and reasoning\n- large context length (64k)\n- fluency in English, French, Italian, German, and Spanish\n\nSee benchmarks on the launch announcement [here](https://mistral.ai/news/mixtral-8x22b/).\n#moe",
      "context_length": 65536,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": "mistral"
      },
      "pricing": {
        "prompt": "0.000002",
        "completion": "0.000006",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 65536,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mistral-large",
      "name": "Mistral Large",
      "created": 1708905600,
      "description": "This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/).\n\nIt supports dozens of languages including French, German, Spanish, Italian, Portuguese, Arabic, Hindi, Russian, Chinese, Japanese, and Korean, along with 80+ coding languages including Python, Java, C, C++, JavaScript, and Bash. Its long context window allows precise information recall from large documents.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000002",
        "completion": "0.000006",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mistral-large-2407",
      "name": "Mistral Large 2407",
      "created": 1731978415,
      "description": "This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/).\n\nIt supports dozens of languages including French, German, Spanish, Italian, Portuguese, Arabic, Hindi, Russian, Chinese, Japanese, and Korean, along with 80+ coding languages including Python, Java, C, C++, JavaScript, and Bash. Its long context window allows precise information recall from large documents.\n",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000002",
        "completion": "0.000006",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mistral-large-2411",
      "name": "Mistral Large 2411",
      "created": 1731978685,
      "description": "Mistral Large 2 2411 is an update of [Mistral Large 2](/mistralai/mistral-large) released together with [Pixtral Large 2411](/mistralai/pixtral-large-2411)\n\nIt provides a significant upgrade on the previous [Mistral Large 24.07](/mistralai/mistral-large-2407), with notable improvements in long context understanding, a new system prompt, and more accurate function calling.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000002",
        "completion": "0.000006",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/pixtral-large-2411",
      "name": "Mistral: Pixtral Large 2411",
      "created": 1731977388,
      "description": "Pixtral Large is a 124B parameter, open-weight, multimodal model built on top of [Mistral Large 2](/mistralai/mistral-large-2411). The model is able to understand documents, charts and natural images.\n\nThe model is available under the Mistral Research License (MRL) for research and educational use, and the Mistral Commercial License for experimentation, testing, and production for commercial purposes.\n\n",
      "context_length": 128000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000002",
        "completion": "0.000006",
        "image": "0.002888",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "ai21/jamba-1-5-large",
      "name": "AI21: Jamba 1.5 Large",
      "created": 1724371200,
      "description": "Jamba 1.5 Large is part of AI21's new family of open models, offering superior speed, efficiency, and quality.\n\nIt features a 256K effective context window, the longest among open models, enabling improved performance on tasks like document summarization and analysis.\n\nBuilt on a novel SSM-Transformer architecture, it outperforms larger models like Llama 3.1 70B on benchmarks while maintaining resource efficiency.\n\nRead their [announcement](https://www.ai21.com/blog/announcing-jamba-model-family) to learn more.",
      "context_length": 256000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Other",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000002",
        "completion": "0.000008",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 256000,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "x-ai/grok-2-1212",
      "name": "xAI: Grok 2 1212",
      "created": 1734232814,
      "description": "Grok 2 1212 introduces significant enhancements to accuracy, instruction adherence, and multilingual support, making it a powerful and flexible choice for developers seeking a highly steerable, intelligent model.",
      "context_length": 131072,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Grok",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000002",
        "completion": "0.00001",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 131072,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "cohere/command-r-plus-08-2024",
      "name": "Cohere: Command R+ (08-2024)",
      "created": 1724976000,
      "description": "command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint the same.\n\nRead the launch post [here](https://docs.cohere.com/changelog/command-gets-refreshed).\n\nUse of this model is subject to Cohere's [Acceptable Use Policy](https://docs.cohere.com/docs/c4ai-acceptable-use-policy).",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Cohere",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000002375",
        "completion": "0.0000095",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 4000,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4o",
      "name": "OpenAI: GPT-4o",
      "created": 1715558400,
      "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.\n\nFor benchmarking against other models, it was briefly called [\"im-also-a-good-gpt2-chatbot\"](https://twitter.com/LiamFedus/status/1790064963966370209)\n\n#multimodal",
      "context_length": 128000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000025",
        "completion": "0.00001",
        "image": "0.003613",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 16384,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4o-2024-08-06",
      "name": "OpenAI: GPT-4o (2024-08-06)",
      "created": 1722902400,
      "description": "The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/).\n\nGPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.\n\nFor benchmarking against other models, it was briefly called [\"im-also-a-good-gpt2-chatbot\"](https://twitter.com/LiamFedus/status/1790064963966370209)",
      "context_length": 128000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000025",
        "completion": "0.00001",
        "image": "0.003613",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 16384,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4o-2024-11-20",
      "name": "OpenAI: GPT-4o (2024-11-20)",
      "created": 1732127594,
      "description": "The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded files, providing deeper insights & more thorough responses.\n\nGPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.",
      "context_length": 128000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.0000025",
        "completion": "0.00001",
        "image": "0.003613",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 16384,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "mistralai/mistral-medium",
      "name": "Mistral Medium",
      "created": 1704844800,
      "description": "This is Mistral AI's closed-source, medium-sided model. It's powered by a closed-source prototype and excels at reasoning, code, JSON, chat, and more. In benchmarks, it compares with many of the flagship models of other companies.",
      "context_length": 32000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Mistral",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000275",
        "completion": "0.0000081",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32000,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "cohere/command-r-plus",
      "name": "Cohere: Command R+",
      "created": 1712188800,
      "description": "Command R+ is a new, 104B-parameter LLM from Cohere. It's useful for roleplay, general consumer usecases, and Retrieval Augmented Generation (RAG).\n\nIt offers multilingual support for ten key languages to facilitate global business operations. See benchmarks and the launch post [here](https://txt.cohere.com/command-r-plus-microsoft-azure/).\n\nUse of this model is subject to Cohere's [Acceptable Use Policy](https://docs.cohere.com/docs/c4ai-acceptable-use-policy).",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Cohere",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000285",
        "completion": "0.00001425",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 4000,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "cohere/command-r-plus-04-2024",
      "name": "Cohere: Command R+ (04-2024)",
      "created": 1712016000,
      "description": "Command R+ is a new, 104B-parameter LLM from Cohere. It's useful for roleplay, general consumer usecases, and Retrieval Augmented Generation (RAG).\n\nIt offers multilingual support for ten key languages to facilitate global business operations. See benchmarks and the launch post [here](https://txt.cohere.com/command-r-plus-microsoft-azure/).\n\nUse of this model is subject to Cohere's [Acceptable Use Policy](https://docs.cohere.com/docs/c4ai-acceptable-use-policy).",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Cohere",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00000285",
        "completion": "0.00001425",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 4000,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-3.5-turbo-16k",
      "name": "OpenAI: GPT-3.5 Turbo 16k",
      "created": 1693180800,
      "description": "This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up to Sep 2021.",
      "context_length": 16385,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000003",
        "completion": "0.000004",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 16385,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3.5-sonnet:beta",
      "name": "Anthropic: Claude 3.5 Sonnet (self-moderated)",
      "created": 1729555200,
      "description": "New Claude 3.5 Sonnet delivers better-than-Opus capabilities, faster-than-Sonnet speeds, at the same Sonnet prices. Sonnet is particularly good at:\n\n- Coding: Scores ~49% on SWE-Bench Verified, higher than the last best score, and without any fancy prompt scaffolding\n- Data science: Augments human data science expertise; navigates unstructured data while using multiple tools for insights\n- Visual processing: excelling at interpreting charts, graphs, and images, accurately transcribing text to derive insights beyond just the text alone\n- Agentic tasks: exceptional tool use, making it great at agentic tasks (i.e. complex, multi-step problem solving tasks that require engaging with other systems)\n\n#multimodal",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000003",
        "completion": "0.000015",
        "image": "0.0048",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3.5-sonnet",
      "name": "Anthropic: Claude 3.5 Sonnet",
      "created": 1729555200,
      "description": "New Claude 3.5 Sonnet delivers better-than-Opus capabilities, faster-than-Sonnet speeds, at the same Sonnet prices. Sonnet is particularly good at:\n\n- Coding: Scores ~49% on SWE-Bench Verified, higher than the last best score, and without any fancy prompt scaffolding\n- Data science: Augments human data science expertise; navigates unstructured data while using multiple tools for insights\n- Visual processing: excelling at interpreting charts, graphs, and images, accurately transcribing text to derive insights beyond just the text alone\n- Agentic tasks: exceptional tool use, making it great at agentic tasks (i.e. complex, multi-step problem solving tasks that require engaging with other systems)\n\n#multimodal",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000003",
        "completion": "0.000015",
        "image": "0.0048",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 8192,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3-sonnet:beta",
      "name": "Anthropic: Claude 3 Sonnet (self-moderated)",
      "created": 1709596800,
      "description": "Claude 3 Sonnet is an ideal balance of intelligence and speed for enterprise workloads. Maximum utility at a lower price, dependable, balanced for scaled deployments.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-family)\n\n#multimodal",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000003",
        "completion": "0.000015",
        "image": "0.0048",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3-sonnet",
      "name": "Anthropic: Claude 3 Sonnet",
      "created": 1709596800,
      "description": "Claude 3 Sonnet is an ideal balance of intelligence and speed for enterprise workloads. Maximum utility at a lower price, dependable, balanced for scaled deployments.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-family)\n\n#multimodal",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000003",
        "completion": "0.000015",
        "image": "0.0048",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3.5-sonnet-20240620:beta",
      "name": "Anthropic: Claude 3.5 Sonnet (2024-06-20) (self-moderated)",
      "created": 1718841600,
      "description": "Claude 3.5 Sonnet delivers better-than-Opus capabilities, faster-than-Sonnet speeds, at the same Sonnet prices. Sonnet is particularly good at:\n\n- Coding: Autonomously writes, edits, and runs code with reasoning and troubleshooting\n- Data science: Augments human data science expertise; navigates unstructured data while using multiple tools for insights\n- Visual processing: excelling at interpreting charts, graphs, and images, accurately transcribing text to derive insights beyond just the text alone\n- Agentic tasks: exceptional tool use, making it great at agentic tasks (i.e. complex, multi-step problem solving tasks that require engaging with other systems)\n\nFor the latest version (2024-10-23), check out [Claude 3.5 Sonnet](/anthropic/claude-3.5-sonnet).\n\n#multimodal",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000003",
        "completion": "0.000015",
        "image": "0.0048",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 8192,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3.5-sonnet-20240620",
      "name": "Anthropic: Claude 3.5 Sonnet (2024-06-20)",
      "created": 1718841600,
      "description": "Claude 3.5 Sonnet delivers better-than-Opus capabilities, faster-than-Sonnet speeds, at the same Sonnet prices. Sonnet is particularly good at:\n\n- Coding: Autonomously writes, edits, and runs code with reasoning and troubleshooting\n- Data science: Augments human data science expertise; navigates unstructured data while using multiple tools for insights\n- Visual processing: excelling at interpreting charts, graphs, and images, accurately transcribing text to derive insights beyond just the text alone\n- Agentic tasks: exceptional tool use, making it great at agentic tasks (i.e. complex, multi-step problem solving tasks that require engaging with other systems)\n\nFor the latest version (2024-10-23), check out [Claude 3.5 Sonnet](/anthropic/claude-3.5-sonnet).\n\n#multimodal",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000003",
        "completion": "0.000015",
        "image": "0.0048",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 8192,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4o-2024-05-13",
      "name": "OpenAI: GPT-4o (2024-05-13)",
      "created": 1715558400,
      "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.\n\nFor benchmarking against other models, it was briefly called [\"im-also-a-good-gpt2-chatbot\"](https://twitter.com/LiamFedus/status/1790064963966370209)\n\n#multimodal",
      "context_length": 128000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000005",
        "completion": "0.000015",
        "image": "0.007225",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "x-ai/grok-beta",
      "name": "xAI: Grok Beta",
      "created": 1729382400,
      "description": "Grok Beta is xAI's experimental language model with state-of-the-art reasoning capabilities, best for complex and multi-step use cases.\n\nIt is the successor of [Grok 2](https://x.ai/blog/grok-2) with enhanced context length.",
      "context_length": 131072,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "Grok",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000005",
        "completion": "0.000015",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 131072,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "x-ai/grok-vision-beta",
      "name": "xAI: Grok Vision Beta",
      "created": 1731976624,
      "description": "Grok Vision Beta is xAI's experimental language model with vision capability.\n\n",
      "context_length": 8192,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Grok",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000005",
        "completion": "0.000015",
        "image": "0.009",
        "request": "0"
      },
      "top_provider": {
        "context_length": 8192,
        "max_completion_tokens": null,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4o:extended",
      "name": "OpenAI: GPT-4o (extended)",
      "created": 1715558400,
      "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.\n\nFor benchmarking against other models, it was briefly called [\"im-also-a-good-gpt2-chatbot\"](https://twitter.com/LiamFedus/status/1790064963966370209)\n\n#multimodal",
      "context_length": 128000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000006",
        "completion": "0.000018",
        "image": "0.007225",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 64000,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4-turbo",
      "name": "OpenAI: GPT-4 Turbo",
      "created": 1712620800,
      "description": "The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.",
      "context_length": 128000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00001",
        "completion": "0.00003",
        "image": "0.01445",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4-1106-preview",
      "name": "OpenAI: GPT-4 Turbo (older v1106)",
      "created": 1699228800,
      "description": "The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to April 2023.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00001",
        "completion": "0.00003",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4-turbo-preview",
      "name": "OpenAI: GPT-4 Turbo Preview",
      "created": 1706140800,
      "description": "The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023.\n\n**Note:** heavily rate limited by OpenAI while in preview.",
      "context_length": 128000,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00001",
        "completion": "0.00003",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 128000,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/o1",
      "name": "OpenAI: o1",
      "created": 1734459999,
      "description": "The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason using chain of thought. \n\nThe o1 models are optimized for math, science, programming, and other STEM-related tasks. They consistently exhibit PhD-level accuracy on benchmarks in physics, chemistry, and biology. Learn more in the [launch announcement](https://openai.com/o1).\n",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000015",
        "completion": "0.00006",
        "image": "0.021675",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 100000,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3-opus:beta",
      "name": "Anthropic: Claude 3 Opus (self-moderated)",
      "created": 1709596800,
      "description": "Claude 3 Opus is Anthropic's most powerful model for highly complex tasks. It boasts top-level performance, intelligence, fluency, and understanding.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-family)\n\n#multimodal",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000015",
        "completion": "0.000075",
        "image": "0.024",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 4096,
        "is_moderated": false
      },
      "per_request_limits": null
    },
    {
      "id": "anthropic/claude-3-opus",
      "name": "Anthropic: Claude 3 Opus",
      "created": 1709596800,
      "description": "Claude 3 Opus is Anthropic's most powerful model for highly complex tasks. It boasts top-level performance, intelligence, fluency, and understanding.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-family)\n\n#multimodal",
      "context_length": 200000,
      "architecture": {
        "modality": "text+image->text",
        "tokenizer": "Claude",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.000015",
        "completion": "0.000075",
        "image": "0.024",
        "request": "0"
      },
      "top_provider": {
        "context_length": 200000,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4",
      "name": "OpenAI: GPT-4",
      "created": 1685232000,
      "description": "OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning capabilities. Training data: up to Sep 2021.",
      "context_length": 8191,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00003",
        "completion": "0.00006",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 8191,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4-0314",
      "name": "OpenAI: GPT-4 (older v0314)",
      "created": 1685232000,
      "description": "GPT-4-0314 is the first version of GPT-4 released, with a context length of 8,192 tokens, and was supported until June 14. Training data: up to Sep 2021.",
      "context_length": 8191,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00003",
        "completion": "0.00006",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 8191,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4-32k-0314",
      "name": "OpenAI: GPT-4 32k (older v0314)",
      "created": 1693180800,
      "description": "GPT-4-32k is an extended version of GPT-4, with the same capabilities but quadrupled context length, allowing for processing up to 40 pages of text in a single pass. This is particularly beneficial for handling longer content like interacting with PDFs without an external vector database. Training data: up to Sep 2021.",
      "context_length": 32767,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00006",
        "completion": "0.00012",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32767,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    },
    {
      "id": "openai/gpt-4-32k",
      "name": "OpenAI: GPT-4 32k",
      "created": 1693180800,
      "description": "GPT-4-32k is an extended version of GPT-4, with the same capabilities but quadrupled context length, allowing for processing up to 40 pages of text in a single pass. This is particularly beneficial for handling longer content like interacting with PDFs without an external vector database. Training data: up to Sep 2021.",
      "context_length": 32767,
      "architecture": {
        "modality": "text->text",
        "tokenizer": "GPT",
        "instruct_type": null
      },
      "pricing": {
        "prompt": "0.00006",
        "completion": "0.00012",
        "image": "0",
        "request": "0"
      },
      "top_provider": {
        "context_length": 32767,
        "max_completion_tokens": 4096,
        "is_moderated": true
      },
      "per_request_limits": null
    }
  ]
}