{"object":"list","data":[{"api":"chat","id":"xai/grok-3-mini:low","object":"model","created":1739923200,"updated":1788861323,"owned_by":"system","input_price":3e-7,"cached_price":3e-7,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":3e-7,"output_price":5e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A lightweight model that thinks before responding. Fast, smart, and great for logic-based tasks that do not require deep domain knowledge. The raw thinking traces are accessible.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-3-mini"},{"api":"chat","id":"xai/grok-4-1-fast-reasoning","object":"model","created":1763510400,"updated":1787843415,"owned_by":"system","input_price":2e-7,"caching_price":2e-7,"cached_price":5e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"caching_price":2e-7,"cached_price":5e-8,"output_price":5e-7},{"prompt_tokens_threshold":128000,"input_price":4e-7,"caching_price":4e-7,"cached_price":5e-8,"output_price":0.000001}],"max_output_tokens":0,"context_window":2000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A frontier multimodal model optimized specifically for high-performance agentic tool calling.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-4-1-fast-reasoning"},{"api":"chat","id":"xai/grok-4.2-beta","object":"model","created":1773944308,"updated":1788220720,"owned_by":"system","input_price":0.000002,"caching_price":0.000002,"cached_price":2e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.000002,"cached_price":2e-7,"output_price":0.000006},{"prompt_tokens_threshold":200000,"input_price":0.000004,"caching_price":0.000004,"cached_price":4e-7,"output_price":0.000012}],"max_output_tokens":0,"context_window":2000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Grok 4.20 Beta is xAI's newest flagship model with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering consistently precise and truthful responses.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-4.2-beta"},{"api":"chat","id":"xai/grok-4-fast","object":"model","created":1758240000,"updated":1787843415,"owned_by":"system","input_price":2e-7,"caching_price":2e-7,"cached_price":5e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"caching_price":2e-7,"cached_price":5e-8,"output_price":5e-7},{"prompt_tokens_threshold":128000,"input_price":4e-7,"caching_price":4e-7,"cached_price":5e-8,"output_price":0.000001}],"max_output_tokens":0,"context_window":2000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"xAI's latest advancement in cost-efficient reasoning models","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-4-fast"},{"api":"chat","id":"xai/grok-4.3","object":"model","created":1777591821,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.00000125,"cached_price":2e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.00000125,"cached_price":2e-7,"output_price":0.0000025},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.0000025,"cached_price":4e-7,"output_price":0.000005}],"max_output_tokens":0,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Grok 4.3 is a reasoning model from xAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual accuracy. Reasoning is always active and cannot be disabled or configured by effort level. It supports a 1 million token context window with no output token limit, making it well-suited for long-document analysis, deep research, and multi-step agentic tasks. Pricing is tiered: requests exceeding 200k total tokens are billed at a higher rate.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-4.3"},{"api":"chat","id":"xai/grok-3-mini","object":"model","created":1739923200,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":3e-7,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":3e-7,"output_price":5e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A lightweight model that thinks before responding. Fast, smart, and great for logic-based tasks that do not require deep domain knowledge. The raw thinking traces are accessible.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-3-mini"},{"api":"chat","id":"xai/grok-4-fast-non-reasoning","object":"model","created":1758240000,"updated":1787843415,"owned_by":"system","input_price":2e-7,"caching_price":2e-7,"cached_price":5e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"caching_price":2e-7,"cached_price":5e-8,"output_price":5e-7},{"prompt_tokens_threshold":128000,"input_price":4e-7,"caching_price":4e-7,"cached_price":5e-8,"output_price":0.000001}],"max_output_tokens":0,"context_window":2000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":true,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"xAI's latest advancement in cost-efficient reasoning models","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-4-fast-non-reasoning"},{"api":"chat","id":"xai/grok-4.5","object":"model","created":1783523154,"updated":1788220514,"owned_by":"system","input_price":0.000002,"caching_price":0.000002,"cached_price":5e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.000002,"cached_price":5e-7,"output_price":0.000006},{"prompt_tokens_threshold":200000,"input_price":0.000004,"caching_price":0.000004,"cached_price":0.000001,"output_price":0.000012}],"max_output_tokens":0,"context_window":500000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Grok 4.5 is xAI's frontier multimodal model for advanced coding, knowledge work, STEM reasoning, and agentic workflows. It supports text and image inputs with text output, tool calling, structured outputs, web search, and extended reasoning. Pricing is tiered: requests exceeding 200k total tokens are billed at a higher rate.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-4.5"},{"api":"chat","id":"xai/grok-code-fast-1","object":"model","created":1756339200,"updated":1787843415,"owned_by":"system","input_price":2e-7,"cached_price":2e-8,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":2e-8,"output_price":0.0000015}],"max_output_tokens":0,"context_window":256000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A speedy and economical reasoning model that excels at agentic coding","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-code-fast-1"},{"api":"chat","id":"xai/grok-3","object":"model","created":1739923200,"updated":1787843415,"owned_by":"system","input_price":0.000005,"cached_price":0.000005,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"cached_price":0.000005,"output_price":0.000025}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Excels at enterprise use cases like data extraction, coding, and text summarization. Possesses deep domain knowledge in finance, healthcare, law, and science.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-3"},{"api":"chat","id":"xai/grok-4.6","object":"model","created":1786492800,"updated":1788220514,"owned_by":"system","input_price":0.000002,"caching_price":0.000002,"cached_price":5e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.000002,"cached_price":5e-7,"output_price":0.000006},{"prompt_tokens_threshold":200000,"input_price":0.000004,"caching_price":0.000004,"cached_price":0.000001,"output_price":0.000012}],"max_output_tokens":0,"context_window":500000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Grok 4.6 is xAI's smartest frontier model, with leading performance on coding, knowledge work, and STEM. It accepts text and image inputs with text output, and supports tool calling, structured outputs, web search, prompt caching, and extended reasoning. 500k token context window. Pricing is tiered: requests exceeding 200k total tokens are billed at a higher rate.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-4.6"},{"api":"chat","id":"xai/grok-build-0.1","object":"model","created":1779298123,"updated":1787843415,"owned_by":"system","input_price":0.000001,"cached_price":1e-7,"output_price":0.000002,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"cached_price":1e-7,"output_price":0.000002}],"max_output_tokens":0,"context_window":256000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Grok Build 0.1 is xAI's dedicated coding model optimized for software engineering tasks including code generation, review, debugging, and technical problem solving.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-build-0.1"},{"api":"chat","id":"xai/grok-4.7","object":"model","created":1790008976,"updated":1790008976,"owned_by":"system","input_price":0.000002,"caching_price":0.000002,"cached_price":5e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.000002,"cached_price":5e-7,"output_price":0.000006}],"max_output_tokens":0,"context_window":500000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and managing long context, and it improves on its predecessor at professional knowledge work such as drafting documents and presentations.\n\nThe model was trained with a longer reinforcement learning run weighted toward problems that take many hours to complete, and natively understands the Grok Bot harness for conversational tasks. It ships with a new safeguard stack that pairs strong jailbreak resistance with low refusal rates for legitimate cybersecurity and biology work. SpaceXAI's reported benchmark results use the xhigh reasoning effort.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-4.7"},{"api":"chat","id":"xai/grok-4-1-fast-non-reasoning","object":"model","created":1763510400,"updated":1787843415,"owned_by":"system","input_price":2e-7,"caching_price":2e-7,"cached_price":5e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"caching_price":2e-7,"cached_price":5e-8,"output_price":5e-7},{"prompt_tokens_threshold":128000,"input_price":4e-7,"caching_price":4e-7,"cached_price":5e-8,"output_price":0.000001}],"max_output_tokens":0,"context_window":2000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A frontier multimodal model optimized specifically for high-performance agentic tool calling.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-4-1-fast-non-reasoning"},{"api":"chat","id":"xai/grok-3-mini:high","object":"model","created":1739923200,"updated":1788861323,"owned_by":"system","input_price":3e-7,"cached_price":3e-7,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":3e-7,"output_price":5e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A lightweight model that thinks before responding. Fast, smart, and great for logic-based tasks that do not require deep domain knowledge. The raw thinking traces are accessible.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-3-mini"},{"api":"chat","id":"xai/grok-4","object":"model","created":1752019200,"updated":1787843415,"owned_by":"system","input_price":0.000003,"caching_price":0.000003,"cached_price":7.5e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.000003,"cached_price":7.5e-7,"output_price":0.000015},{"prompt_tokens_threshold":128000,"input_price":0.000006,"caching_price":0.000006,"cached_price":7.5e-7,"output_price":0.00003}],"max_output_tokens":0,"context_window":256000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"xAI's latest and greatest flagship model, offering unparalleled performance in natural language, math and reasoning - the perfect jack of all trades.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"xai","model_canonical_name":"grok-4"},{"api":"chat","id":"mistral/mistral-medium-latest","object":"model","created":1777570439,"updated":1787843415,"owned_by":"system","input_price":4.4e-7,"cached_price":4.4e-7,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.4e-7,"cached_price":4.4e-7,"output_price":0.0000022}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"mistral-large-latest is the routing tag used by Mistral AI to point to their most advanced flagship medium language model","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"mistral","model_canonical_name":"mistral-medium-latest"},{"api":"chat","id":"mistral/glm-5-2","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":0.00000154,"cached_price":1.54e-7,"output_price":0.00000484,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000154,"cached_price":1.54e-7,"output_price":0.00000484}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM 5.2 by Z.ai is an open weights MoE model (744B total / 40B active parameters) served by Mistral on EU infrastructure with no Mistral modifications. 1M token context window, 128K max output. Text only. Supports reasoning, function calling, structured outputs, predicted outputs and prefix. Built for long context coding and agentic workflows. Public preview since 6 August 2026.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"mistral/mistral-large-latest","object":"model","created":1764624472,"updated":1787843415,"owned_by":"system","input_price":5.5e-7,"cached_price":5.5e-7,"output_price":0.00000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":5.5e-7,"cached_price":5.5e-7,"output_price":0.00000165}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"mistral-large-latest is the routing tag used by Mistral AI to point to their most advanced flagship large language model","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"mistral","model_canonical_name":"mistral-large-latest"},{"api":"chat","id":"mistral/devstral-small-latest","object":"model","created":1752105600,"updated":1787843415,"owned_by":"system","input_price":1.1e-7,"cached_price":1.1e-7,"output_price":3.3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.1e-7,"cached_price":1.1e-7,"output_price":3.3e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"mistral-large-latest is the routing tag used by Mistral AI to point to their most advanced flagship small language model","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"mistral","model_canonical_name":"devstral-small-latest"},{"api":"chat","id":"mistral/mistral-small-2503","object":"model","created":1742238937,"updated":1787843415,"owned_by":"system","input_price":1.1e-7,"cached_price":1.1e-7,"output_price":3.3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.1e-7,"cached_price":1.1e-7,"output_price":3.3e-7}],"max_output_tokens":0,"context_window":32768,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Mistral Small 3.1 (2503) is a powerful 24-billion parameter multimodal AI model developed by Mistral AI that processes both text and image inputs. Designed for enterprise efficiency and fast-response agents, it is highly sought after for its exceptional long-context processing, function calling, and local deployment capabilities. ","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"mistral","model_canonical_name":"mistral-small-2503","retires":1785456000},{"api":"chat","id":"mistral/mistral-medium-3-5","object":"model","created":1777570439,"updated":1787843415,"owned_by":"system","input_price":0.00000165,"cached_price":0.00000165,"output_price":0.00000825,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000165,"cached_price":0.00000165,"output_price":0.00000825}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Mistral Medium 3.5 is a dense 128B instruction following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex multi step reasoning. It is particularly strong at reliable multi tool calling and long horizon tasks, with a 256K context window, configurable reasoning effort per request, and a custom vision encoder that handles variable image sizes and aspect ratios. Self hostable on as few as four GPUs and available under open weights.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"mistral","model_canonical_name":"mistral-medium-3-5"},{"api":"chat","id":"mistral/mistral-small-2603","object":"model","created":1773695685,"updated":1787843415,"owned_by":"system","input_price":1.65e-7,"cached_price":1.65e-7,"output_price":6.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.65e-7,"cached_price":1.65e-7,"output_price":6.6e-7}],"max_output_tokens":0,"context_window":256000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Mistral's powerful hybrid model unifying instruct, reasoning, and coding capabilities in a single model. 119B parameters with 6.5B active.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"mistral","model_canonical_name":"mistral-small-2603"},{"api":"chat","id":"mistral/leanstral-1-5","object":"model","created":1779840000,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":32768,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Leanstral 1.5 is an updated Lean 4 formal proof engineering model from Mistral AI, optimized for automated theorem proving and autoformalization. It has 119B total parameters with 6.5B active and supports a 256K token context window. It supports native function calling and structured output.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"mistral","model_canonical_name":"leanstral-1-5"},{"api":"chat","id":"mistral/codestral-latest","object":"model","created":1754079630,"updated":1787843415,"owned_by":"system","input_price":3.3e-7,"cached_price":3.3e-7,"output_price":9.9e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.3e-7,"cached_price":3.3e-7,"output_price":9.9e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Codestral-latest is Mistral AI’s state-of-the-art open-weight language model built specifically for software development and code generation. It is fine-tuned to assist developers with code generation, completion, refactoring, and debugging across more than 80 programming languages.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"mistral","model_canonical_name":"codestral-latest"},{"api":"chat","id":"mistral/devstral-latest","object":"model","created":1779840000,"updated":1787843415,"owned_by":"system","input_price":4.4e-7,"cached_price":4.4e-7,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.4e-7,"cached_price":4.4e-7,"output_price":0.0000022}],"max_output_tokens":0,"context_window":256000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"An enterprise grade text model, that excels at using tools to explore codebases, editing multiple files and power software engineering agents.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"mistral","model_canonical_name":"devstral-latest"},{"api":"chat","id":"mistral/mistral-small-latest","object":"model","created":1773695685,"updated":1787843415,"owned_by":"system","input_price":1.65e-7,"cached_price":1.65e-7,"output_price":6.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.65e-7,"cached_price":1.65e-7,"output_price":6.6e-7}],"max_output_tokens":0,"context_window":256000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Mistral's powerful hybrid model unifying instruct, reasoning, and coding capabilities in a single model. 119B parameters with 6.5B active.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"mistral","model_canonical_name":"mistral-small-2603"},{"api":"chat","id":"mistral/open-mistral-7b","object":"model","created":1695772800,"updated":1787843415,"owned_by":"system","input_price":2.75e-7,"cached_price":2.75e-7,"output_price":2.75e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.75e-7,"cached_price":2.75e-7,"output_price":2.75e-7}],"max_output_tokens":0,"context_window":32768,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"OpenMistral-7B (officially Mistral 7B) is a highly efficient, 7.3-billion parameter open-weight language model released by Mistral AI. Praised for its excellent performance-to-size ratio, it frequently outperforms much larger proprietary and open-source models while being light enough to run on standard consumer hardware. ","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"mistral","model_canonical_name":"open-mistral-7b","retires":1788134400},{"api":"chat","id":"coding/gemini-2.5-pro@us-east5","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-pro@europe-north1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-pro@us-east1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash@us-south1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash@us-central1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash@europe-north1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash@europe-west1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash@europe-west4","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-pro@us-central1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-pro@europe-west4","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash@europe-west8","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash@us-east1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-pro@us-south1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash@us-west1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash@europe-central2","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-pro@us-west1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-pro","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-pro@europe-west1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-flash@us-east5","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-pro@europe-central2","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"coding/gemini-2.5-pro@europe-west8","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"doubleword/deepseek-v4-pro-0424:flex","object":"model","created":1777000679,"updated":1787843415,"owned_by":"system","input_price":0.00000131,"output_price":0.00000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000131,"output_price":0.00000275}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek V4-Pro is DeepSeek's flagship open-weights MoE model, with performance in the same class as Claude Sonnet 4.6 and stronger overall results than GLM-5.1.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro-0424"},{"api":"chat","id":"doubleword/deepseek-v4-flash-0424:flex","object":"model","created":1777000666,"updated":1787843415,"owned_by":"system","input_price":1e-7,"output_price":2e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"output_price":2e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek V4-Flash is the smaller model in the V4 family, ahead of models like Gemini 3 Flash and Qwen3.5-397B-A17B.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0424"},{"api":"chat","id":"doubleword/glm-5.2:flex","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":0.00000105,"output_price":0.0000033,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000105,"output_price":0.0000033}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Z.ai's next-generation flagship model for agentic software engineering, with significantly stronger coding capabilities than GLM-5.1.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"doubleword/glm-5.2","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":0.0000014,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"output_price":0.0000044}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Z.ai's next-generation flagship model for agentic software engineering, with significantly stronger coding capabilities than GLM-5.1.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"doubleword/deepseek-v4-flash-0424","object":"model","created":1777000666,"updated":1787843415,"owned_by":"system","input_price":1.4e-7,"output_price":2.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"output_price":2.8e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek V4-Flash is the smaller model in the V4 family, ahead of models like Gemini 3 Flash and Qwen3.5-397B-A17B.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0424"},{"api":"chat","id":"doubleword/nvidia-nemotron-3-ultra:flex","object":"model","created":1782230296,"updated":1787843415,"owned_by":"system","input_price":3.7e-7,"output_price":0.00000187,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.7e-7,"output_price":0.00000187}],"max_output_tokens":131072,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3 Ultra is NVIDIA's strongest open-weights reasoning model, positioned near GPT-5.4 Mini (xhigh) and ahead of DeepSeek V4-Flash and Qwen3.5-397B-A17B.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nvidia-nemotron-3-ultra"},{"api":"chat","id":"doubleword/deepseek-v4-pro-0424","object":"model","created":1777000679,"updated":1787843415,"owned_by":"system","input_price":0.00000174,"output_price":0.00000348,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000174,"output_price":0.00000348}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek V4-Pro is DeepSeek's flagship open-weights MoE model, with performance in the same class as Claude Sonnet 4.6 and stronger overall results than GLM-5.1.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro-0424"},{"api":"chat","id":"doubleword/nvidia-nemotron-3-ultra","object":"model","created":1782230290,"updated":1787843415,"owned_by":"system","input_price":5e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"output_price":0.0000025}],"max_output_tokens":131072,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3 Ultra is NVIDIA's strongest open-weights reasoning model, positioned near GPT-5.4 Mini (xhigh) and ahead of DeepSeek V4-Flash and Qwen3.5-397B-A17B.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nvidia-nemotron-3-ultra"},{"api":"chat","id":"sail/gemma-4-31b-it-nvfp4","object":"model","created":1775088000,"updated":1788116243,"owned_by":"system","input_price":1.4e-7,"cached_price":7e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":7e-8,"output_price":4e-7}],"max_output_tokens":65536,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVFP4-quantized Gemma 4 31B IT, NVIDIA's efficient quantization of Google's open multimodal model, at lower cost than the BF16 checkpoint. Served via Sail Research.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"nvfp4","open_weights":true,"model_lab":"google","model_canonical_name":"gemma-4-31b"},{"api":"chat","id":"sail/glm-5.2","object":"model","created":1781568000,"updated":1788116259,"owned_by":"system","input_price":8e-7,"cached_price":1.6e-7,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":8e-7,"cached_price":1.6e-7,"output_price":0.000003}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.2 is Z.ai's flagship model for long-horizon coding, agentic engineering, and sustained multi-step execution, with a 1M token context window. Served via Sail Research.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"sail/gpt-oss-120b","object":"model","created":1754352000,"updated":1788116250,"owned_by":"system","input_price":6e-8,"cached_price":3e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-8,"cached_price":3e-8,"output_price":4e-7}],"max_output_tokens":131072,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"gpt-oss-120b is OpenAI's open-weight flagship reasoning MoE model with configurable reasoning effort, native tool use, and structured output. Served via Sail Research.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"mxfp4","open_weights":true,"model_lab":"openai","model_canonical_name":"gpt-oss-120b"},{"api":"chat","id":"sail/gemma-4-31b-it","object":"model","created":1775088000,"updated":1788116247,"owned_by":"system","input_price":4e-7,"cached_price":2e-7,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"cached_price":2e-7,"output_price":6e-7}],"max_output_tokens":65536,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemma 4 31B IT is Google's open multimodal model with strong multilingual, vision, and general chat performance. Served via Sail Research.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"google","model_canonical_name":"gemma-4-31b"},{"api":"chat","id":"sail/kimi-k2.6","object":"model","created":1776729600,"updated":1788116253,"owned_by":"system","input_price":0.000001,"cached_price":2e-7,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"cached_price":2e-7,"output_price":0.000004}],"max_output_tokens":131072,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.6 is Moonshot AI's 1T-parameter MoE model with strong coding, agentic, vision, and long-context performance. Served via Sail Research.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"moonshotai","model_canonical_name":"kimi-k2.6"},{"api":"chat","id":"sail/deepseek-v4-flash-0731","object":"model","created":1785456000,"updated":1788116256,"owned_by":"system","input_price":9e-8,"cached_price":2e-8,"output_price":1.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":9e-8,"cached_price":2e-8,"output_price":1.8e-7}],"max_output_tokens":384000,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash 0731 is DeepSeek's fast, cost-efficient frontier MoE model for coding, reasoning, and agent workflows, with a 1M token context window. Served via Sail Research.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0731"},{"api":"chat","id":"openai/o4-mini:low","object":"model","created":1744824542,"updated":1788861323,"owned_by":"system","input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"openai/gpt-5-mini:flex","object":"model","created":1754587407,"updated":1787843415,"owned_by":"system","input_price":1.25e-7,"cached_price":1.25e-8,"output_price":0.000001,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.25e-7,"cached_price":1.25e-8,"output_price":0.000001}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-mini","retires":1796860800},{"api":"chat","id":"openai/gpt-4o-2024-08-06","object":"model","created":1722902400,"updated":1787843415,"owned_by":"system","input_price":0.0000025,"cached_price":0.00000125,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":0.00000125,"output_price":0.00001}],"max_output_tokens":16384,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance \u0026 readability. It’s also better at working with uploaded files, providing deeper insights \u0026 more thorough responses.\n\nGPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4o-2024-08-06"},{"api":"chat","id":"openai/gpt-5.1:flex","object":"model","created":1763060305,"updated":1787843415,"owned_by":"system","input_price":6.25e-7,"cached_price":6.25e-8,"output_price":0.000005,"pricing":[{"prompt_tokens_threshold":0,"input_price":6.25e-7,"cached_price":6.25e-8,"output_price":0.000005}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.1"},{"api":"chat","id":"openai/o4-mini:high","object":"model","created":1744824542,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"2026-09-08 09:55:23.812184+00","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"openai/o3-mini:medium","object":"model","created":1738351721,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"cached_price":5.5e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"cached_price":5.5e-7,"output_price":0.0000044}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"2026-09-08 09:55:23.812184+00","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o3-mini"},{"api":"chat","id":"openai/gpt-5.6-sol","object":"model","created":1783590850,"updated":1787843415,"owned_by":"system","input_price":0.000004,"caching_price":0.000005,"cached_price":4e-7,"output_price":0.00002,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000004,"caching_price":0.000005,"cached_price":4e-7,"output_price":0.00002},{"prompt_tokens_threshold":272000,"input_price":0.000008,"caching_price":0.00001,"cached_price":8e-7,"output_price":0.00003}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks and long-horizon problem solving.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-sol"},{"api":"chat","id":"openai/gpt-4o","object":"model","created":1732127594,"updated":1787843415,"owned_by":"system","input_price":0.0000025,"cached_price":0.00000125,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":0.00000125,"output_price":0.00001}],"max_output_tokens":16384,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance \u0026 readability. It’s also better at working with uploaded files, providing deeper insights \u0026 more thorough responses.\n\nGPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4o-2024-11-20"},{"api":"chat","id":"openai/o3","object":"model","created":1744827057,"updated":1787843415,"owned_by":"system","input_price":0.000002,"cached_price":5e-7,"output_price":0.000008,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"cached_price":5e-7,"output_price":0.000008}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o3 is an advanced reasoning model that utilizes deep reinforcement learning and chain-of-thought processing to solve complex, multimodal coding, mathematics, and scientific problems with up to 100,000 output tokens.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o3","retires":1796860800},{"api":"chat","id":"openai/gpt-5.6-terra","object":"model","created":1783590857,"updated":1790847020,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.000012},{"prompt_tokens_threshold":272000,"input_price":0.000004,"caching_price":0.000005,"cached_price":4e-7,"output_price":0.000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic tasks where capability and cost need to be balanced, offering strong performance at roughly half the cost of Sol.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-terra"},{"api":"chat","id":"openai/gpt-6-sol:flex","object":"model","created":1790102579,"updated":1790847020,"owned_by":"system","input_price":0.000001,"caching_price":0.00000125,"cached_price":1e-7,"output_price":0.000005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"caching_price":0.00000125,"cached_price":1e-7,"output_price":0.000005},{"prompt_tokens_threshold":272000,"input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.0000075}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional work, agentic coding, business workflow automation, and computer use, and is particularly strong at long-horizon software engineering tasks in real codebases. It approaches Astra-level factual reliability at a much lower cost.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-sol"},{"api":"chat","id":"openai/gpt-5-mini:priority","object":"model","created":1754587407,"updated":1787843415,"owned_by":"system","input_price":4.5e-7,"cached_price":4.5e-8,"output_price":0.0000036,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.5e-7,"cached_price":4.5e-8,"output_price":0.0000036}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-mini","retires":1796860800},{"api":"chat","id":"openai/o1:medium","object":"model","created":1734459999,"updated":1788861323,"owned_by":"system","input_price":0.000015,"cached_price":0.0000075,"output_price":0.00006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000015,"cached_price":0.0000075,"output_price":0.00006}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o1"},{"api":"chat","id":"openai/o4-mini:medium","object":"model","created":1744824542,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"2026-09-08 09:55:23.812184+00","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"openai/gpt-5.4:flex","object":"model","created":1772734352,"updated":1787843415,"owned_by":"system","input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.0000075,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.0000075},{"prompt_tokens_threshold":272000,"input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.00001125}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 is OpenAI's latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling high-context reasoning, coding, and multimodal analysis within the same workflow. The model delivers improved performance in coding, document understanding, tool use, and instruction following, capable of generating production-quality code, synthesizing information across multiple sources, and executing complex multi-step workflows with fewer iterations and greater token efficiency.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"openai/gpt-4o-2024-05-13","object":"model","created":1715558400,"updated":1787843415,"owned_by":"system","input_price":0.0000025,"cached_price":0.0000025,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":0.0000025,"output_price":0.00001}],"max_output_tokens":4096,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance \u0026 readability. It’s also better at working with uploaded files, providing deeper insights \u0026 more thorough responses.\n\nGPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4o-2024-05-13","retires":1792713600},{"api":"chat","id":"openai/gpt-6-sol","object":"model","created":1790102552,"updated":1790847020,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001},{"prompt_tokens_threshold":272000,"input_price":0.000004,"caching_price":0.000005,"cached_price":4e-7,"output_price":0.000015}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional work, agentic coding, business workflow automation, and computer use, and is particularly strong at long-horizon software engineering tasks in real codebases. It approaches Astra-level factual reliability at a much lower cost.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-sol"},{"api":"chat","id":"openai/gpt-5.3-codex","object":"model","created":1771959164,"updated":1787843415,"owned_by":"system","input_price":0.00000175,"cached_price":1.75e-7,"output_price":0.000014,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":1.75e-7,"output_price":0.000014}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.3-codex"},{"api":"chat","id":"openai/gpt-5-mini","object":"model","created":1754586319,"updated":1787843415,"owned_by":"system","input_price":2.5e-7,"cached_price":2.5e-8,"output_price":0.000002,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"cached_price":2.5e-8,"output_price":0.000002}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-mini","retires":1796860800},{"api":"chat","id":"openai/gpt-5.4-nano","object":"model","created":1773748187,"updated":1787843415,"owned_by":"system","input_price":2e-7,"cached_price":2e-8,"output_price":0.00000125,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":2e-8,"output_price":0.00000125}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding, and tool use, while reducing latency and cost for large-scale deployments.\n\nThe model is designed for production environments that require a balance of capability and efficiency, making it well suited for chat applications, coding assistants, and agent workflows that operate at scale. GPT-5.4 mini delivers reliable instruction following, solid multi-step reasoning, and consistent performance across diverse tasks with improved cost efficiency.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4-nano"},{"api":"chat","id":"openai/o3-pro","object":"model","created":1749601952,"updated":1787843415,"owned_by":"system","input_price":0.00002,"cached_price":0.00002,"output_price":0.00008,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00002,"cached_price":0.00002,"output_price":0.00008}],"max_output_tokens":100000,"context_window":200000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The o3 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o3-pro","retires":1796860800},{"api":"chat","id":"openai/gpt-5.1","object":"model","created":1763060305,"updated":1787843415,"owned_by":"system","input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.1"},{"api":"chat","id":"openai/gpt-6.1-sol","object":"model","created":1790703300,"updated":1790847020,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"cached_price":1e-7,"output_price":0.00001},{"prompt_tokens_threshold":272000,"input_price":0.000004,"caching_price":0.000005,"cached_price":2e-7,"output_price":0.000015}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6.1 Sol delivers near-Astra performance at a lower cost for complex coding, computer use, and professional work. It succeeds GPT-6 Sol with the same 1M context and 128K output, cheaper cached input, and reasoning effort from low to max.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6.1-sol"},{"api":"chat","id":"openai/gpt-5.5:flex","object":"model","created":1777051893,"updated":1787843415,"owned_by":"system","input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.000015},{"prompt_tokens_threshold":272000,"input_price":0.000005,"cached_price":5e-7,"output_price":0.0000225}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.5 is OpenAI's frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large-scale reasoning, coding, and multimodal workflows within a single system.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.5"},{"api":"chat","id":"openai/gpt-5.5-pro","object":"model","created":1777051896,"updated":1787843415,"owned_by":"system","input_price":0.00003,"output_price":0.00018,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00003,"output_price":0.00018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.5 Pro is OpenAI's high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, and is designed for long-horizon problem solving, agentic coding, and precise execution across multi-step workflows.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.5-pro"},{"api":"chat","id":"openai/gpt-5.4-nano:flex","object":"model","created":1773748187,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":1e-8,"output_price":6.25e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":1e-8,"output_price":6.25e-7}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding, and tool use, while reducing latency and cost for large-scale deployments.\n\nThe model is designed for production environments that require a balance of capability and efficiency, making it well suited for chat applications, coding assistants, and agent workflows that operate at scale. GPT-5.4 mini delivers reliable instruction following, solid multi-step reasoning, and consistent performance across diverse tasks with improved cost efficiency.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4-nano"},{"api":"chat","id":"openai/gpt-5","object":"model","created":1754586506,"updated":1787843415,"owned_by":"system","input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5"},{"api":"chat","id":"openai/gpt-5.6-sol:flex","object":"model","created":1783590850,"updated":1790847020,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001},{"prompt_tokens_threshold":272000,"input_price":0.000004,"caching_price":0.000005,"cached_price":4e-7,"output_price":0.000015}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks and long-horizon problem solving.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-sol"},{"api":"chat","id":"openai/gpt-5-nano","object":"model","created":1754586455,"updated":1787843415,"owned_by":"system","input_price":5e-8,"cached_price":5e-9,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"cached_price":5e-9,"output_price":4e-7}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 nano is OpenAI's fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-nano","retires":1796860800},{"api":"chat","id":"openai/gpt-5.4-mini:flex","object":"model","created":1773748178,"updated":1787843415,"owned_by":"system","input_price":3.75e-7,"cached_price":3.75e-8,"output_price":0.00000225,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.75e-7,"cached_price":3.75e-8,"output_price":0.00000225}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding, and tool use, while reducing latency and cost for large-scale deployments.\n\nThe model is designed for production environments that require a balance of capability and efficiency, making it well suited for chat applications, coding assistants, and agent workflows that operate at scale. GPT-5.4 mini delivers reliable instruction following, solid multi-step reasoning, and consistent performance across diverse tasks with improved cost efficiency.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4-mini"},{"api":"chat","id":"openai/o3-mini:high","object":"model","created":1738351721,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"cached_price":5.5e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"cached_price":5.5e-7,"output_price":0.0000044}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"2026-09-08 09:55:23.812184+00","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o3-mini"},{"api":"chat","id":"openai/gpt-4.1-nano","object":"model","created":1744654969,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":2.5e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":2.5e-8,"output_price":4e-7}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-nano"},{"api":"chat","id":"openai/gpt-5.6-luna","object":"model","created":1783590864,"updated":1790847020,"owned_by":"system","input_price":2e-7,"caching_price":2.5e-7,"cached_price":2e-8,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"caching_price":2.5e-7,"cached_price":2e-8,"output_price":0.0000012},{"prompt_tokens_threshold":272000,"input_price":4e-7,"caching_price":5e-7,"cached_price":4e-8,"output_price":0.0000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for its price tier.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-luna"},{"api":"chat","id":"openai/gpt-6.1-sol:flex","object":"model","created":1790703300,"updated":1790847020,"owned_by":"system","input_price":0.000001,"caching_price":0.00000125,"cached_price":5e-8,"output_price":0.000005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"caching_price":0.00000125,"cached_price":5e-8,"output_price":0.000005},{"prompt_tokens_threshold":272000,"input_price":0.000002,"caching_price":0.0000025,"cached_price":1e-7,"output_price":0.0000075}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6.1 Sol delivers near-Astra performance at a lower cost for complex coding, computer use, and professional work. It succeeds GPT-6 Sol with the same 1M context and 128K output, cheaper cached input, and reasoning effort from low to max.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6.1-sol"},{"api":"chat","id":"openai/gpt-5.4-pro","object":"model","created":1772734366,"updated":1787843415,"owned_by":"system","input_price":0.00003,"cached_price":0.00003,"output_price":0.00018,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00003,"cached_price":0.00003,"output_price":0.00018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs. Optimized for step-by-step reasoning, instruction following, and accuracy, GPT-5.4 Pro excels at agentic coding, long-context workflows, and multi-step problem solving.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4-pro"},{"api":"chat","id":"openai/o1:high","object":"model","created":1734459999,"updated":1788861323,"owned_by":"system","input_price":0.000015,"cached_price":0.0000075,"output_price":0.00006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000015,"cached_price":0.0000075,"output_price":0.00006}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o1"},{"api":"chat","id":"openai/gpt-4o-mini","object":"model","created":1721264400,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":7.5e-8,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":7.5e-8,"output_price":6e-7}],"max_output_tokens":16384,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs.\n\nAs their most advanced small model, it is many multiples more affordable than other recent frontier models, and more than 60% cheaper than [GPT-3.5 Turbo](/models/openai/gpt-3.5-turbo). It maintains SOTA intelligence, while being significantly more cost-effective.\n\nGPT-4o mini achieves an 82% score on MMLU and presently ranks higher than GPT-4 on chat preferences [common leaderboards](https://arena.lmsys.org/).\n\nCheck out the [launch announcement](https://openai.com/index/gpt-4o-mini-advancing-cost-efficient-intelligence/) to learn more.\n\n#multimodal","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4o-mini"},{"api":"chat","id":"openai/gpt-5-pro","object":"model","created":1759748728,"updated":1787843415,"owned_by":"system","input_price":0.000015,"cached_price":0.000015,"output_price":0.00012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000015,"cached_price":0.000015,"output_price":0.00012}],"max_output_tokens":272000,"context_window":400000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 Pro is OpenAI’s extended-reasoning tier of GPT-5, built to push reliability on hard problems, long tool chains, and agentic workflows. It keeps GPT-5’s multimodal skills and very large context (API page lists up to 400K tokens) while allocating more compute to think longer and plan better, improving code generation, math, and complex writing beyond standard GPT-5/“Thinking.” OpenAI positions Pro as the version that “uses extended reasoning for even more comprehensive and accurate answers,” targeting high-stakes tasks and enterprise use.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-pro","retires":1796860800},{"api":"chat","id":"openai/gpt-6-astra","object":"model","created":1788555169,"updated":1790847020,"owned_by":"system","input_price":0.00001,"caching_price":0.0000125,"cached_price":0.000001,"output_price":0.00005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00001,"caching_price":0.0000125,"cached_price":0.000001,"output_price":0.00005},{"prompt_tokens_threshold":272000,"input_price":0.00002,"caching_price":0.000025,"cached_price":0.000002,"output_price":0.000075}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Astra is OpenAI's most capable model, built for the hardest end-to-end work. It is suited for complex reasoning, coding, computer use, research, and document creation, with reasoning effort levels from low up to xhigh and max.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-astra"},{"api":"chat","id":"openai/gpt-5.6-luna:flex","object":"model","created":1783590864,"updated":1790847020,"owned_by":"system","input_price":1e-7,"caching_price":1.25e-7,"cached_price":1e-8,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.25e-7,"cached_price":1e-8,"output_price":6e-7},{"prompt_tokens_threshold":272000,"input_price":2e-7,"caching_price":2.5e-7,"cached_price":2e-8,"output_price":9e-7}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for its price tier.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-luna"},{"api":"chat","id":"openai/gpt-5.4-mini","object":"model","created":1773748178,"updated":1787843415,"owned_by":"system","input_price":7.5e-7,"cached_price":7.5e-8,"output_price":0.0000045,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.5e-7,"cached_price":7.5e-8,"output_price":0.0000045}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding, and tool use, while reducing latency and cost for large-scale deployments.\n\nThe model is designed for production environments that require a balance of capability and efficiency, making it well suited for chat applications, coding assistants, and agent workflows that operate at scale. GPT-5.4 mini delivers reliable instruction following, solid multi-step reasoning, and consistent performance across diverse tasks with improved cost efficiency.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4-mini"},{"api":"chat","id":"openai/o1:low","object":"model","created":1734459999,"updated":1788861323,"owned_by":"system","input_price":0.000015,"cached_price":0.0000075,"output_price":0.00006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000015,"cached_price":0.0000075,"output_price":0.00006}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o1"},{"api":"chat","id":"openai/gpt-6-astra:flex","object":"model","created":1788556144,"updated":1790847020,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025},{"prompt_tokens_threshold":272000,"input_price":0.00001,"caching_price":0.0000125,"cached_price":0.000001,"output_price":0.0000375}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Astra is OpenAI's most capable model, built for the hardest end-to-end work. It is suited for complex reasoning, coding, computer use, research, and document creation, with reasoning effort levels from low up to xhigh and max.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-astra"},{"api":"chat","id":"openai/gpt-5.6-terra:flex","object":"model","created":1783590857,"updated":1790847020,"owned_by":"system","input_price":0.000001,"caching_price":0.00000125,"cached_price":1e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"caching_price":0.00000125,"cached_price":1e-7,"output_price":0.000006},{"prompt_tokens_threshold":272000,"input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.000009}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic tasks where capability and cost need to be balanced, offering strong performance at roughly half the cost of Sol.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-terra"},{"api":"chat","id":"openai/o4-mini:flex","object":"model","created":1744820942,"updated":1787843415,"owned_by":"system","input_price":5.5e-7,"cached_price":1.38e-7,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":5.5e-7,"cached_price":1.38e-7,"output_price":0.0000022}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"openai/o3-mini:low","object":"model","created":1738351721,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"cached_price":5.5e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"cached_price":5.5e-7,"output_price":0.0000044}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"2026-09-08 09:55:23.812184+00","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o3-mini"},{"api":"chat","id":"openai/gpt-6-luna","object":"model","created":1790102582,"updated":1790847020,"owned_by":"system","input_price":1e-7,"caching_price":1.25e-7,"cached_price":1e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.25e-7,"cached_price":1e-8,"output_price":5e-7},{"prompt_tokens_threshold":272000,"input_price":2e-7,"caching_price":2.5e-7,"cached_price":2e-8,"output_price":7.5e-7}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic tasks, and at higher reasoning effort it can take on complex software engineering and computer-use tasks that previously called for a Sol-tier model.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-luna"},{"api":"chat","id":"openai/gpt-5.4","object":"model","created":1772734352,"updated":1790847020,"owned_by":"system","input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.000015},{"prompt_tokens_threshold":272000,"input_price":0.000005,"cached_price":5e-7,"output_price":0.0000225}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 is OpenAI's latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling high-context reasoning, coding, and multimodal analysis within the same workflow. The model delivers improved performance in coding, document understanding, tool use, and instruction following, capable of generating production-quality code, synthesizing information across multiple sources, and executing complex multi-step workflows with fewer iterations and greater token efficiency.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"openai/gpt-5.5","object":"model","created":1777051893,"updated":1787843415,"owned_by":"system","input_price":0.000005,"cached_price":5e-7,"output_price":0.00003,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"cached_price":5e-7,"output_price":0.00003},{"prompt_tokens_threshold":272000,"input_price":0.00001,"cached_price":0.000001,"output_price":0.000045}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.5 is OpenAI's frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large-scale reasoning, coding, and multimodal workflows within a single system.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.5"},{"api":"chat","id":"openai/gpt-5:flex","object":"model","created":1754587413,"updated":1787843415,"owned_by":"system","input_price":6.25e-7,"cached_price":6.25e-8,"output_price":0.000005,"pricing":[{"prompt_tokens_threshold":0,"input_price":6.25e-7,"cached_price":6.25e-8,"output_price":0.000005}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5","retires":1796860800},{"api":"chat","id":"openai/o3:flex","object":"model","created":1744823457,"updated":1787843415,"owned_by":"system","input_price":0.000001,"cached_price":2.5e-7,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"cached_price":2.5e-7,"output_price":0.000004}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"O3 Flex is a cheaper version of the o3 model","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o3"},{"api":"chat","id":"openai/gpt-4o-2024-11-20","object":"model","created":1732127594,"updated":1787843415,"owned_by":"system","input_price":0.0000025,"cached_price":0.00000125,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":0.00000125,"output_price":0.00001}],"max_output_tokens":16384,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance \u0026 readability. It’s also better at working with uploaded files, providing deeper insights \u0026 more thorough responses.\n\nGPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4o-2024-11-20"},{"api":"chat","id":"openai/gpt-6-luna:flex","object":"model","created":1790102585,"updated":1790847020,"owned_by":"system","input_price":5e-8,"caching_price":6.25e-8,"cached_price":5e-9,"output_price":2.5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"caching_price":6.25e-8,"cached_price":5e-9,"output_price":2.5e-7},{"prompt_tokens_threshold":272000,"input_price":1e-7,"caching_price":1.25e-7,"cached_price":1e-8,"output_price":3.75e-7}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic tasks, and at higher reasoning effort it can take on complex software engineering and computer-use tasks that previously called for a Sol-tier model.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"No training on customer data. Data retained for 30 days.","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-luna"},{"api":"chat","id":"openai/o4-mini","object":"model","created":1744824542,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"openai/gpt-4.1-mini","object":"model","created":1744654981,"updated":1787843415,"owned_by":"system","input_price":4e-7,"cached_price":1e-7,"output_price":0.0000016,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"cached_price":1e-7,"output_price":0.0000016}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-mini"},{"api":"chat","id":"openai/gpt-5-nano:flex","object":"model","created":1754587402,"updated":1787843415,"owned_by":"system","input_price":2.5e-8,"cached_price":2.5e-9,"output_price":2e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-8,"cached_price":2.5e-9,"output_price":2e-7}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 nano is OpenAI's fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-nano","retires":1796860800},{"api":"chat","id":"openai/o3-mini","object":"model","created":1738351721,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"cached_price":5.5e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"cached_price":5.5e-7,"output_price":0.0000044}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o3-mini"},{"api":"chat","id":"openai/o1","object":"model","created":1734459999,"updated":1787843415,"owned_by":"system","input_price":0.000015,"cached_price":0.0000075,"output_price":0.00006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000015,"cached_price":0.0000075,"output_price":0.00006}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"o1"},{"api":"chat","id":"openai/gpt-5.2","object":"model","created":1765389775,"updated":1787843415,"owned_by":"system","input_price":0.00000175,"cached_price":1.75e-7,"output_price":0.000014,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":1.75e-7,"output_price":0.000014}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The best model for coding and agentic tasks across industries","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.2"},{"api":"chat","id":"openai/gpt-5:priority","object":"model","created":1754587413,"updated":1787843415,"owned_by":"system","input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.00002,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.00002}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5","retires":1796860800},{"api":"chat","id":"openai/gpt-4.1","object":"model","created":1744654985,"updated":1787843415,"owned_by":"system","input_price":0.000002,"cached_price":5e-7,"output_price":0.000008,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"cached_price":5e-7,"output_price":0.000008}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1"},{"api":"chat","id":"openai/gpt-5.2:flex","object":"model","created":1765389775,"updated":1787843415,"owned_by":"system","input_price":8.75e-7,"cached_price":8.75e-8,"output_price":0.000007,"pricing":[{"prompt_tokens_threshold":0,"input_price":8.75e-7,"cached_price":8.75e-8,"output_price":0.000007}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The best model for coding and agentic tasks across industries","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.2"},{"api":"chat","id":"alibaba/qwen3.8-flash","object":"model","created":1787702400,"updated":1787843415,"owned_by":"system","input_price":1.6e-7,"caching_price":2e-7,"cached_price":1.6e-8,"output_price":4.7e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.6e-7,"caching_price":2e-7,"cached_price":1.6e-8,"output_price":4.7e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long video analysis.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3.8-flash"},{"api":"chat","id":"alibaba/qwen3-coder-flash","object":"model","created":1758115536,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":3e-7,"cached_price":8e-8,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":3e-7,"cached_price":8e-8,"output_price":0.0000015},{"prompt_tokens_threshold":32000,"input_price":5e-7,"caching_price":5e-7,"cached_price":1.2e-7,"output_price":0.0000025}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3 Coder Flash","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3-coder-flash"},{"api":"chat","id":"alibaba/qwen-plus","object":"model","created":1738409840,"updated":1787843415,"owned_by":"system","input_price":4e-7,"caching_price":4e-7,"cached_price":4e-7,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"caching_price":4e-7,"cached_price":4e-7,"output_price":0.0000012}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen-plus"},{"api":"chat","id":"alibaba/qwen-max","object":"model","created":1737763200,"updated":1787843415,"owned_by":"system","input_price":0.0000016,"caching_price":0.0000016,"cached_price":0.0000016,"output_price":0.0000064,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000016,"caching_price":0.0000016,"cached_price":0.0000016,"output_price":0.0000064}],"max_output_tokens":0,"context_window":32768,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen-max"},{"api":"chat","id":"alibaba/qwen3-max","object":"model","created":1758662808,"updated":1787843415,"owned_by":"system","input_price":8.61e-7,"output_price":0.000003441,"pricing":[{"prompt_tokens_threshold":0,"input_price":8.61e-7,"output_price":0.000003441},{"prompt_tokens_threshold":32678,"input_price":0.000001434,"output_price":0.000005735},{"prompt_tokens_threshold":131072,"input_price":0.000002151,"output_price":0.000008602}],"max_output_tokens":65536,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"This is the best-performing model in the Qwen series. It is ideal for complex, multi-step tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3-max"},{"api":"chat","id":"alibaba/qwen3.8-max","object":"model","created":1785731612,"updated":1787843415,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2.5e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"cached_price":2.5e-7,"output_price":0.000006}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.8-Max is Alibaba's 2.4 trillion parameter MoE flagship, delivering a comprehensive leap in coding and professional work. It autonomously codes and delivers complete projects spanning 10+ days, and handles hundreds of specialized tasks across legal, financial, design, and other professional domains, producing production grade results end to end in a single conversation. Native visual understanding runs through the full cycle of planning, execution, and verification, enabling deep semantic analysis of ultra long documents and extended video content. In long horizon tasks it plans autonomously, iterates through closed feedback loops, and continuously evolves.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3.8-max"},{"api":"chat","id":"alibaba/qwen3.6-plus","object":"model","created":1775133557,"updated":1787843415,"owned_by":"system","input_price":5e-7,"caching_price":6.25e-7,"cached_price":5e-8,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"caching_price":6.25e-7,"cached_price":5e-8,"output_price":0.000003}],"max_output_tokens":65536,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.6-Plus is Alibaba's flagship with hybrid attention and sparse MoE architecture. Features 1M-token native context with always-on reasoning and near-frontier performance on coding and agentic benchmarks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3.6-plus"},{"api":"chat","id":"alibaba/qwen3-30b-a3b-instruct-2507","object":"model","created":1753806965,"updated":1787843415,"owned_by":"system","input_price":2e-7,"caching_price":8e-7,"cached_price":2e-7,"output_price":8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"caching_price":8e-7,"cached_price":2e-7,"output_price":8e-7}],"max_output_tokens":65536,"context_window":131072,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and agentic tool use. Post-trained on instruction data, it demonstrates competitive performance across reasoning (AIME, ZebraLogic), coding (MultiPL-E, LiveCodeBench), and alignment (IFEval, WritingBench) benchmarks. It outperforms its non-instruct variant on subjective and open-ended tasks while retaining strong factual and coding performance.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-30b-a3b-instruct-2507"},{"api":"chat","id":"alibaba/qwen3-coder-plus","object":"model","created":1758662707,"updated":1787843415,"owned_by":"system","input_price":0.000001,"output_price":0.000005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"output_price":0.000005},{"prompt_tokens_threshold":32678,"input_price":0.0000018,"output_price":0.000009},{"prompt_tokens_threshold":131072,"input_price":0.000003,"output_price":0.000015},{"prompt_tokens_threshold":262144,"input_price":0.000006,"output_price":0.00006}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3 Coder Plus","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3-coder-plus"},{"api":"chat","id":"alibaba/qwen-turbo","object":"model","created":1730419200,"updated":1787843415,"owned_by":"system","input_price":5e-8,"caching_price":5e-8,"cached_price":5e-8,"output_price":2e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"caching_price":5e-8,"cached_price":5e-8,"output_price":2e-7}],"max_output_tokens":0,"context_window":1000000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen-turbo"},{"api":"chat","id":"alibaba/qwen3.7-max","object":"model","created":1779376861,"updated":1787843415,"owned_by":"system","input_price":0.0000025,"caching_price":0.000003125,"cached_price":2.5e-7,"output_price":0.0000075,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"caching_price":0.000003125,"cached_price":2.5e-7,"output_price":0.0000075}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks, and long-horizon autonomous execution. The model offers notable gains in coding and agentic performance over prior Qwen generations and supports explicit prompt caching for efficient repeated context use.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3.7-max"},{"api":"chat","id":"alibaba/qwen3.7-plus","object":"model","created":1780491783,"updated":1787843415,"owned_by":"system","input_price":3.2e-7,"caching_price":4e-7,"cached_price":3.2e-8,"output_price":0.00000128,"discount_percentage":20,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.2e-7,"caching_price":4e-7,"cached_price":3.2e-8,"output_price":0.00000128}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its vision-language abilities while retaining full-stack, agent-level intelligence for coding, tool use, and productivity workflows. Its distinguishing trait is multi-modal interactive hybrid agent capability: it can perceive real-world scenes, read screens and interact with GUIs, generate code from visual references, and perform end-to-end navigation within mobile apps.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3.7-plus"},{"api":"chat","id":"xiaomi/mimo-v2.6-flash","object":"model","created":1790075515,"updated":1790075515,"owned_by":"system","input_price":1.4e-7,"cached_price":2.8e-9,"output_price":2.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":2.8e-9,"output_price":2.8e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Xiaomi MiMo V2.6 Flash. 309B total / 15B active MoE with hybrid attention, native omni-modal input (text, image, video, audio), text output. 1M token context window, 128K max output. OpenAI-compatible.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.6-flash"},{"api":"chat","id":"xiaomi/mimo-v2.6-pro","object":"model","created":1790075519,"updated":1790075519,"owned_by":"system","input_price":4.35e-7,"cached_price":3.6e-9,"output_price":8.7e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.35e-7,"cached_price":3.6e-9,"output_price":8.7e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Xiaomi MiMo V2.6 Pro. Flagship 1T+ parameter MoE, native omni-modal input (text, image, video, audio), text output. 1M token context window, 128K max output. OpenAI-compatible.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.6-pro"},{"api":"chat","id":"xiaomi/mimo-v2.5","object":"model","created":1776874269,"updated":1787843415,"owned_by":"system","input_price":1.4e-7,"cached_price":2.8e-9,"output_price":2.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":2.8e-9,"output_price":2.8e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Xiaomi MiMo v2.5. Native omni-modal model: text, image, video, and audio input, text output. 1M context window for multimodal agent scenarios. OpenAI-compatible.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.5"},{"api":"chat","id":"xiaomi/mimo-v2.6-pro-ultraspeed","object":"model","created":1790075438,"updated":1790075438,"owned_by":"system","input_price":0.00000435,"cached_price":3.6e-8,"output_price":0.0000087,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000435,"cached_price":3.6e-8,"output_price":0.0000087}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Xiaomi MiMo V2.6 Pro UltraSpeed. Same 1T+ MiMo V2.6 Pro checkpoint served at roughly 10x output speed for latency sensitive workloads. Omni-modal input (text, image, video, audio), text output. 1M token context window, 128K max output. OpenAI-compatible.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.6-pro-ultraspeed"},{"api":"chat","id":"xiaomi/mimo-v2.5-pro","object":"model","created":1776874273,"updated":1787843415,"owned_by":"system","input_price":4.35e-7,"cached_price":3.6e-9,"output_price":8.7e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.35e-7,"cached_price":3.6e-9,"output_price":8.7e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Xiaomi MiMo v2.5 Pro. 1M token context window, 128K max output. OpenAI-compatible.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.5-pro"},{"api":"chat","id":"zai/glm-5.2","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-5.2 is the latest model in Z.ai's GLM series, continuing the line's focus on coding and long-horizon agentic tasks. Like GLM-5.1, it is designed to plan, execute, and iterate autonomously on extended, engineering-grade tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"zai/GLM-5","object":"model","created":1770829182,"updated":1787843415,"owned_by":"system","input_price":0.000001,"cached_price":2e-7,"output_price":0.0000032,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"cached_price":2e-7,"output_price":0.0000032}],"max_output_tokens":128000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-5 is Zai’s new-generation flagship foundation model, designed for Agentic Engineering, capable of providing reliable productivity in complex system engineering and long-range Agent tasks. In terms of Coding and Agent capabilities, GLM-5 has achieved state-of-the-art (SOTA) performance in open source, with its usability in real programming scenarios approaching that of Claude Opus 4.5.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5"},{"api":"chat","id":"zai/glm-5.3","object":"model","created":1787077232,"updated":1787843415,"owned_by":"system","input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.3 is the latest flagship model for complex software engineering and long-horizon Agent tasks. It delivers a 50% improvement in coding experience over its predecessor, matches Mythos 5 on selected cybersecurity capabilities, and achieves a better balance between performance and token efficiency.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3"},{"api":"chat","id":"zai/GLM-4.5","object":"model","created":1753471347,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":1.1e-7,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":1.1e-7,"output_price":0.0000022}],"max_output_tokens":98304,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-4.5 and GLM-4.5-Air are Z AI's latest flagship models, purpose-built as foundational models for agent-oriented applications. Both leverage a Mixture-of-Experts (MoE) architecture. GLM-4.5 has a total parameter count of 355B with 32B active parameters per forward pass, while GLM-4.5-Air adopts a more streamlined design with 106B total parameters and 12B active parameters.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-4.5"},{"api":"chat","id":"zai/glm-5.1","object":"model","created":1775578025,"updated":1787843415,"owned_by":"system","input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044}],"max_output_tokens":128000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Compared with GLM-5, GLM-5.1 delivers significant improvements in coding, agentic tool usage, reasoning, role-play, and general chat quality. Besides, GLM-5.1 has outstanding capabilities in long-horizon agentic tasks like CUDA kernel optimization.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.1"},{"api":"chat","id":"zai/glm-5.3-flash","object":"model","created":1787761346,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while reducing compute overhead.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":false,"model_lab":"zai","model_canonical_name":"glm-5.3-flash"},{"api":"chat","id":"zai/GLM-4.7","object":"model","created":1766378014,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":1.1e-7,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":1.1e-7,"output_price":0.0000022}],"max_output_tokens":128000,"context_window":200000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-4.7 is Z AI’s latest flagship model, designed to push agentic and coding performance further. It expands the context window from 128K to 200K tokens, improves reasoning and tool-use capabilities, and delivers stronger results in coding benchmarks and real-world development workflows. GLM-4.6 demonstrates refined writing quality, more capable agent behavior, and higher token efficiency (≈15% fewer tokens vs. GLM-4.5).\n\nEvaluations show clear gains over GLM-4.5 across reasoning, agents, and coding, reaching near parity with Claude Sonnet 4 in practical tasks while outperforming other open-source baselines. GLM-4.6 is available through the Z.ai API platform, OpenRouter, coding agents (Claude Code, Roo Code, Cline, Kilo Code), and soon as downloadable weights on HuggingFace and ModelScope.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","quantization":"fp4","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-4.7"},{"api":"chat","id":"zai/GLM-4.6","object":"model","created":1759235576,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":1.1e-7,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":1.1e-7,"output_price":0.0000022}],"max_output_tokens":128000,"context_window":200000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-4.6 is Z AI’s latest flagship model, designed to push agentic and coding performance further. It expands the context window from 128K to 200K tokens, improves reasoning and tool-use capabilities, and delivers stronger results in coding benchmarks and real-world development workflows. GLM-4.6 demonstrates refined writing quality, more capable agent behavior, and higher token efficiency (≈15% fewer tokens vs. GLM-4.5).\n\nEvaluations show clear gains over GLM-4.5 across reasoning, agents, and coding, reaching near parity with Claude Sonnet 4 in practical tasks while outperforming other open-source baselines. GLM-4.6 is available through the Z.ai API platform, OpenRouter, coding agents (Claude Code, Roo Code, Cline, Kilo Code), and soon as downloadable weights on HuggingFace and ModelScope.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","quantization":"fp4","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-4.6"},{"api":"chat","id":"minimaxi/minimax-m2.5","object":"model","created":1770908502,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000012,"cached_price":6e-8,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000012,"cached_price":6e-8,"output_price":0.0000012}],"max_output_tokens":128000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1 to extend into general office work, reaching fluency in generating and operating Word, Excel, and Powerpoint files, context switching between diverse software environments, and working across different agent and human teams. Scoring 80.2% on SWE-Bench Verified, 51.3% on Multi-SWE-Bench, and 76.3% on BrowseComp, M2.5 is also more token efficient than previous generations, having been trained to optimize its actions and output through planning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5"},{"api":"chat","id":"minimaxi/minimax-m2","object":"model","created":1761252093,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000012,"cached_price":3e-7,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000012,"cached_price":3e-7,"output_price":0.0000012}],"max_output_tokens":128000,"context_window":200000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning, tool use, and multi-step task execution while maintaining low latency and deployment efficiency.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2"},{"api":"chat","id":"minimaxi/minimax-m2.7","object":"model","created":1773836697,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000012,"cached_price":6e-8,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000012,"cached_price":6e-8,"output_price":0.0000012}],"max_output_tokens":128000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent collaboration, enabling it to plan, execute, and refine complex tasks across dynamic environments.\n\nTrained for production-grade performance, M2.7 handles workflows such as live debugging, root cause analysis, financial modeling, and full document generation across Word, Excel, and PowerPoint. It delivers strong results on benchmarks including 56.2% on SWE-Pro and 57.0% on Terminal Bench 2, while achieving a 1495 ELO on GDPval-AA, setting a new standard for multi-agent systems operating in real-world digital workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.7"},{"api":"chat","id":"minimaxi/minimax-m2.7-highspeed","object":"model","created":1773915388,"updated":1787843415,"owned_by":"system","input_price":6e-7,"caching_price":0.0000012,"cached_price":6e-8,"output_price":0.0000024,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"caching_price":0.0000012,"cached_price":6e-8,"output_price":0.0000024}],"max_output_tokens":128000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent collaboration, enabling it to plan, execute, and refine complex tasks across dynamic environments.\n\nTrained for production-grade performance, M2.7 handles workflows such as live debugging, root cause analysis, financial modeling, and full document generation across Word, Excel, and PowerPoint. It delivers strong results on benchmarks including 56.2% on SWE-Pro and 57.0% on Terminal Bench 2, while achieving a 1495 ELO on GDPval-AA, setting a new standard for multi-agent systems operating in real-world digital workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.7-highspeed"},{"api":"chat","id":"minimaxi/minimax-m2.5-highspeed","object":"model","created":1770891388,"updated":1787843415,"owned_by":"system","input_price":6e-7,"caching_price":0.0000024,"cached_price":6e-8,"output_price":0.0000024,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"caching_price":0.0000024,"cached_price":6e-8,"output_price":0.0000024}],"max_output_tokens":128000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1 to extend into general office work, reaching fluency in generating and operating Word, Excel, and Powerpoint files, context switching between diverse software environments, and working across different agent and human teams. Scoring 80.2% on SWE-Bench Verified, 51.3% on Multi-SWE-Bench, and 76.3% on BrowseComp, M2.5 is also more token efficient than previous generations, having been trained to optimize its actions and output through planning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5-highspeed"},{"api":"chat","id":"minimaxi/minimax-m3","object":"model","created":1780245374,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":6e-8,"output_price":0.0000012,"discount_percentage":50,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":6e-8,"output_price":0.0000012}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax M3 is a frontier multimodal model with a 1M-token context window built on MiniMax Sparse Attention (MSA). It reaches frontier-level performance on coding and agentic tasks, surpassing GPT-5.5 and Gemini 3.1 Pro on SWE-Bench Pro and approaching Claude Opus 4.7. It natively supports image and video input and is the first open-weight model to combine frontier coding, ultra-long context, and native multimodality.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://platform.minimax.io/docs/guides/text-generation","geolocation":"global","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m3"},{"api":"chat","id":"google/gemini-3.6-flash:flex","object":"model","created":1784646733,"updated":1787843415,"owned_by":"system","input_price":3.75e-7,"cached_price":3.75e-8,"output_price":0.000001875,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.75e-7,"cached_price":3.75e-8,"output_price":0.000001875}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.6 Flash is the latest Flash model in the Gemini series, delivering frontier level intelligence at high speed and low cost. Built for the agentic era, it excels at sub agent deployment, multi step workflows, and long horizon tasks at scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.6-flash"},{"api":"chat","id":"google/gemini-3.1-pro-preview","object":"model","created":1771509627,"updated":1788220514,"owned_by":"system","input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012},{"prompt_tokens_threshold":200000,"input_price":0.000004,"caching_price":0.000009,"cached_price":4e-7,"output_price":0.000018}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Pro is the next iteration in the Gemini 3 series of models, a suite of highly capable, natively multimodal reasoning models. As of this model card’s date of publication, Gemini 3.1 Pro is Google’s most advanced model for complex tasks. Geminin 3.1 Pro can comprehend vast datasets and challenging problems from massively multimodal information sources, including text, audio, images, video, and entire code repositories.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-pro-preview"},{"api":"chat","id":"google/gemini-3-pro-preview:flex","object":"model","created":1771509627,"updated":1788220514,"owned_by":"system","input_price":0.000001,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000006},{"prompt_tokens_threshold":200000,"input_price":0.000002,"caching_price":0.000009,"cached_price":4e-7,"output_price":0.000009}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Pro is the next iteration in the Gemini 3 series of models, a suite of highly capable, natively multimodal reasoning models. As of this model card’s date of publication, Gemini 3.1 Pro is Google’s most advanced model for complex tasks. Geminin 3.1 Pro can comprehend vast datasets and challenging problems from massively multimodal information sources, including text, audio, images, video, and entire code repositories.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-pro-preview"},{"api":"chat","id":"google/gemma-4-31b-it","object":"model","created":1775148486,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":8192,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function calling, and multilingual support across 140+ languages. Strong on coding, reasoning, and document understanding tasks. Apache 2.0 license.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"google","model_canonical_name":"gemma-4-31b-it"},{"api":"chat","id":"google/gemini-3-pro-image","object":"model","created":1781754054,"updated":1788220122,"owned_by":"system","input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012}],"max_output_tokens":32768,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3 Pro Image, or Gemini 3 Pro (with Nano Banana), is designed to tackle the most challenging image generation by incorporating state-of-the-art reasoning capabilities. It's the best model for complex and multi-turn image generation and editing, having improved accuracy and enhanced image quality.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3-pro-image"},{"api":"chat","id":"google/gemini-3.5-flash-lite","object":"model","created":1784646726,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":3e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":3e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash-Lite is a high efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi agent workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash-lite"},{"api":"chat","id":"google/gemini-3.5-flash-lite:flex","object":"model","created":1784646726,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":2e-8,"output_price":0.00000125,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":2e-8,"output_price":0.00000125}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash-Lite is a high efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi agent workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash-lite"},{"api":"chat","id":"google/gemini-3.1-flash-lite:flex","object":"model","created":1778168828,"updated":1787843415,"owned_by":"system","input_price":1.25e-7,"cached_price":1.25e-8,"output_price":7.5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.25e-7,"cached_price":1.25e-8,"output_price":7.5e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Flash Lite Preview is the most cost-efficient model in the Gemini family, optimized for high-volume, low-latency tasks. It delivers fast responses with solid quality for everyday use cases including summarization, classification, and simple reasoning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-flash-lite"},{"api":"chat","id":"google/gemini-2.5-flash:flex","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":3e-8,"output_price":0.00000125,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":3e-8,"output_price":0.00000125}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"google/gemini-3.1-pro-preview:flex","object":"model","created":1771509627,"updated":1788220514,"owned_by":"system","input_price":0.000001,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000006},{"prompt_tokens_threshold":200000,"input_price":0.000002,"caching_price":0.000009,"cached_price":4e-7,"output_price":0.000009}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Pro is the next iteration in the Gemini 3 series of models, a suite of highly capable, natively multimodal reasoning models. As of this model card’s date of publication, Gemini 3.1 Pro is Google’s most advanced model for complex tasks. Geminin 3.1 Pro can comprehend vast datasets and challenging problems from massively multimodal information sources, including text, audio, images, video, and entire code repositories.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-pro-preview"},{"api":"chat","id":"google/gemini-3.6-flash","object":"model","created":1784646733,"updated":1787843415,"owned_by":"system","input_price":0.0000015,"cached_price":1.5e-7,"output_price":0.000007,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000015,"cached_price":1.5e-7,"output_price":0.000007}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.6 Flash is the latest Flash model in the Gemini series, delivering frontier level intelligence at high speed and low cost. Built for the agentic era, it excels at sub agent deployment, multi step workflows, and long horizon tasks at scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.6-flash"},{"api":"chat","id":"google/gemini-3-flash-preview:flex","object":"model","created":1765987078,"updated":1787843415,"owned_by":"system","input_price":2.5e-7,"cached_price":5e-8,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"cached_price":5e-8,"output_price":0.0000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3 Flash Preview is designed to deliver strong agentic capabilities (near-Pro level) at substantial speed and value. Making it perfect for engaging multi-turn chats, and collaborating back and forth with your coding agent without getting out of flow. Compared to 2.5 Flash it delivers significant improvements across the board.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3-flash-preview"},{"api":"chat","id":"google/gemini-3-pro-preview","object":"model","created":1771509627,"updated":1788220514,"owned_by":"system","input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012},{"prompt_tokens_threshold":200000,"input_price":0.000004,"caching_price":0.000009,"cached_price":4e-7,"output_price":0.000018}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Pro is the next iteration in the Gemini 3 series of models, a suite of highly capable, natively multimodal reasoning models. As of this model card’s date of publication, Gemini 3.1 Pro is Google’s most advanced model for complex tasks. Geminin 3.1 Pro can comprehend vast datasets and challenging problems from massively multimodal information sources, including text, audio, images, video, and entire code repositories.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-pro-preview"},{"api":"chat","id":"google/gemini-3-flash-preview","object":"model","created":1765987078,"updated":1787843415,"owned_by":"system","input_price":5e-7,"cached_price":5e-8,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"cached_price":5e-8,"output_price":0.000003}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3 Flash Preview is designed to deliver strong agentic capabilities (near-Pro level) at substantial speed and value. Making it perfect for engaging multi-turn chats, and collaborating back and forth with your coding agent without getting out of flow. Compared to 2.5 Flash it delivers significant improvements across the board.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3-flash-preview"},{"api":"chat","id":"google/gemini-2.5-flash-lite:flex","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":5e-8,"cached_price":1e-8,"output_price":2e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"cached_price":1e-8,"output_price":2e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite","retires":1792108800},{"api":"chat","id":"google/gemini-3.1-flash-lite","object":"model","created":1778168828,"updated":1787843415,"owned_by":"system","input_price":2.5e-7,"cached_price":2.5e-8,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"cached_price":2.5e-8,"output_price":0.0000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Flash Lite Preview is the most cost-efficient model in the Gemini family, optimized for high-volume, low-latency tasks. It delivers fast responses with solid quality for everyday use cases including summarization, classification, and simple reasoning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-flash-lite"},{"api":"chat","id":"google/gemini-3.5-flash:flex","object":"model","created":1779193800,"updated":1787843415,"owned_by":"system","input_price":7.5e-7,"caching_price":0.000001583,"cached_price":8e-8,"output_price":0.0000045,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.5e-7,"caching_price":0.000001583,"cached_price":8e-8,"output_price":0.0000045}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash provides sustained frontier-level intelligence optimized for real-world tasks at a higher speed and lower cost. Designed for the agentic era, it excels at sub-agent deployment, multi-step workflows, and long-horizon tasks at scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash"},{"api":"chat","id":"google/gemini-3.1-flash-image","object":"model","created":1781754065,"updated":1788220122,"owned_by":"system","input_price":5e-7,"caching_price":0,"cached_price":0,"output_price":0.000002,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"caching_price":0,"cached_price":0,"output_price":0.000002}],"max_output_tokens":65536,"context_window":131072,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Flash Image is optimized for image understanding and generation and offers a balance of price and performance.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-flash-image","retires":1782345600},{"api":"chat","id":"google/gemini-3.5-flash","object":"model","created":1779193800,"updated":1787843415,"owned_by":"system","input_price":0.0000015,"caching_price":0.000001583,"cached_price":1.5e-7,"output_price":0.000009,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000015,"caching_price":0.000001583,"cached_price":1.5e-7,"output_price":0.000009}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash provides sustained frontier-level intelligence optimized for real-world tasks at a higher speed and lower cost. Designed for the agentic era, it excels at sub-agent deployment, multi-step workflows, and long-horizon tasks at scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash"},{"api":"chat","id":"google/gemini-2.5-pro","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"google/gemini-2.5-pro:flex","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":6.25e-7,"caching_price":0.000002375,"cached_price":1.25e-7,"output_price":0.000005,"pricing":[{"prompt_tokens_threshold":0,"input_price":6.25e-7,"caching_price":0.000002375,"cached_price":1.25e-7,"output_price":0.000005},{"prompt_tokens_threshold":200000,"input_price":0.00000125,"caching_price":0.00000475,"cached_price":2.5e-7,"output_price":0.0000075}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro","retires":1792108800},{"api":"chat","id":"google/gemini-2.5-flash-lite","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite","retires":1792108800},{"api":"chat","id":"google/gemini-2.5-flash","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash","retires":1792108800},{"api":"chat","id":"google/gemini-2.5-flash-image","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"bedrock/claude-opus-4-8","object":"model","created":1779905091,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.8 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.8 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-8"},{"api":"chat","id":"bedrock/gpt-5.6-terra@us-east-1","object":"model","created":1783555200,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"cached_price":2.2e-7,"output_price":0.0000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"cached_price":2.2e-7,"output_price":0.0000132}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Terra is OpenAI's balanced frontier model for everyday production workloads, delivering near flagship reasoning, coding, and agentic performance at roughly half the cost of the previous generation. It features a 1M+ token context window with support for text and image inputs, server side tool calling, and prompt caching.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-terra"},{"api":"chat","id":"bedrock/claude-sonnet-4@eu-central-1","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"bedrock/claude-opus-5@eu-west-3","object":"model","created":1784911391,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 5 is the most capable model in Anthropic's Opus family, built for long running, asynchronous agents operating with full autonomy. It advances the coding and agentic strengths of Opus 4.8 with stronger planning, more reliable execution across extended multi step workflows, and better judgment on when to act and when to escalate. It excels in asynchronous agent pipelines spanning large codebases, multi stage debugging, and end to end project orchestration. Beyond coding, Opus 5 delivers frontier knowledge work, from drafting documents and building presentations to deep data analysis, and maintains coherence across very long outputs and extended sessions, making it the default choice for work that requires persistence, judgment, and follow through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5"},{"api":"chat","id":"bedrock/claude-opus-5@eu-north-1","object":"model","created":1784911391,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 5 is the most capable model in Anthropic's Opus family, built for long running, asynchronous agents operating with full autonomy. It advances the coding and agentic strengths of Opus 4.8 with stronger planning, more reliable execution across extended multi step workflows, and better judgment on when to act and when to escalate. It excels in asynchronous agent pipelines spanning large codebases, multi stage debugging, and end to end project orchestration. Beyond coding, Opus 5 delivers frontier knowledge work, from drafting documents and building presentations to deep data analysis, and maintains coherence across very long outputs and extended sessions, making it the default choice for work that requires persistence, judgment, and follow through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5"},{"api":"chat","id":"bedrock/gpt-5.6-sol@us-east-1","object":"model","created":1783555200,"updated":1787843415,"owned_by":"system","input_price":0.0000044,"cached_price":4.4e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"cached_price":4.4e-7,"output_price":0.000022}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Sol is OpenAI's most capable frontier model, delivering state of the art reasoning and agentic performance across coding, cybersecurity, and scientific research. It features a 1M+ token context window with support for text and image inputs, server side tool calling, and prompt caching, and is designed for the hardest multi step reasoning and long horizon agent workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-sol"},{"api":"chat","id":"bedrock/claude-sonnet-4@us-east-1","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"bedrock/claude-opus-5","object":"model","created":1784911391,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 5 is the most capable model in Anthropic's Opus family, built for long running, asynchronous agents operating with full autonomy. It advances the coding and agentic strengths of Opus 4.8 with stronger planning, more reliable execution across extended multi step workflows, and better judgment on when to act and when to escalate. It excels in asynchronous agent pipelines spanning large codebases, multi stage debugging, and end to end project orchestration. Beyond coding, Opus 5 delivers frontier knowledge work, from drafting documents and building presentations to deep data analysis, and maintains coherence across very long outputs and extended sessions, making it the default choice for work that requires persistence, judgment, and follow through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5"},{"api":"chat","id":"bedrock/claude-sonnet-4-5@us-east-2","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165},{"prompt_tokens_threshold":200000,"input_price":0.0000066,"caching_price":0.00000825,"caching_5m_price":0.00000825,"caching_1h_price":0.0000132,"cached_price":6e-7,"output_price":0.00002475}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-4-5","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-4-5@us-west-2","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165},{"prompt_tokens_threshold":200000,"input_price":0.0000066,"caching_price":0.00000825,"caching_5m_price":0.00000825,"caching_1h_price":0.0000132,"cached_price":6e-7,"output_price":0.00002475}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-4-5@eu-west-1","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165},{"prompt_tokens_threshold":200000,"input_price":0.0000066,"caching_price":0.00000825,"caching_5m_price":0.00000825,"caching_1h_price":0.0000132,"cached_price":6e-7,"output_price":0.00002475}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"bedrock/gpt-5.4@us-west-2","object":"model","created":1772734352,"updated":1787843415,"owned_by":"system","input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 is OpenAI's latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling high-context reasoning, coding, and multimodal analysis within the same workflow. The model delivers improved performance in coding, document understanding, tool use, and instruction following, capable of generating production-quality code, synthesizing information across multiple sources, and executing complex multi-step workflows with fewer iterations and greater token efficiency.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"bedrock/gpt-5.6-sol@us-east-2","object":"model","created":1783555200,"updated":1787843415,"owned_by":"system","input_price":0.0000044,"cached_price":4.4e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"cached_price":4.4e-7,"output_price":0.000022}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Sol is OpenAI's most capable frontier model, delivering state of the art reasoning and agentic performance across coding, cybersecurity, and scientific research. It features a 1M+ token context window with support for text and image inputs, server side tool calling, and prompt caching, and is designed for the hardest multi step reasoning and long horizon agent workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-sol"},{"api":"chat","id":"bedrock/claude-opus-4-7@eu-west-3","object":"model","created":1776418380,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.7 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.7 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.7 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-7"},{"api":"chat","id":"bedrock/claude-haiku-4-5@eu-north-1","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"caching_price":0.000001375,"cached_price":1.1e-7,"output_price":0.0000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"caching_price":0.000001375,"caching_5m_price":0.000001375,"caching_1h_price":0.0000022,"cached_price":1.1e-7,"output_price":0.0000055}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-4-5@eu-central-1","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165},{"prompt_tokens_threshold":200000,"input_price":0.0000066,"caching_price":0.00000825,"caching_5m_price":0.00000825,"caching_1h_price":0.0000132,"cached_price":6e-7,"output_price":0.00002475}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"bedrock/claude-haiku-4-5@us-east-2","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"caching_price":0.000001375,"cached_price":1.1e-7,"output_price":0.0000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"caching_price":0.000001375,"caching_5m_price":0.000001375,"caching_1h_price":0.0000022,"cached_price":1.1e-7,"output_price":0.0000055}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-4-6@us-east-2","object":"model","created":1771342990,"updated":1787843415,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.6 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"bedrock/gpt-5.6-luna@us-east-2","object":"model","created":1783555200,"updated":1787843415,"owned_by":"system","input_price":2.2e-7,"cached_price":2.2e-8,"output_price":0.00000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.2e-7,"cached_price":2.2e-8,"output_price":0.00000132}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Luna is OpenAI's fast, cost efficient model for high volume tasks such as classification, extraction, summarization, and lightweight agent steps. It features a 1M+ token context window with support for text and image inputs, server side tool calling, and prompt caching.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-luna"},{"api":"chat","id":"bedrock/claude-opus-4-5@us-east-1","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-4@eu-west-3","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"bedrock/minimax-m2.5@eu-west-1","object":"model","created":1770908502,"updated":1787843415,"owned_by":"system","input_price":3.6e-7,"output_price":0.00000144,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.6e-7,"output_price":0.00000144}],"max_output_tokens":16000,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax M2.5 is MiniMax's latest large language model, delivering strong performance across reasoning, coding, and tool use with improved efficiency for production workloads.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5"},{"api":"chat","id":"bedrock/claude-sonnet-5@us-east-2","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"bedrock/claude-opus-5-5@eu-central-1","object":"model","created":1790095261,"updated":1790095261,"owned_by":"system","input_price":0.0000044,"caching_price":0.0000055,"cached_price":2.2e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"caching_price":0.0000055,"caching_5m_price":0.0000055,"caching_1h_price":0.0000088,"cached_price":2.2e-7,"output_price":0.000022}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 5.5 is a step-change improvement over Opus 5, with the biggest gains in agentic coding, long-running agentic tasks, and knowledge work. It is also a much better collaborator on long-running work. It reports back in plain language on what it did, what it found, and what it needs next.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5-5"},{"api":"chat","id":"bedrock/claude-sonnet-5@eu-west-1","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"bedrock/claude-opus-4-5","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"bedrock/claude-haiku-4-5@eu-west-3","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"caching_price":0.000001375,"cached_price":1.1e-7,"output_price":0.0000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"caching_price":0.000001375,"caching_5m_price":0.000001375,"caching_1h_price":0.0000022,"cached_price":1.1e-7,"output_price":0.0000055}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"bedrock/gpt-5.5@us-east-2","object":"model","created":1777051893,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.000033,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.000033}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.5 is OpenAI's frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large-scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.5"},{"api":"chat","id":"bedrock/claude-opus-5-5@eu-west-1","object":"model","created":1790095261,"updated":1790095261,"owned_by":"system","input_price":0.0000044,"caching_price":0.0000055,"cached_price":2.2e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"caching_price":0.0000055,"caching_5m_price":0.0000055,"caching_1h_price":0.0000088,"cached_price":2.2e-7,"output_price":0.000022}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 5.5 is a step-change improvement over Opus 5, with the biggest gains in agentic coding, long-running agentic tasks, and knowledge work. It is also a much better collaborator on long-running work. It reports back in plain language on what it did, what it found, and what it needs next.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5-5"},{"api":"chat","id":"bedrock/claude-opus-5-5","object":"model","created":1790095744,"updated":1790095744,"owned_by":"system","input_price":0.000004,"caching_price":0.000005,"cached_price":2e-7,"output_price":0.00002,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000004,"caching_price":0.000005,"caching_5m_price":0.000005,"caching_1h_price":0.000008,"cached_price":2e-7,"output_price":0.00002}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 5.5 is a step-change improvement over Opus 5, with the biggest gains in agentic coding, long-running agentic tasks, and knowledge work. It is also a much better collaborator on long-running work. It reports back in plain language on what it did, what it found, and what it needs next.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5-5"},{"api":"chat","id":"bedrock/claude-sonnet-4-6@eu-west-1","object":"model","created":1771342990,"updated":1787843415,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.6 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"bedrock/kimi-k2.5@us-west-2","object":"model","created":1769487076,"updated":1787843415,"owned_by":"system","input_price":6e-7,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"output_price":0.000003}],"max_output_tokens":16000,"context_window":128000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.5 is an open-source, native multimodal agentic model built through continual pretraining on approximately 15 trillion mixed visual and text tokens atop Kimi-K2-Base. It seamlessly integrates vision and language understanding with advanced agentic capabilities, instant and thinking modes, as well as conversational and agentic paradigms.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.5"},{"api":"chat","id":"bedrock/claude-opus-4-8@eu-central-1","object":"model","created":1779905091,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.8 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.8 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-8"},{"api":"chat","id":"bedrock/claude-haiku-4-5","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.000001,"caching_price":0.00000125,"cached_price":1e-7,"output_price":0.000005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"caching_price":0.00000125,"caching_5m_price":0.00000125,"caching_1h_price":0.000002,"cached_price":1e-7,"output_price":0.000005}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-5@eu-west-2","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"uk","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"bedrock/claude-opus-4-5@eu-west-1","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-5@eu-north-1","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"bedrock/claude-haiku-4-5@ap-northeast-1","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"caching_price":0.000001375,"cached_price":1.1e-7,"output_price":0.0000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"caching_price":0.000001375,"caching_5m_price":0.000001375,"caching_1h_price":0.0000022,"cached_price":1.1e-7,"output_price":0.0000055}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"ap","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"bedrock/minimax-m2.5@us-east-2","object":"model","created":1770908502,"updated":1787843415,"owned_by":"system","input_price":3e-7,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"output_price":0.0000012}],"max_output_tokens":16000,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax M2.5 is MiniMax's latest large language model, delivering strong performance across reasoning, coding, and tool use with improved efficiency for production workloads.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5"},{"api":"chat","id":"bedrock/claude-opus-4-5@us-west-2","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"bedrock/minimax-m2.5@eu-central-1","object":"model","created":1770908502,"updated":1787843415,"owned_by":"system","input_price":3.6e-7,"output_price":0.00000144,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.6e-7,"output_price":0.00000144}],"max_output_tokens":16000,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax M2.5 is MiniMax's latest large language model, delivering strong performance across reasoning, coding, and tool use with improved efficiency for production workloads.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5"},{"api":"chat","id":"bedrock/claude-opus-4-8@eu-west-3","object":"model","created":1779905091,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.8 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.8 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-8"},{"api":"chat","id":"bedrock/claude-opus-5-5@eu-west-3","object":"model","created":1790095261,"updated":1790095261,"owned_by":"system","input_price":0.0000044,"caching_price":0.0000055,"cached_price":2.2e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"caching_price":0.0000055,"caching_5m_price":0.0000055,"caching_1h_price":0.0000088,"cached_price":2.2e-7,"output_price":0.000022}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 5.5 is a step-change improvement over Opus 5, with the biggest gains in agentic coding, long-running agentic tasks, and knowledge work. It is also a much better collaborator on long-running work. It reports back in plain language on what it did, what it found, and what it needs next.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5-5"},{"api":"chat","id":"bedrock/gpt-5.5@us-east-1","object":"model","created":1777051893,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.000033,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.000033}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.5 is OpenAI's frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large-scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.5"},{"api":"chat","id":"bedrock/claude-opus-4-5@us-east-2","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-5-5","object":"model","created":1790620530,"updated":1790620530,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"caching_5m_price":0.0000025,"caching_1h_price":0.000004,"cached_price":2e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5.5 is a step-change improvement over Sonnet 5. It performs great with well-scoped everyday tasks like building features, fixing bugs, and creating polished documents, slides, and spreadsheets. Sonnet 5.5 also improves on communication and writes more clearly than Sonnet 5.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5-5"},{"api":"chat","id":"bedrock/gpt-5.4@us-east-2","object":"model","created":1772734352,"updated":1787843415,"owned_by":"system","input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 is OpenAI's latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling high-context reasoning, coding, and multimodal analysis within the same workflow. The model delivers improved performance in coding, document understanding, tool use, and instruction following, capable of generating production-quality code, synthesizing information across multiple sources, and executing complex multi-step workflows with fewer iterations and greater token efficiency.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"bedrock/claude-sonnet-4-6@us-east-1","object":"model","created":1771342990,"updated":1787843415,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.6 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"bedrock/kimi-k2.5@eu-west-2","object":"model","created":1769487076,"updated":1787843415,"owned_by":"system","input_price":7.2e-7,"caching_price":7.2e-7,"cached_price":7.2e-7,"output_price":0.0000036,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.2e-7,"caching_price":7.2e-7,"cached_price":7.2e-7,"output_price":0.0000036}],"max_output_tokens":16000,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.5 is an open-source, native multimodal agentic model built through continual pretraining on approximately 15 trillion mixed visual and text tokens atop Kimi-K2-Base. It seamlessly integrates vision and language understanding with advanced agentic capabilities, instant and thinking modes, as well as conversational and agentic paradigms.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"uk","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.5"},{"api":"chat","id":"bedrock/claude-opus-4-8@eu-west-1","object":"model","created":1779905091,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.8 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.8 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-8"},{"api":"chat","id":"bedrock/claude-sonnet-4@eu-west-1","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"bedrock/gpt-5.6-luna@us-west-2","object":"model","created":1783555200,"updated":1787843415,"owned_by":"system","input_price":2.2e-7,"cached_price":2.2e-8,"output_price":0.00000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.2e-7,"cached_price":2.2e-8,"output_price":0.00000132}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Luna is OpenAI's fast, cost efficient model for high volume tasks such as classification, extraction, summarization, and lightweight agent steps. It features a 1M+ token context window with support for text and image inputs, server side tool calling, and prompt caching.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-luna"},{"api":"chat","id":"bedrock/kimi-k2.5@us-east-1","object":"model","created":1769487076,"updated":1787843415,"owned_by":"system","input_price":6e-7,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"output_price":0.000003}],"max_output_tokens":16000,"context_window":128000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.5 is an open-source, native multimodal agentic model built through continual pretraining on approximately 15 trillion mixed visual and text tokens atop Kimi-K2-Base. It seamlessly integrates vision and language understanding with advanced agentic capabilities, instant and thinking modes, as well as conversational and agentic paradigms.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.5"},{"api":"chat","id":"bedrock/claude-sonnet-4-5@eu-north-1","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165},{"prompt_tokens_threshold":200000,"input_price":0.0000066,"caching_price":0.00000825,"caching_5m_price":0.00000825,"caching_1h_price":0.0000132,"cached_price":6e-7,"output_price":0.00002475}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"bedrock/claude-opus-5@eu-central-1","object":"model","created":1784911391,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 5 is the most capable model in Anthropic's Opus family, built for long running, asynchronous agents operating with full autonomy. It advances the coding and agentic strengths of Opus 4.8 with stronger planning, more reliable execution across extended multi step workflows, and better judgment on when to act and when to escalate. It excels in asynchronous agent pipelines spanning large codebases, multi stage debugging, and end to end project orchestration. Beyond coding, Opus 5 delivers frontier knowledge work, from drafting documents and building presentations to deep data analysis, and maintains coherence across very long outputs and extended sessions, making it the default choice for work that requires persistence, judgment, and follow through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5"},{"api":"chat","id":"bedrock/claude-sonnet-4","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"bedrock/gpt-5.6-terra@us-east-2","object":"model","created":1783555200,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"cached_price":2.2e-7,"output_price":0.0000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"cached_price":2.2e-7,"output_price":0.0000132}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Terra is OpenAI's balanced frontier model for everyday production workloads, delivering near flagship reasoning, coding, and agentic performance at roughly half the cost of the previous generation. It features a 1M+ token context window with support for text and image inputs, server side tool calling, and prompt caching.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-terra"},{"api":"chat","id":"bedrock/claude-sonnet-4-6@ap-northeast-1","object":"model","created":1771342990,"updated":1787843415,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.6 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"ap","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"bedrock/claude-sonnet-4@us-west-2","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"bedrock/minimax-m2.5@eu-north-1","object":"model","created":1770908502,"updated":1787843415,"owned_by":"system","input_price":3.6e-7,"output_price":0.00000144,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.6e-7,"output_price":0.00000144}],"max_output_tokens":16000,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax M2.5 is MiniMax's latest large language model, delivering strong performance across reasoning, coding, and tool use with improved efficiency for production workloads.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5"},{"api":"chat","id":"bedrock/claude-opus-4-5@eu-west-3","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"bedrock/gpt-5.6-luna@us-east-1","object":"model","created":1783555200,"updated":1787843415,"owned_by":"system","input_price":2.2e-7,"cached_price":2.2e-8,"output_price":0.00000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.2e-7,"cached_price":2.2e-8,"output_price":0.00000132}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Luna is OpenAI's fast, cost efficient model for high volume tasks such as classification, extraction, summarization, and lightweight agent steps. It features a 1M+ token context window with support for text and image inputs, server side tool calling, and prompt caching.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-luna"},{"api":"chat","id":"bedrock/claude-opus-4-5@eu-central-1","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"bedrock/gpt-5.6-terra@us-west-2","object":"model","created":1783555200,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"cached_price":2.2e-7,"output_price":0.0000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"cached_price":2.2e-7,"output_price":0.0000132}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Terra is OpenAI's balanced frontier model for everyday production workloads, delivering near flagship reasoning, coding, and agentic performance at roughly half the cost of the previous generation. It features a 1M+ token context window with support for text and image inputs, server side tool calling, and prompt caching.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-terra"},{"api":"chat","id":"bedrock/claude-sonnet-5@us-east-1","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"bedrock/claude-sonnet-5","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"caching_5m_price":0.0000025,"caching_1h_price":0.000004,"cached_price":2e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"bedrock/claude-opus-4-8@eu-north-1","object":"model","created":1779905091,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.8 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.8 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-8"},{"api":"chat","id":"bedrock/minimax-m2.5@us-west-2","object":"model","created":1770908502,"updated":1787843415,"owned_by":"system","input_price":3e-7,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"output_price":0.0000012}],"max_output_tokens":16000,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax M2.5 is MiniMax's latest large language model, delivering strong performance across reasoning, coding, and tool use with improved efficiency for production workloads.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5"},{"api":"chat","id":"bedrock/claude-opus-4-8@ap-northeast-1","object":"model","created":1779905091,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.8 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.8 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"ap","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-8"},{"api":"chat","id":"bedrock/claude-haiku-4-5@eu-central-1","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"caching_price":0.000001375,"cached_price":1.1e-7,"output_price":0.0000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"caching_price":0.000001375,"caching_5m_price":0.000001375,"caching_1h_price":0.0000022,"cached_price":1.1e-7,"output_price":0.0000055}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"bedrock/claude-haiku-4-5@eu-west-1","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"caching_price":0.000001375,"cached_price":1.1e-7,"output_price":0.0000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"caching_price":0.000001375,"caching_5m_price":0.000001375,"caching_1h_price":0.0000022,"cached_price":1.1e-7,"output_price":0.0000055}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-4-5@eu-west-3","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165},{"prompt_tokens_threshold":200000,"input_price":0.0000066,"caching_price":0.00000825,"caching_5m_price":0.00000825,"caching_1h_price":0.0000132,"cached_price":6e-7,"output_price":0.00002475}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"bedrock/claude-opus-5-5@eu-north-1","object":"model","created":1790095261,"updated":1790095261,"owned_by":"system","input_price":0.0000044,"caching_price":0.0000055,"cached_price":2.2e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"caching_price":0.0000055,"caching_5m_price":0.0000055,"caching_1h_price":0.0000088,"cached_price":2.2e-7,"output_price":0.000022}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 5.5 is a step-change improvement over Opus 5, with the biggest gains in agentic coding, long-running agentic tasks, and knowledge work. It is also a much better collaborator on long-running work. It reports back in plain language on what it did, what it found, and what it needs next.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5-5"},{"api":"chat","id":"bedrock/claude-opus-4-7","object":"model","created":1776418380,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.7 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.7 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.7 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-7"},{"api":"chat","id":"bedrock/kimi-k2.5@eu-north-1","object":"model","created":1769487076,"updated":1787843415,"owned_by":"system","input_price":7.2e-7,"caching_price":7.2e-7,"cached_price":7.2e-7,"output_price":0.0000036,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.2e-7,"caching_price":7.2e-7,"cached_price":7.2e-7,"output_price":0.0000036}],"max_output_tokens":16000,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.5 is an open-source, native multimodal agentic model built through continual pretraining on approximately 15 trillion mixed visual and text tokens atop Kimi-K2-Base. It seamlessly integrates vision and language understanding with advanced agentic capabilities, instant and thinking modes, as well as conversational and agentic paradigms.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.5"},{"api":"chat","id":"bedrock/claude-sonnet-4@eu-north-1","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"bedrock/claude-sonnet-5@ap-southeast-2","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"ap","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"bedrock/claude-opus-4-6","object":"model","created":1770315610,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.6 is Anthropic's most powerful model yet and the best coding model in the world.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-6"},{"api":"chat","id":"bedrock/claude-opus-4-7@eu-north-1","object":"model","created":1776418380,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.7 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.7 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.7 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-7"},{"api":"chat","id":"bedrock/claude-sonnet-4-6","object":"model","created":1771342990,"updated":1787843415,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.6 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"bedrock/claude-haiku-4-5@us-west-2","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"caching_price":0.000001375,"cached_price":1.1e-7,"output_price":0.0000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"caching_price":0.000001375,"caching_5m_price":0.000001375,"caching_1h_price":0.0000022,"cached_price":1.1e-7,"output_price":0.0000055}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"bedrock/claude-haiku-4-5@us-east-1","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"caching_price":0.000001375,"cached_price":1.1e-7,"output_price":0.0000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"caching_price":0.000001375,"caching_5m_price":0.000001375,"caching_1h_price":0.0000022,"cached_price":1.1e-7,"output_price":0.0000055}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"bedrock/claude-opus-4-5@eu-north-1","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-4-6@eu-north-1","object":"model","created":1771342990,"updated":1787843415,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.6 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"bedrock/gpt-5.4@us-east-1","object":"model","created":1772734352,"updated":1787843415,"owned_by":"system","input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 is OpenAI's latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling high-context reasoning, coding, and multimodal analysis within the same workflow. The model delivers improved performance in coding, document understanding, tool use, and instruction following, capable of generating production-quality code, synthesizing information across multiple sources, and executing complex multi-step workflows with fewer iterations and greater token efficiency.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"bedrock/claude-sonnet-5@eu-central-1","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"bedrock/claude-sonnet-4-6@us-west-2","object":"model","created":1771342990,"updated":1787843415,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.6 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"bedrock/claude-sonnet-4@us-east-2","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"bedrock/claude-fable-5@us-east-1","object":"model","created":1781016885,"updated":1787843415,"owned_by":"system","input_price":0.000011,"caching_price":0.00001375,"cached_price":0.0000011,"output_price":0.000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000011,"caching_price":0.00001375,"caching_5m_price":0.00001375,"caching_1h_price":0.000022,"cached_price":0.0000011,"output_price":0.000055}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Fable 5 is Anthropic's most capable generally available model for autonomous knowledge work and coding. It is a Mythos-class model with robust safeguards. It can handle long-running, complex, and asynchronous tasks where previous models would have needed more frequent check-ins.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-fable-5"},{"api":"chat","id":"bedrock/claude-opus-5@eu-west-1","object":"model","created":1784911391,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 5 is the most capable model in Anthropic's Opus family, built for long running, asynchronous agents operating with full autonomy. It advances the coding and agentic strengths of Opus 4.8 with stronger planning, more reliable execution across extended multi step workflows, and better judgment on when to act and when to escalate. It excels in asynchronous agent pipelines spanning large codebases, multi stage debugging, and end to end project orchestration. Beyond coding, Opus 5 delivers frontier knowledge work, from drafting documents and building presentations to deep data analysis, and maintains coherence across very long outputs and extended sessions, making it the default choice for work that requires persistence, judgment, and follow through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5"},{"api":"chat","id":"bedrock/claude-opus-4-7@eu-west-1","object":"model","created":1776418380,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.7 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.7 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.7 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-7"},{"api":"chat","id":"bedrock/claude-sonnet-4-6@eu-west-3","object":"model","created":1771342990,"updated":1787843415,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.6 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"bedrock/kimi-k2.5@us-east-2","object":"model","created":1769487076,"updated":1787843415,"owned_by":"system","input_price":6e-7,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"output_price":0.000003}],"max_output_tokens":16000,"context_window":128000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.5 is an open-source, native multimodal agentic model built through continual pretraining on approximately 15 trillion mixed visual and text tokens atop Kimi-K2-Base. It seamlessly integrates vision and language understanding with advanced agentic capabilities, instant and thinking modes, as well as conversational and agentic paradigms.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.5"},{"api":"chat","id":"bedrock/minimax-m2.5@us-east-1","object":"model","created":1770908502,"updated":1787843415,"owned_by":"system","input_price":3e-7,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"output_price":0.0000012}],"max_output_tokens":16000,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax M2.5 is MiniMax's latest large language model, delivering strong performance across reasoning, coding, and tool use with improved efficiency for production workloads.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5"},{"api":"chat","id":"bedrock/claude-sonnet-4-5@us-east-1","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165},{"prompt_tokens_threshold":200000,"input_price":0.0000066,"caching_price":0.00000825,"caching_5m_price":0.00000825,"caching_1h_price":0.0000132,"cached_price":6e-7,"output_price":0.00002475}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"bedrock/claude-sonnet-5@us-west-2","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"bedrock/claude-opus-4-7@eu-central-1","object":"model","created":1776418380,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.7 is Anthropic's most capable generally available model, advancing performance across coding, enterprise workflows, and long-running agentic tasks. Coding: Claude Opus 4.7 is built for agentic coding at scale, excelling at long-horizon projects, complex implementations, and polished UI design. It handles the full lifecycle from architecture to deployment, including design-quality UI so senior engineers can delegate complex work with confidence. Enterprise workflows: Claude Opus 4.7 sets the standard for enterprise knowledge work, carrying context across sessions to manage complex, multi-day projects end-to-end. It delivers professional polish on the documents, spreadsheets, and presentations that move work forward.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-7"},{"api":"chat","id":"bedrock/minimax-m2.5@eu-south-1","object":"model","created":1770908502,"updated":1787843415,"owned_by":"system","input_price":3.6e-7,"output_price":0.00000144,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.6e-7,"output_price":0.00000144}],"max_output_tokens":16000,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax M2.5 is MiniMax's latest large language model, delivering strong performance across reasoning, coding, and tool use with improved efficiency for production workloads.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5"},{"api":"chat","id":"bedrock/claude-sonnet-4-6@eu-central-1","object":"model","created":1771342990,"updated":1787843415,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3e-7,"output_price":0.0000165}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.6 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"anthropic/claude-opus-4-8","object":"model","created":1779905091,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 4.8 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.7, it delivers stronger performance on complex, multi-step tasks and more reliable agentic execution across extended workflows. It is especially effective for asynchronous agent pipelines where tasks unfold over time - large codebases, multi-stage debugging, and end-to-end project orchestration.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-8"},{"api":"chat","id":"anthropic/claude-opus-4-6","object":"model","created":1770219050,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.6 is Anthropic's most powerful model yet and the best coding model in the world.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-6"},{"api":"chat","id":"anthropic/claude-fable-5","object":"model","created":1781007515,"updated":1787843415,"owned_by":"system","input_price":0.00001,"caching_price":0.0000125,"cached_price":0.000001,"output_price":0.00005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00001,"caching_price":0.0000125,"caching_5m_price":0.0000125,"caching_1h_price":0.00002,"cached_price":0.000001,"output_price":0.00005}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Fable 5 is Anthropic's most capable generally available model for autonomous knowledge work and coding. It is a Mythos-class model with robust safeguards. It can handle long-running, complex, and asynchronous tasks where previous models would have needed more frequent check-ins.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-fable-5"},{"api":"chat","id":"anthropic/claude-sonnet-4-6","object":"model","created":1771342990,"updated":1787843415,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.6 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"anthropic/claude-haiku-4-5","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.000001,"caching_price":0.00000125,"cached_price":1e-7,"output_price":0.000005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"caching_price":0.00000125,"caching_5m_price":0.00000125,"caching_1h_price":0.000002,"cached_price":1e-7,"output_price":0.000005}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"anthropic/claude-sonnet-5","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"caching_5m_price":0.0000025,"caching_1h_price":0.000004,"cached_price":2e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"anthropic/claude-opus-5-5","object":"model","created":1790098200,"updated":1790098200,"owned_by":"system","input_price":0.000004,"caching_price":0.000005,"cached_price":2e-7,"output_price":0.00002,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000004,"caching_price":0.000005,"caching_5m_price":0.000005,"caching_1h_price":0.000008,"cached_price":2e-7,"output_price":0.00002}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 5.5 is a step-change improvement over Opus 5, with the biggest gains in agentic coding, long-running agentic tasks, and knowledge work. It is also a much better collaborator on long-running work. It reports back in plain language on what it did, what it found, and what it needs next.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5-5"},{"api":"chat","id":"anthropic/claude-opus-4-5","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"anthropic/claude-opus-4-7","object":"model","created":1776351100,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on complex, multi-step tasks and more reliable agentic execution across extended workflows. It is especially effective for asynchronous agent pipelines where tasks unfold over time - large codebases, multi-stage debugging, and end-to-end project orchestration.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-7"},{"api":"chat","id":"anthropic/claude-sonnet-5-5","object":"model","created":1790616534,"updated":1790616534,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"caching_5m_price":0.0000025,"caching_1h_price":0.000004,"cached_price":2e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5.5 is a step-change improvement over Sonnet 5. It performs great with well-scoped everyday tasks like building features, fixing bugs, and creating polished documents, slides, and spreadsheets. Sonnet 5.5 also improves on communication and writes more clearly than Sonnet 5.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5-5"},{"api":"chat","id":"anthropic/claude-fable-5.1","object":"model","created":1788220800,"updated":1788288205,"owned_by":"system","input_price":0.00001,"caching_price":0.0000125,"cached_price":2.5e-7,"output_price":0.00005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00001,"caching_price":0.0000125,"caching_5m_price":0.0000125,"caching_1h_price":0.00002,"cached_price":2.5e-7,"output_price":0.00005}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Fable 5.1 is Anthropic's most capable model, improving on Claude Fable 5 across agentic coding, long running agentic workflows, and knowledge work such as large code refactors, front end and visual code generation, and finance and analysis tasks. It ships a 1M token context window by default with 128K max output, is more token efficient and concise than Fable 5, and supports per message effort control. It is a Mythos class model with robust safeguards.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-fable-5.1"},{"api":"chat","id":"anthropic/claude-sonnet-4-5","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Sonnet 4.5 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"anthropic/claude-opus-5","object":"model","created":1784912544,"updated":1788220122,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 5 is a workhorse model for agentic coding. It is strongest on difficult tasks, including\nmulti-file features, larger refactors, and end-to-end feature work. It also performs well on\neasier tasks like single-turn edits, though the difference from prior models is smaller. Opus\n5 completes full tasks rather than leaving stubs or placeholders.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5"},{"api":"chat","id":"groq/openai/gpt-oss-120b","object":"model","created":1754414231,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":1.5e-7,"output_price":7.5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":1.5e-7,"output_price":7.5e-7}],"max_output_tokens":32768,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"OpenAI's gpt-oss-120b is a powerful open-weight Large Language Model (LLM) designed for advanced chain-of-thought reasoning, agentic workflows, and complex tool-calling.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"openai","model_canonical_name":"gpt-oss-120b"},{"api":"chat","id":"groq/openai/gpt-oss-20b","object":"model","created":1754414229,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":1e-7,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":1e-7,"output_price":5e-7}],"max_output_tokens":32768,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"OpenAI's gpt-oss-20b is a 21-billion parameter, open-weight reasoning model released under the Apache 2.0 license. Designed for local execution on consumer hardware and low-latency workflows, it delivers reasoning capabilities and benchmark performance comparable to the OpenAI o3-mini models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"openai","model_canonical_name":"gpt-oss-20b"},{"api":"chat","id":"inceptron/kimi-k2.6","object":"model","created":1776699402,"updated":1787843415,"owned_by":"system","input_price":8e-7,"cached_price":2e-7,"output_price":0.0000035,"pricing":[{"prompt_tokens_threshold":0,"input_price":8e-7,"cached_price":2e-7,"output_price":0.0000035}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.6 is Moonshot AI's latest open weight reasoning model, built for long horizon coding, agentic execution, and multimodal reasoning. It retains the trillion parameter MoE architecture with roughly 32B active parameters and a 256k token context window, while improving agentic benchmark performance and knowledge reliability over K2.5. Native text, image, and video input plus tool driven workflows make it well suited for coding, research, and complex multi step tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","quantization":"int4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.6"},{"api":"chat","id":"inceptron/minimax-m2.5","object":"model","created":1770908502,"updated":1787843415,"owned_by":"system","input_price":2.8e-7,"cached_price":3e-8,"output_price":0.0000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.8e-7,"cached_price":3e-8,"output_price":0.0000011}],"max_output_tokens":196608,"context_window":196608,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax M2.5 is a state of the art language model built for real world productivity and autonomous agent execution. Trained with large scale reinforcement learning across hundreds of thousands of complex digital environments, it delivers leading performance in coding, search, tool use, and professional office workflows, operating with significantly improved speed and token efficiency. Designed to plan like an architect and act with cost efficient precision, M2.5 extends beyond software development into finance, research, and enterprise grade office tasks, bringing high end agentic capability at a fraction of the typical frontier model cost.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5"},{"api":"chat","id":"inceptron/kimi-k2.7-Code","object":"model","created":1781266361,"updated":1787843415,"owned_by":"system","input_price":7.5e-7,"cached_price":2e-7,"output_price":0.0000035,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.5e-7,"cached_price":2e-7,"output_price":0.0000035}],"max_output_tokens":256000,"context_window":256000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Kimi K2.7 Code is a coding-focused agentic model built on Kimi K2.6, tuned for long-horizon software engineering workflows, multi-step tool use, and end-to-end task completion. It keeps the trillion-parameter MoE architecture with roughly 32B active parameters, a 256K-token context window, native INT4 quantization, and multimodal image and video input support while reducing thinking-token usage compared with K2.6.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data.","geolocation":"eu","quantization":"int4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.7-code"},{"api":"chat","id":"inceptron/glm-5.2","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":0.0000012,"cached_price":2.6e-7,"output_price":0.0000042,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000012,"cached_price":2.6e-7,"output_price":0.0000042}],"max_output_tokens":1000000,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM 5.2 is the latest Z.ai flagship model for long horizon coding, reasoning, and agentic workflows. It improves on GLM 5.1 with a 1M token context window, stronger engineering task performance, multiple thinking effort levels, and architectural changes that reduce long context inference cost while improving speculative decoding. Released under the MIT license, it is designed for sustained work over large repositories, complex tool use, and multi step technical tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","quantization":"fp4","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"sakana/fugu-ultra","object":"model","created":1782132799,"updated":1787843415,"owned_by":"system","input_price":0.000005,"cached_price":5e-7,"output_price":0.00003,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"cached_price":5e-7,"output_price":0.00003},{"prompt_tokens_threshold":272000,"input_price":0.00001,"cached_price":0.000001,"output_price":0.000045}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Fugu Ultra is built on a pool of publicly accessible frontier models, rather than running as a single model. It coordinates several models, routing work to 1-3 agents depending on the problem and combining their results into a single answer.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"sakana","model_canonical_name":"fugu-ultra"},{"api":"chat","id":"nebius/zai-org/glm-5.1","object":"model","created":1775578025,"updated":1787843415,"owned_by":"system","input_price":0.0000014,"cached_price":0.0000014,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"cached_price":0.0000014,"output_price":0.0000044}],"max_output_tokens":0,"context_window":200000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Zhipu AI flagship multimodal model with strong bilingual Chinese English reasoning, long context understanding, advanced tool use, and agent oriented capabilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.1"},{"api":"chat","id":"nebius/qwen/qwen3-32b","object":"model","created":1745875945,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":1e-7,"output_price":3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":1e-7,"output_price":3e-7}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Generalist model offering strong multilingual reasoning, coding, and long-context performance at mid scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-32b"},{"api":"chat","id":"nebius/nousresearch/hermes-4-405b","object":"model","created":1756235463,"updated":1787843415,"owned_by":"system","input_price":0.000001,"cached_price":0.000001,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"cached_price":0.000001,"output_price":0.000003}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Hybrid reasoning model trained on verified CoT traces for strong math, coding, and step by step reliability.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp8","open_weights":true,"model_lab":"nousresearch","model_canonical_name":"hermes-4-405b"},{"api":"chat","id":"nebius/nousresearch/hermes-4-70b","object":"model","created":1756236182,"updated":1787843415,"owned_by":"system","input_price":1.3e-7,"cached_price":1.3e-7,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.3e-7,"cached_price":1.3e-7,"output_price":4e-7}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Compact version of Hermes-4 delivering high-quality reasoning and coding with lower inference cost.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"nousresearch","model_canonical_name":"hermes-4-70b"},{"api":"chat","id":"nebius/deepseek-v4-pro-0424","object":"model","created":1777000679,"updated":1787843415,"owned_by":"system","input_price":0.00000175,"cached_price":0.00000175,"output_price":0.0000035,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":0.00000175,"output_price":0.0000035}],"max_output_tokens":128000,"context_window":164000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek-V4 is designed for advanced reasoning, coding, and long-horizon agent workflows, with strong performance across knowledge, math, and software engineering benchmarks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"uk","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro-0424"},{"api":"chat","id":"nebius/nvidia/nemotron-3-nano-omni","object":"model","created":1779285239,"updated":1787843415,"owned_by":"system","input_price":6e-8,"cached_price":6e-8,"output_price":2.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-8,"cached_price":6e-8,"output_price":2.4e-7}],"max_output_tokens":0,"context_window":300000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The most open, efficient, and accurate omni modal reasoning model for agentic AI.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3-nano-omni"},{"api":"chat","id":"nebius/nvidia/nemotron-3-ultra-550b-a55b","object":"model","created":1780551208,"updated":1787843415,"owned_by":"system","input_price":0.000001,"cached_price":0.000001,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"cached_price":0.000001,"output_price":0.000003}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Nemotron 3 Ultra is a 550B hybrid MoE model from NVIDIA, optimized for the most demanding multi-agent AI and complex reasoning tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3-ultra-550b-a55b"},{"api":"chat","id":"nebius/google/gemma-3-27b-it","object":"model","created":1741756359,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":1e-7,"output_price":3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":1e-7,"output_price":3e-7}],"max_output_tokens":8192,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google mid-size model optimized for high-quality instruction following, coding, and multilingual performance.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp8","open_weights":true,"model_lab":"google","model_canonical_name":"gemma-3-27b-it"},{"api":"chat","id":"nebius/glm-5.2","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":0.0000014,"cached_price":0,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"cached_price":0,"output_price":0.0000044}],"max_output_tokens":131072,"context_window":1000000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-5.2 is Zhipu AI's latest flagship model with strong bilingual (Chinese-English) reasoning, long-context understanding, advanced tool use, and agent-oriented capabilities. The latest model in the GLM series, designed to plan, execute, and iterate autonomously on extended, engineering-grade tasks. Hosted on Nebius at fp8 quantization.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp4","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"nebius/qwen/qwen3-next-80b-a3b-thinking","object":"model","created":1757612284,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":1.5e-7,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":1.5e-7,"output_price":0.0000012}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen thinking-optimized 80B model designed for sustained multi-step reasoning, structured deliberation, and high-precision problem-solving across math, code, and complex planning tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-next-80b-a3b-thinking"},{"api":"chat","id":"nebius/meta-llama/Llama-3.3-70B-Instruct","object":"model","created":1733506137,"updated":1787843415,"owned_by":"system","input_price":1.3e-7,"cached_price":1.3e-7,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.3e-7,"cached_price":1.3e-7,"output_price":4e-7}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"eu","quantization":"fp8","open_weights":true,"model_lab":"meta","model_canonical_name":"llama-3.3-70b-instruct"},{"api":"chat","id":"nebius/openai/gpt-oss-120b","object":"model","created":1754414231,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":1.5e-7,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":1.5e-7,"output_price":6e-7}],"max_output_tokens":128000,"context_window":131000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Open-weight agentic model with configurable reasoning, full CoT visibility, strong tool use, and fine-tuning support.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"eu","quantization":"fp4","open_weights":true,"model_lab":"openai","model_canonical_name":"gpt-oss-120b"},{"api":"chat","id":"nebius/moonshotai/kimi-k2.6","object":"model","created":1776699402,"updated":1787843415,"owned_by":"system","input_price":9.5e-7,"cached_price":9.5e-7,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":9.5e-7,"cached_price":9.5e-7,"output_price":0.000004}],"max_output_tokens":128000,"context_window":256000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.6 is an open-source, native multimodal agentic model built through continual pretraining on approximately 15 trillion mixed visual and text tokens atop Kimi-K2-Base.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"int4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.6"},{"api":"chat","id":"nebius/minimaxi/minimax-m2.5","object":"model","created":1770908502,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":3e-7,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":3e-7,"output_price":0.0000012}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Open-source agentic coding model built for polyglot development and precision refactoring, using interleaved-thinking tool calls to reliably execute long, multi-step coding and office workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.5"},{"api":"chat","id":"nebius/qwen/qwen3.5-397b-a17b","object":"model","created":1771223018,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":6e-7,"output_price":0.0000036,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":6e-7,"output_price":0.0000036}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Multimodal model featuring a Hybrid Mixture-of-Experts architecture, designed for state-of-the-art performance across chat, retrieval-augmented generation, vision-language understanding, video understanding, and agentic workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.5-397b-a17b"},{"api":"chat","id":"nebius/qwen/qwen3-235b-a22b-instruct-2507","object":"model","created":1753119555,"updated":1787843415,"owned_by":"system","input_price":2e-7,"cached_price":2e-7,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":2e-7,"output_price":6e-7}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Balanced Qwen3 flagship tuned for strong general reasoning, chat quality, and tool use.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-235b-a22b-instruct-2507"},{"api":"chat","id":"nebius/kimi-k3","object":"model","created":1784215858,"updated":1787843415,"owned_by":"system","input_price":0.000003,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"output_price":0.000015}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Kimi K3 is Moonshot AI's flagship reasoning model with a 1M token context window, strong agentic tool use, and long horizon task execution. Hosted on Nebius.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k3"},{"api":"chat","id":"nebius/qwen/qwen3-30b-a3b-instruct-2507","object":"model","created":1753806965,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":1e-7,"output_price":3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":1e-7,"output_price":3e-7}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Versatile 30B instruct model optimized for high-quality chat, reasoning, and coding.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-30b-a3b-instruct-2507"},{"api":"chat","id":"typesafe/jev-1.13.0","object":"model","created":1789746304,"updated":1789746304,"owned_by":"system","input_price":4.2e-8,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.2e-8,"output_price":0}],"max_output_tokens":1024,"context_window":64000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Jev is TypeSafe’s flagship model and the first System One model.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Data retention period is not specified","geolocation":"global","open_weights":false,"model_lab":"typesafe","model_canonical_name":"jev"},{"api":"chat","id":"deepseek/deepseek-v4.1-flash","object":"model","created":1789010266,"updated":1789028588,"owned_by":"system","input_price":3e-7,"cached_price":6e-9,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":6e-9,"output_price":0.0000012}],"max_output_tokens":384000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek V4.1 Flash is a 552B parameter Mixture of Experts model built on a new Causal Encoder Decoder architecture that activates only 8B parameters for input and 16B for output, giving frontier level intelligence at a fraction of the cost. It is the smallest model in the new DeepSeek architecture family and the first Flash with native vision and multimodal understanding. V4.1 Flash beats DeepSeek V4 Pro on performance, cost, speed and task completion time and scores 88.1 on CyberGym. Its KV cache is compressed to 890 bytes per token, four times smaller than V4 Flash and over 400 times smaller than V1, cutting the cache hit costs that dominate agent workloads. 1M token context window, 384K max output, thinking and non thinking modes, tool calling, JSON output and prompt caching. Pricing per 1M tokens starts at $0.003 for cache hits, $0.15 input and $0.60 output off peak, with peak rates at double (01:00 to 04:00 and 06:00 to 10:00 UTC, Monday to Friday). Open weights are published on Hugging Face. Ideal for coding agents, long context RAG, high throughput batch processing and multimodal assistants. Replaces deepseek-v4-flash and deepseek-v4-flash-vision-exp, both still accepted as aliases.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No online information","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4.1-flash"},{"api":"chat","id":"deepseek/deepseek-reasoner","object":"model","created":1737381095,"updated":1787843415,"owned_by":"system","input_price":1.4e-7,"cached_price":2.8e-8,"output_price":2.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":2.8e-8,"output_price":2.8e-7}],"max_output_tokens":384000,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"An alias for DeepSeek v4 Flash in thinking mode.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No online information","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-reasoner","retires":1784851200},{"api":"chat","id":"deepseek/deepseek-chat","object":"model","created":1735241320,"updated":1787843415,"owned_by":"system","input_price":1.4e-7,"cached_price":2.8e-8,"output_price":2.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":2.8e-8,"output_price":2.8e-7}],"max_output_tokens":384000,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"An alias for DeepSeek v4 Flash in non-thinking mode.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No online information","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-chat","retires":1784851200},{"api":"chat","id":"deepseek/deepseek-v4-pro-0813","object":"model","created":1777000679,"updated":1787843415,"owned_by":"system","input_price":0.00000132,"cached_price":4.4e-8,"output_price":0.00000396,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000132,"cached_price":4.4e-8,"output_price":0.00000396}],"max_output_tokens":384000,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek v4 Pro - 1.6T total / 49B active params. Performance rivaling the world's top closed-source models.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No online information","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro-0813"},{"api":"chat","id":"deepinfra/zai-org/GLM-4.5-Air","object":"model","created":1753471258,"updated":1787843415,"owned_by":"system","input_price":2e-7,"cached_price":2e-7,"output_price":0.0000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":2e-7,"output_price":0.0000011}],"max_output_tokens":4096,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The GLM-4.5 series models are foundation models designed for intelligent agents. GLM-4.5 has 355 billion total parameters with 32 billion active parameters, while GLM-4.5-Air adopts a more compact design with 106 billion total parameters and 12 billion active parameters. GLM-4.5 models unify reasoning, coding, and intelligent agent capabilities to meet the complex demands of intelligent agent applications.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-4.5-air"},{"api":"chat","id":"deepinfra/deepseek-ai/DeepSeek-V3:flex","object":"model","created":1735241320,"updated":1787843415,"owned_by":"system","input_price":2.56e-7,"output_price":7.12e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.56e-7,"output_price":7.12e-7}],"max_output_tokens":8192,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek-V3, a strong Mixture-of-Experts (MoE) language model with 671B total parameters with 37B activated for each token. To achieve efficient inference and cost-effective training, DeepSeek-V3 adopts Multi-head Latent Attention (MLA) and DeepSeekMoE architectures.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v3"},{"api":"chat","id":"deepinfra/deepseek-v4.1-flash","object":"model","created":1788912000,"updated":1789221770,"owned_by":"system","input_price":2e-7,"cached_price":6e-9,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":6e-9,"output_price":6e-7}],"max_output_tokens":0,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4.1 Flash is a 552B parameter Mixture of Experts model built on a new Causal Encoder Decoder architecture that activates only 8B parameters for input and 16B for output, giving frontier level intelligence at a fraction of the cost. It is the smallest model in the new DeepSeek architecture family and the first Flash with native vision and multimodal understanding. V4.1 Flash beats DeepSeek V4 Pro on performance, cost, speed and task completion time and scores 88.1 on CyberGym. Its KV cache is compressed to 890 bytes per token, four times smaller than V4 Flash and over 400 times smaller than V1, cutting the cache hit costs that dominate agent workloads. 1M token context window, 384K max output, thinking and non thinking modes, tool calling, structured JSON output and prompt caching. Served on DeepInfra in FP8 at $0.20 input, $0.006 cache read and $0.60 output per 1M tokens, with a flex tier at 0.8x. Open weights are published on Hugging Face. Ideal for coding agents, long context RAG, high throughput batch processing and multimodal assistants.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4.1-flash"},{"api":"chat","id":"deepinfra/moonshotai/Kimi-K2.6","object":"model","created":1776699402,"updated":1787843415,"owned_by":"system","input_price":7.5e-7,"cached_price":1.5e-7,"output_price":0.0000035,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.5e-7,"cached_price":1.5e-7,"output_price":0.0000035}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.6 is an open-source, native multimodal agentic model that advances practical capabilities in long-horizon coding, coding-driven design, proactive autonomous execution, and swarm-based task orchestration.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.6"},{"api":"chat","id":"deepinfra/deepseek-v4-flash-0424:flex","object":"model","created":1777000666,"updated":1787843415,"owned_by":"system","input_price":7.2e-8,"cached_price":1.44e-8,"output_price":1.44e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.2e-8,"cached_price":1.44e-8,"output_price":1.44e-7}],"max_output_tokens":0,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash is an efficiency-focused MoE model with 284B total parameters (13B active) and a 1M-token context window. It's tuned for fast inference and high-throughput use cases while still holding up on reasoning and coding tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0424"},{"api":"chat","id":"deepinfra/nvidia/Nemotron-3-Nano-30B-A3B:flex","object":"model","created":1765731275,"updated":1787843415,"owned_by":"system","input_price":4e-8,"cached_price":2e-8,"output_price":1.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-8,"cached_price":2e-8,"output_price":1.6e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3 Nano is an open small reasoning model optimized for fast, cost-efficient inference in agentic and production workloads. Built with a hybrid Mixture-of-Experts (MoE) and Mamba-Transformer architecture, it delivers strong multi-step reasoning, high token throughput, stable latency with predictable cost, and efficient deployment for agent-based systems. Designed for real-world AI systems where reasoning can generate significantly more tokens per prompt, Nemotron Nano reduces compu","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3-nano-30b-a3b"},{"api":"chat","id":"deepinfra/glm-5.2","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":7.5e-7,"cached_price":1.4e-7,"output_price":0.0000024,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.5e-7,"cached_price":1.4e-7,"output_price":0.0000024}],"max_output_tokens":128000,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.2 is Z-AI's latest flagship model for long-horizon tasks. It marks a substantial leap in long-horizon task capability over its predecessor GLM-5.1 and, for the first time, delivers that capability on a **solid 1M-token context**.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepinfra","model_canonical_name":"glm-5.2"},{"api":"chat","id":"deepinfra/microsoft/phi-4","object":"model","created":1736489872,"updated":1787843415,"owned_by":"system","input_price":7e-8,"cached_price":7e-8,"output_price":1.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":7e-8,"cached_price":7e-8,"output_price":1.4e-7}],"max_output_tokens":0,"context_window":16384,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Phi-4-reasoning-plus is an enhanced 14B parameter model from Microsoft, fine-tuned from Phi-4 with additional reinforcement learning to boost accuracy on math, science, and code reasoning tasks. It uses the same dense decoder-only transformer architecture as Phi-4, but generates longer, more comprehensive outputs structured into a step-by-step reasoning trace and final answer.\n\nWhile it offers improved benchmark scores over Phi-4-reasoning across tasks like AIME, OmniMath, and HumanEvalPlus, its responses are typically ~50% longer, resulting in higher latency. Designed for English-only applications, it is well-suited for structured reasoning workflows where output quality takes priority over response speed.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"microsoft","model_canonical_name":"phi-4"},{"api":"chat","id":"deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo:flex","object":"model","created":1733506137,"updated":1787843415,"owned_by":"system","input_price":8e-8,"output_price":2.56e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":8e-8,"output_price":2.56e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"meta","model_canonical_name":"llama-3.3-70b-instruct-turbo"},{"api":"chat","id":"deepinfra/XiaomiMiMo/MiMo-V2.5-Pro","object":"model","created":1776874273,"updated":1790520685,"owned_by":"system","input_price":0.000001,"cached_price":2e-7,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"cached_price":2e-7,"output_price":0.000003}],"max_output_tokens":0,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. It utilizes the hybrid attention architecture and 3-layers Multi-Token Prediction (MTP) introduced in [MiMo-V2-Flash](https://github.com/XiaomiMiMo/MiMo-V2-Flash).","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.5-pro","retires":1790696493},{"api":"chat","id":"deepinfra/deepseek-v4-pro-0424","object":"model","created":1777000679,"updated":1787843415,"owned_by":"system","input_price":0.0000013,"cached_price":1e-7,"output_price":0.0000026,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000013,"cached_price":1e-7,"output_price":0.0000026}],"max_output_tokens":0,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Pro is an MoE model with 1.6T total parameters (49B active) and a 1M-token context window. It's built for advanced reasoning, coding, and long-running agent tasks, and performs well on knowledge, math, and software engineering benchmarks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro-0424"},{"api":"chat","id":"deepinfra/deepseek-v4-flash-0731:flex","object":"model","created":1785478908,"updated":1787843415,"owned_by":"system","input_price":7.2e-8,"cached_price":1.44e-8,"output_price":1.44e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.2e-8,"cached_price":1.44e-8,"output_price":1.44e-7}],"max_output_tokens":0,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash 0731 is the official release of DeepSeek V4 Flash, superseding the preview version, with substantially enhanced agentic capabilities. A 304B parameter open-source MoE model with a 1M token context window, tuned for fast inference and high-throughput use cases with strong reasoning, coding, and function calling performance.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0731"},{"api":"chat","id":"deepinfra/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B","object":"model","created":1773245239,"updated":1787843415,"owned_by":"system","input_price":1e-7,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"output_price":5e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3 Super is a hybrid Mixture-of-Experts (MoE) model engineered for highest compute efficiency and accuracy in multi-agent applications and specialized agentic systems. It is optimized to run many collaborating agents per application on a single GPU, delivering high accuracy for reasoning, tool use, and instruction following.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nvidia-nemotron-3-super-120b-a12b"},{"api":"chat","id":"deepinfra/Qwen/Qwen3.5-35B-A3B","object":"model","created":1772053822,"updated":1787843415,"owned_by":"system","input_price":1.4e-7,"cached_price":5e-8,"output_price":0.000001,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":5e-8,"output_price":0.000001}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.5-35B-A3B is an efficient Mixture-of-Experts model from Alibaba's Qwen3.5 series with 35B total parameters and only 3B activated per token. It features a 262K token context window (extensible to 1M with YaRN), thinking/reasoning mode, tool calling, and support for 201 languages. Delivers strong performance on reasoning, coding, and vision-language tasks at a fraction of the compute cost.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.5-35b-a3b"},{"api":"chat","id":"deepinfra/deepseek-ai/DeepSeek-V3","object":"model","created":1735241320,"updated":1787843415,"owned_by":"system","input_price":8.5e-7,"cached_price":8.5e-7,"output_price":9e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":8.5e-7,"cached_price":8.5e-7,"output_price":9e-7}],"max_output_tokens":8192,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek-V3, a strong Mixture-of-Experts (MoE) language model with 671B total parameters with 37B activated for each token. To achieve efficient inference and cost-effective training, DeepSeek-V3 adopts Multi-head Latent Attention (MLA) and DeepSeekMoE architectures.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v3"},{"api":"chat","id":"deepinfra/google/gemma-4-31B-it","object":"model","created":1775148486,"updated":1787843415,"owned_by":"system","input_price":1.3e-7,"output_price":3.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.3e-7,"output_price":3.8e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemma is a family of open models built by Google DeepMind. Gemma 4 models are multimodal, handling text and image input and generating text output.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"google","model_canonical_name":"gemma-4-31b-it"},{"api":"chat","id":"deepinfra/Qwen/Qwen2.5-72B-Instruct","object":"model","created":1726704000,"updated":1787843415,"owned_by":"system","input_price":2.3e-7,"cached_price":2.3e-7,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.3e-7,"cached_price":2.3e-7,"output_price":4e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen2.5-72b-instruct"},{"api":"chat","id":"deepinfra/Qwen/Qwen3.5-397B-A17B","object":"model","created":1771223018,"updated":1787843415,"owned_by":"system","input_price":4.9e-7,"cached_price":3e-7,"output_price":0.0000036,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.9e-7,"cached_price":3e-7,"output_price":0.0000036}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.5-397B-A17B is Alibaba's most capable Qwen3.5 model, a Mixture-of-Experts architecture with 397B total parameters and 17B activated per token. It features a 262K token context window (extensible to 1M with YaRN), thinking/reasoning mode, tool calling with MCP integration, and support for 201 languages. Sets state-of-the-art results on reasoning, coding, math, and multimodal benchmarks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.5-397b-a17b"},{"api":"chat","id":"deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo","object":"model","created":1721692800,"updated":1787843415,"owned_by":"system","input_price":2e-8,"cached_price":2e-8,"output_price":5e-8,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-8,"cached_price":2e-8,"output_price":5e-8}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"meta","model_canonical_name":"meta-llama-3.1-8b-instruct-turbo"},{"api":"chat","id":"deepinfra/ByteDance/Seed-1.8","object":"model","created":1779873100,"updated":1787843415,"owned_by":"system","input_price":2.5e-7,"cached_price":5e-8,"output_price":0.000002,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"cached_price":5e-8,"output_price":0.000002}],"max_output_tokens":0,"context_window":256000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Optimized specifically for multimodal agent scenarios. It features enhanced agent capabilities, upgraded multimodal comprehension, and more flexible context management.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"bytedance","model_canonical_name":"seed-1.8"},{"api":"chat","id":"deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo","object":"model","created":1733506137,"updated":1787843415,"owned_by":"system","input_price":1.2e-7,"cached_price":1.2e-7,"output_price":3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.2e-7,"cached_price":1.2e-7,"output_price":3e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"meta","model_canonical_name":"llama-3.3-70b-instruct-turbo"},{"api":"chat","id":"deepinfra/deepseek-v4-flash-0731","object":"model","created":1785478908,"updated":1787843415,"owned_by":"system","input_price":9e-8,"cached_price":1.8e-8,"output_price":1.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":9e-8,"cached_price":1.8e-8,"output_price":1.8e-7}],"max_output_tokens":0,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash 0731 is the official release of DeepSeek V4 Flash, superseding the preview version, with substantially enhanced agentic capabilities. A 304B parameter open-source MoE model with a 1M token context window, tuned for fast inference and high-throughput use cases with strong reasoning, coding, and function calling performance.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0731"},{"api":"chat","id":"deepinfra/Qwen/Qwen3.5-35B-A3B:flex","object":"model","created":1772053822,"updated":1787843415,"owned_by":"system","input_price":1.12e-7,"cached_price":4e-8,"output_price":8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.12e-7,"cached_price":4e-8,"output_price":8e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.5-35B-A3B is an efficient Mixture-of-Experts model from Alibaba's Qwen3.5 series with 35B total parameters and only 3B activated per token. It features a 262K token context window (extensible to 1M with YaRN), thinking/reasoning mode, tool calling, and support for 201 languages. Delivers strong performance on reasoning, coding, and vision-language tasks at a fraction of the compute cost.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.5-35b-a3b"},{"api":"chat","id":"deepinfra/deepseek-v4.1-flash:flex","object":"model","created":1788912000,"updated":1789221770,"owned_by":"system","input_price":1.6e-7,"cached_price":4.8e-9,"output_price":4.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.6e-7,"cached_price":4.8e-9,"output_price":4.8e-7}],"max_output_tokens":0,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4.1 Flash is a 552B parameter Mixture of Experts model built on a new Causal Encoder Decoder architecture that activates only 8B parameters for input and 16B for output, giving frontier level intelligence at a fraction of the cost. It is the smallest model in the new DeepSeek architecture family and the first Flash with native vision and multimodal understanding. V4.1 Flash beats DeepSeek V4 Pro on performance, cost, speed and task completion time and scores 88.1 on CyberGym. Its KV cache is compressed to 890 bytes per token, four times smaller than V4 Flash and over 400 times smaller than V1, cutting the cache hit costs that dominate agent workloads. 1M token context window, 384K max output, thinking and non thinking modes, tool calling, structured JSON output and prompt caching. Served on DeepInfra in FP8 at $0.20 input, $0.006 cache read and $0.60 output per 1M tokens, with a flex tier at 0.8x. Open weights are published on Hugging Face. Ideal for coding agents, long context RAG, high throughput batch processing and multimodal assistants.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4.1-flash"},{"api":"chat","id":"deepinfra/Qwen/Qwen3.5-2B","object":"model","created":1773152396,"updated":1787843415,"owned_by":"system","input_price":2e-8,"output_price":1e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-8,"output_price":1e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.5-2B is a compact yet capable model from Alibaba's Qwen3.5 series. It features a 262K token context window, support for 201 languages, thinking/reasoning mode, and tool calling for agentic workflows. A strong choice for prototyping, fine-tuning, and efficient multilingual deployments.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.5-2b"},{"api":"chat","id":"deepinfra/meta-llama/Llama-3.2-90B-Vision-Instruct","object":"model","created":1727222400,"updated":1787843415,"owned_by":"system","input_price":3.5e-7,"cached_price":3.5e-7,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.5e-7,"cached_price":3.5e-7,"output_price":4e-7}],"max_output_tokens":4096,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"meta","model_canonical_name":"llama-3.2-90b-vision-instruct"},{"api":"chat","id":"deepinfra/meta-llama/Llama-3.3-70B-Instruct","object":"model","created":1733506137,"updated":1787843415,"owned_by":"system","input_price":2.3e-7,"cached_price":2.3e-7,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.3e-7,"cached_price":2.3e-7,"output_price":4e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"meta","model_canonical_name":"llama-3.3-70b-instruct"},{"api":"chat","id":"deepinfra/Qwen/Qwen3.5-397B-A17B:flex","object":"model","created":1771223018,"updated":1787843415,"owned_by":"system","input_price":3.6e-7,"cached_price":1.76e-7,"output_price":0.0000024,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.6e-7,"cached_price":1.76e-7,"output_price":0.0000024}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.5-397B-A17B is Alibaba's most capable Qwen3.5 model, a Mixture-of-Experts architecture with 397B total parameters and 17B activated per token. It features a 262K token context window (extensible to 1M with YaRN), thinking/reasoning mode, tool calling with MCP integration, and support for 201 languages. Sets state-of-the-art results on reasoning, coding, math, and multimodal benchmarks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.5-397b-a17b"},{"api":"chat","id":"deepinfra/moonshotai/Kimi-K2.6:flex","object":"model","created":1776699402,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":1.2e-7,"output_price":0.0000028,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":1.2e-7,"output_price":0.0000028}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.6 is an open-source, native multimodal agentic model that advances practical capabilities in long-horizon coding, coding-driven design, proactive autonomous execution, and swarm-based task orchestration.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.6"},{"api":"chat","id":"deepinfra/mimo-v2.6-flash","object":"model","created":1790520673,"updated":1790520673,"owned_by":"system","input_price":1.4e-7,"cached_price":2.7e-9,"output_price":2.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":2.7e-9,"output_price":2.8e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiMo-V2.6-Flash is the efficiency-balanced model in Xiaomi's MiMo-V2.6 series: a sparse Mixture-of-Experts model with 309B total and 15B activated parameters, natively omnimodal across text, image, video and audio, with a 1M-token context window. Trained with large-scale mixed reinforcement learning across coding, general agents, visual and cybersecurity tasks, it is built for agentic workloads such as long-horizon coding, tool use and computer use.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.6-flash"},{"api":"chat","id":"deepinfra/glm-5.2:flex","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":1.12e-7,"output_price":0.00000192,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":1.12e-7,"output_price":0.00000192}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.2 is Z-AI's latest flagship model for long-horizon tasks. It marks a substantial leap in long-horizon task capability over its predecessor GLM-5.1 and, for the first time, delivers that capability on a **solid 1M-token context**.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepinfra","model_canonical_name":"glm-5.2"},{"api":"chat","id":"deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct","object":"model","created":1721692800,"updated":1787843415,"owned_by":"system","input_price":2.3e-7,"cached_price":2.3e-7,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.3e-7,"cached_price":2.3e-7,"output_price":4e-7}],"max_output_tokens":0,"context_window":130815,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"meta","model_canonical_name":"meta-llama-3.1-70b-instruct"},{"api":"chat","id":"deepinfra/Qwen/Qwen3-235B-A22B","object":"model","created":1745875757,"updated":1787843415,"owned_by":"system","input_price":2e-7,"cached_price":2e-7,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":2e-7,"output_price":6e-7}],"max_output_tokens":4096,"context_window":40960,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-235b-a22b"},{"api":"chat","id":"deepinfra/qwen3.8","object":"model","created":1786551702,"updated":1787843415,"owned_by":"system","input_price":0.000002,"cached_price":2e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"cached_price":2e-7,"output_price":0.000006}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of Qwen3.8 Max, with 95 billion active parameters out of 2.4 trillion total. It is suited for coding, research, complex reasoning, and agentic workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepinfra","model_canonical_name":"qwen3.8-2.4T-A95B"},{"api":"chat","id":"deepinfra/qwen3.8-27b","object":"model","created":1786924800,"updated":1790761156,"owned_by":"system","input_price":2e-7,"cached_price":5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":5e-8,"output_price":0.0000025}],"max_output_tokens":131072,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.8-27B is an open weight dense vision language model from Qwen, built on the Qwen3.5 architecture. It understands images and video, offers flexible thinking control, and is suited for coding, professional work, research, and long horizon agentic tasks. Compact and deployment friendly with a 262K native context.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.8-27b"},{"api":"chat","id":"deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct","object":"model","created":1753230546,"updated":1787843415,"owned_by":"system","input_price":4e-7,"cached_price":4e-7,"output_price":0.0000016,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"cached_price":4e-7,"output_price":0.0000016}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3-Coder-480B-A35B-Instruct is Alibaba’s flagship open-weight AI model designed for autonomous software development and agentic workflows. Using a sparse Mixture-of-Experts (MoE) architecture, it achieves state-of-the-art coding performance that regularly rivals or outperforms proprietary models like Claude Sonnet. ","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-coder-480b-a35b-instruct"},{"api":"chat","id":"deepinfra/Qwen/Qwen3-32B:flex","object":"model","created":1745875945,"updated":1787843415,"owned_by":"system","input_price":6.4e-8,"output_price":2.24e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":6.4e-8,"output_price":2.24e-7}],"max_output_tokens":0,"context_window":40960,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-32b"},{"api":"chat","id":"deepinfra/XiaomiMiMo/MiMo-V2.5:flex","object":"model","created":1776874269,"updated":1790520685,"owned_by":"system","input_price":3.2e-7,"cached_price":6.4e-8,"output_price":0.0000016,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.2e-7,"cached_price":6.4e-8,"output_price":0.0000016}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities, supporting text, image, video, and audio understanding within a unified architecture. Built upon the MiMo-V2-Flash backbone and extended with dedicated vision and audio encoders, it delivers robust performance across multimodal perception, long-context reasoning, and agentic workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.5","retires":1790696493},{"api":"chat","id":"deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507","object":"model","created":1753119555,"updated":1787843415,"owned_by":"system","input_price":7.1e-8,"output_price":1e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.1e-8,"output_price":1e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3-235B-A22B-Instruct-2507 is the updated version of the Qwen3-235B-A22B non-thinking mode, featuring Significant improvements in general capabilities, including instruction following, logical reasoning, text comprehension, mathematics, science, coding and tool usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-235b-a22b-instruct-2507"},{"api":"chat","id":"deepinfra/XiaomiMiMo/MiMo-V2.5","object":"model","created":1776874269,"updated":1790520685,"owned_by":"system","input_price":4e-7,"cached_price":8e-8,"output_price":0.000002,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"cached_price":8e-8,"output_price":0.000002}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiMo-V2.5 is a native omnimodal model with strong agentic capabilities, supporting text, image, video, and audio understanding within a unified architecture. Built upon the MiMo-V2-Flash backbone and extended with dedicated vision and audio encoders, it delivers robust performance across multimodal perception, long-context reasoning, and agentic workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.5","retires":1790696493},{"api":"chat","id":"deepinfra/Qwen/Qwen3.5-27B","object":"model","created":1772053810,"updated":1787843415,"owned_by":"system","input_price":2.6e-7,"output_price":0.0000026,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.6e-7,"output_price":0.0000026}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.5-27B is Alibaba's largest dense Qwen3.5 model, delivering near-frontier quality across reasoning, coding, and instruction following. It features a 262K token context window (extensible to 1M), thinking/reasoning mode, tool calling, multi-token prediction, and support for 201 languages. Best suited for production deployments and complex enterprise tasks requiring top-tier performance.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.5-27b"},{"api":"chat","id":"deepinfra/zai-org/GLM-4.5","object":"model","created":1753471347,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":6e-7,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":6e-7,"output_price":0.0000022}],"max_output_tokens":4096,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"The GLM-4.5 series models are foundation models designed for intelligent agents. GLM-4.5 has 355 billion total parameters with 32 billion active parameters, while GLM-4.5-Air adopts a more compact design with 106 billion total parameters and 12 billion active parameters. GLM-4.5 models unify reasoning, coding, and intelligent agent capabilities to meet the complex demands of intelligent agent applications.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-4.5"},{"api":"chat","id":"deepinfra/ByteDance/Seed-2.0-pro","object":"model","created":1773157231,"updated":1787843415,"owned_by":"system","input_price":5e-7,"cached_price":1e-7,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"cached_price":1e-7,"output_price":0.000003}],"max_output_tokens":0,"context_window":256000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Built for the Agent era, it delivers stable performance in complex reasoning and long-horizon tasks, including multi-step planning, visual-text reasoning, video understanding, and advanced analysis.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"bytedance","model_canonical_name":"seed-2.0-pro"},{"api":"chat","id":"deepinfra/ByteDance/Seed-2.0-code","object":"model","created":1773157231,"updated":1787843415,"owned_by":"system","input_price":5e-7,"cached_price":1e-7,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"cached_price":1e-7,"output_price":0.000003}],"max_output_tokens":0,"context_window":256000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"A coding model optimized for real-world development environments, with reliable tool use in common IDEs such as Claude Code. It delivers strong front-end performance and supports Skills.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"bytedance","model_canonical_name":"seed-2.0-code"},{"api":"chat","id":"deepinfra/Qwen/Qwen2.5-Coder-32B-Instruct","object":"model","created":1731368400,"updated":1787843415,"owned_by":"system","input_price":7e-8,"cached_price":7e-8,"output_price":1.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":7e-8,"cached_price":7e-8,"output_price":1.6e-7}],"max_output_tokens":40960,"context_window":16384,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen2.5-coder-32b-instruct"},{"api":"chat","id":"deepinfra/ByteDance/Seed-2.0-mini","object":"model","created":1772131107,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":2e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":2e-8,"output_price":4e-7}],"max_output_tokens":0,"context_window":256000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Built for low-latency, high-concurrency, cost-sensitive use cases, with flexible deployment, four-tier thinking, and multimodal","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"bytedance","model_canonical_name":"seed-2.0-mini"},{"api":"chat","id":"deepinfra/Qwen/Qwen3-Max","object":"model","created":1758662808,"updated":1787843415,"owned_by":"system","input_price":0.0000012,"cached_price":2.4e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000012,"cached_price":2.4e-7,"output_price":0.000006}],"max_output_tokens":0,"context_window":256000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"The latest flagship model in the Qwen family. State-of-the-art results across a comprehensive suite of benchmarks — including knowledge, reasoning, coding, instruction following, human preference alignment, agent tasks, and multilingual understanding.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3-max"},{"api":"chat","id":"deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo:flex","object":"model","created":1753230546,"updated":1787843415,"owned_by":"system","input_price":2.4e-7,"cached_price":8e-8,"output_price":8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.4e-7,"cached_price":8e-8,"output_price":8e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3-Coder-480B-A35B-Instruct is the Qwen3's most agentic code model, featuring Significant Performance on Agentic Coding, Agentic Browser-Use and other foundational coding tasks, achieving results comparable to Claude Sonnet.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-coder-480b-a35b-instruct-turbo"},{"api":"chat","id":"deepinfra/zai-org/GLM-5.1","object":"model","created":1775578025,"updated":1787843415,"owned_by":"system","input_price":0.00000105,"cached_price":2.05e-7,"output_price":0.0000035,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000105,"cached_price":2.05e-7,"output_price":0.0000035}],"max_output_tokens":0,"context_window":202752,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.1 is Z-AI's next-generation flagship model for agentic engineering, with significantly stronger coding capabilities than its predecessor. It achieves state-of-the-art performance on SWE-Bench Pro and leads GLM-5 by a wide margin on NL2Repo (repo generation) and Terminal-Bench 2.0 (real-world terminal tasks).","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.1"},{"api":"chat","id":"deepinfra/glm-5.3-flash","object":"model","created":1787702400,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.3-Flash is a native multimodal model from Z.ai, suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while reducing compute overhead. Accepts text, image and video input.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3-flash"},{"api":"chat","id":"deepinfra/zai-org/GLM-5.1:flex","object":"model","created":1775578025,"updated":1787843415,"owned_by":"system","input_price":8.4e-7,"cached_price":1.64e-7,"output_price":0.0000028,"pricing":[{"prompt_tokens_threshold":0,"input_price":8.4e-7,"cached_price":1.64e-7,"output_price":0.0000028}],"max_output_tokens":0,"context_window":202752,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.1 is Z-AI's next-generation flagship model for agentic engineering, with significantly stronger coding capabilities than its predecessor. It achieves state-of-the-art performance on SWE-Bench Pro and leads GLM-5 by a wide margin on NL2Repo (repo generation) and Terminal-Bench 2.0 (real-world terminal tasks).","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.1"},{"api":"chat","id":"deepinfra/moonshotai/Kimi-K2.5","object":"model","created":1774953861,"updated":1787843415,"owned_by":"system","input_price":4.5e-7,"cached_price":7e-8,"output_price":0.00000225,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.5e-7,"cached_price":7e-8,"output_price":0.00000225}],"max_output_tokens":131072,"context_window":262100,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.5 is an open-source, native multimodal agentic model built through continual pretraining on approximately 15 trillion mixed visual and text tokens atop Kimi-K2-Base. It seamlessly integrates vision and language understanding with advanced agentic capabilities, instant and thinking modes, as well as conversational and agentic paradigms.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi k2.5"},{"api":"chat","id":"deepinfra/deepseek-ai/DeepSeek-V3.1","object":"model","created":1755779628,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":3e-7,"output_price":0.000001,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":3e-7,"output_price":0.000001}],"max_output_tokens":0,"context_window":163840,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek-V3.1 by DeepSeek-AI is a state-of-the-art, hybrid-inference artificial intelligence model that allows users to toggle between fast conversational responses and deep logical reasoning modes within a single architecture.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v3.1"},{"api":"chat","id":"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B","object":"model","created":1737663169,"updated":1787843415,"owned_by":"system","input_price":2.3e-7,"cached_price":2.3e-7,"output_price":6.9e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.3e-7,"cached_price":2.3e-7,"output_price":6.9e-7}],"max_output_tokens":8192,"context_window":64000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek-R1-Distill-Llama-70B is an open-weights, 70-billion parameter language model developed by DeepSeek. It distills the advanced mathematical, coding, and logical reasoning patterns of the flagship DeepSeek-R1 model into the highly optimized Llama-3.3-70B-Instruct architecture.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-r1-distill-llama-70b"},{"api":"chat","id":"deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507:flex","object":"model","created":1753449557,"updated":1787843415,"owned_by":"system","input_price":1.84e-7,"cached_price":1.6e-7,"output_price":0.00000184,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.84e-7,"cached_price":1.6e-7,"output_price":0.00000184}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3-235B-A22B-Thinking-2507 is the Qwen3's new model with scaling the thinking capability of Qwen3-235B-A22B, improving both the quality and depth of reasoning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-235b-a22b-thinking-2507"},{"api":"chat","id":"deepinfra/XiaomiMiMo/MiMo-V2.5-Pro:flex","object":"model","created":1776874273,"updated":1790520685,"owned_by":"system","input_price":8e-7,"cached_price":1.6e-7,"output_price":0.0000024,"pricing":[{"prompt_tokens_threshold":0,"input_price":8e-7,"cached_price":1.6e-7,"output_price":0.0000024}],"max_output_tokens":0,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiMo-V2.5-Pro is an open-source Mixture-of-Experts (MoE) language model with 1.02T total parameters and 42B active parameters. It utilizes the hybrid attention architecture and 3-layers Multi-Token Prediction (MTP) introduced in [MiMo-V2-Flash](https://github.com/XiaomiMiMo/MiMo-V2-Flash).","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.5-pro","retires":1790696493},{"api":"chat","id":"deepinfra/google/gemma-4-26B-A4B-it:flex","object":"model","created":1775227989,"updated":1787843415,"owned_by":"system","input_price":5.6e-8,"output_price":2.72e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5.6e-8,"output_price":2.72e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemma 4 26B A4B is a Mixture of Experts model from Google DeepMind built for scalable performance with a 256K context window. Supports text and image input, native thinking mode, function calling, and 140+ languages. Ideal for long context RAG, agentic workflows, coding, and multimodal understanding.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"google","model_canonical_name":"gemma-4-26b-a4b-it"},{"api":"chat","id":"deepinfra/deepseek-v4-pro-0424:flex","object":"model","created":1777000679,"updated":1787843415,"owned_by":"system","input_price":0.00000104,"cached_price":8e-8,"output_price":0.00000208,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000104,"cached_price":8e-8,"output_price":0.00000208}],"max_output_tokens":0,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Pro is an MoE model with 1.6T total parameters (49B active) and a 1M-token context window. It's built for advanced reasoning, coding, and long-running agent tasks, and performs well on knowledge, math, and software engineering benchmarks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro-0424"},{"api":"chat","id":"deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo","object":"model","created":1753230546,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":1e-7,"output_price":0.000001,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":1e-7,"output_price":0.000001}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3-Coder-480B-A35B-Instruct is the Qwen3's most agentic code model, featuring Significant Performance on Agentic Coding, Agentic Browser-Use and other foundational coding tasks, achieving results comparable to Claude Sonnet.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-coder-480b-a35b-instruct-turbo"},{"api":"chat","id":"deepinfra/meta-llama/Meta-Llama-3.1-405B-Instruct","object":"model","created":1721692800,"updated":1787843415,"owned_by":"system","input_price":8e-7,"cached_price":8e-7,"output_price":8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":8e-7,"cached_price":8e-7,"output_price":8e-7}],"max_output_tokens":0,"context_window":130815,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"meta","model_canonical_name":"meta-llama-3.1-405b-instruct"},{"api":"chat","id":"deepinfra/Qwen/Qwen3.5-27B:flex","object":"model","created":1772053810,"updated":1787843415,"owned_by":"system","input_price":2.08e-7,"output_price":0.00000208,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.08e-7,"output_price":0.00000208}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.5-27B is Alibaba's largest dense Qwen3.5 model, delivering near-frontier quality across reasoning, coding, and instruction following. It features a 262K token context window (extensible to 1M), thinking/reasoning mode, tool calling, multi-token prediction, and support for 201 languages. Best suited for production deployments and complex enterprise tasks requiring top-tier performance.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.5-27b"},{"api":"chat","id":"deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507","object":"model","created":1753449557,"updated":1787843415,"owned_by":"system","input_price":2.3e-7,"cached_price":2e-7,"output_price":0.0000023,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.3e-7,"cached_price":2e-7,"output_price":0.0000023}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3-235B-A22B-Thinking-2507 is the Qwen3's new model with scaling the thinking capability of Qwen3-235B-A22B, improving both the quality and depth of reasoning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-235b-a22b-thinking-2507"},{"api":"chat","id":"deepinfra/deepseek-ai/DeepSeek-V3.1:flex","object":"model","created":1755779628,"updated":1787843415,"owned_by":"system","input_price":2e-7,"cached_price":1.04e-7,"output_price":7.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":1.04e-7,"output_price":7.6e-7}],"max_output_tokens":0,"context_window":163840,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek-V3.1 by DeepSeek-AI is a state-of-the-art, hybrid-inference artificial intelligence model that allows users to toggle between fast conversational responses and deep logical reasoning modes within a single architecture.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v3.1"},{"api":"chat","id":"deepinfra/deepseek-ai/DeepSeek-R1","object":"model","created":1737381095,"updated":1787843415,"owned_by":"system","input_price":8.5e-7,"cached_price":8.5e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":8.5e-7,"cached_price":8.5e-7,"output_price":0.0000025}],"max_output_tokens":8192,"context_window":64000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek-R1 is an advanced, open-weights reasoning model by DeepSeek. Designed to compete with flagship reasoning models like OpenAI's o1, it focuses heavily on complex problem-solving in mathematics, coding, and science by \"thinking\" before it speaks, natively outputting a detailed, step-by-step chain of thought.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-r1"},{"api":"chat","id":"deepinfra/Qwen/Qwen3-32B","object":"model","created":1745875945,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":1e-7,"output_price":3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":1e-7,"output_price":3e-7}],"max_output_tokens":0,"context_window":40960,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-32b"},{"api":"chat","id":"deepinfra/mimo-v2.6-pro","object":"model","created":1790520673,"updated":1790520673,"owned_by":"system","input_price":4.3e-7,"cached_price":3.6e-9,"output_price":8.7e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.3e-7,"cached_price":3.6e-9,"output_price":8.7e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiMo-V2.6-Pro is the flagship of Xiaomi's MiMo-V2.6 series: a sparse Mixture-of-Experts model with 1.02T total and 42B activated parameters, natively omnimodal across text, image, video and audio, with a 1M-token context window. Trained with large-scale mixed reinforcement learning and groupwise agentic grading, it is built for demanding agentic work including long-horizon coding, multi-tool agents and cybersecurity tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.6-pro"},{"api":"chat","id":"deepinfra/microsoft/phi-4:flex","object":"model","created":1736489872,"updated":1787843415,"owned_by":"system","input_price":5.6e-8,"output_price":1.12e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5.6e-8,"output_price":1.12e-7}],"max_output_tokens":0,"context_window":16384,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Phi-4-reasoning-plus is an enhanced 14B parameter model from Microsoft, fine-tuned from Phi-4 with additional reinforcement learning to boost accuracy on math, science, and code reasoning tasks. It uses the same dense decoder-only transformer architecture as Phi-4, but generates longer, more comprehensive outputs structured into a step-by-step reasoning trace and final answer.\n\nWhile it offers improved benchmark scores over Phi-4-reasoning across tasks like AIME, OmniMath, and HumanEvalPlus, its responses are typically ~50% longer, resulting in higher latency. Designed for English-only applications, it is well-suited for structured reasoning workflows where output quality takes priority over response speed.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"microsoft","model_canonical_name":"phi-4"},{"api":"chat","id":"deepinfra/glm-5.3","object":"model","created":1787077232,"updated":1787941142,"owned_by":"system","input_price":0.0000012,"cached_price":1.2e-7,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000012,"cached_price":1.2e-7,"output_price":0.000004}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM 5.3 is the latest Z.ai flagship model for long horizon coding, reasoning, and agentic workflows. It builds on GLM 5.2 with a 1M token context window, stronger engineering task performance, and multiple thinking effort levels. Released under the MIT license. Served via DeepInfra.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3"},{"api":"chat","id":"deepinfra/deepseek-v4-flash-0424","object":"model","created":1777000666,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":2e-8,"output_price":2e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":2e-8,"output_price":2e-7}],"max_output_tokens":0,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash is an efficiency-focused MoE model with 284B total parameters (13B active) and a 1M-token context window. It's tuned for fast inference and high-throughput use cases while still holding up on reasoning and coding tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0424"},{"api":"chat","id":"deepinfra/nvidia/Nemotron-3-Nano-30B-A3B","object":"model","created":1765731275,"updated":1787843415,"owned_by":"system","input_price":5e-8,"output_price":2e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"output_price":2e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3 Nano is an open small reasoning model optimized for fast, cost-efficient inference in agentic and production workloads. Built with a hybrid Mixture-of-Experts (MoE) and Mamba-Transformer architecture, it delivers strong multi-step reasoning, high token throughput, stable latency with predictable cost, and efficient deployment for agent-based systems. Designed for real-world AI systems where reasoning can generate significantly more tokens per prompt, Nemotron Nano reduces compu","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3-nano-30b-a3b"},{"api":"chat","id":"deepinfra/google/gemma-4-26B-A4B-it","object":"model","created":1775227989,"updated":1787843415,"owned_by":"system","input_price":7e-8,"cached_price":7e-8,"output_price":3.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":7e-8,"cached_price":7e-8,"output_price":3.4e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemma 4 26B A4B is a Mixture of Experts model from Google DeepMind built for scalable performance with a 256K context window. Supports text and image input, native thinking mode, function calling, and 140+ languages. Ideal for long context RAG, agentic workflows, coding, and multimodal understanding.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"Don't store prompts and responses","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"google","model_canonical_name":"gemma-4-26b-a4b-it"},{"api":"chat","id":"deepinfra/deepseek-v4-pro-0813","object":"model","created":1786579200,"updated":1787843415,"owned_by":"system","input_price":0.0000013,"cached_price":1e-7,"output_price":0.0000026,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000013,"cached_price":1e-7,"output_price":0.0000026}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek-V4-Pro-0813 is the official release of DeepSeek-V4-Pro, superseding the preview version, with greatly enhanced agentic capabilities and performance improvements that are especially pronounced in production environments. It is built on the DeepSeek-V4-Pro (Preview) model structure, with a DSpark speculative decoding module attached.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro-0813"},{"api":"chat","id":"deepinfra/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B:flex","object":"model","created":1773245239,"updated":1787843415,"owned_by":"system","input_price":6.8e-8,"output_price":3.2e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":6.8e-8,"output_price":3.2e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3 Super is a hybrid Mixture-of-Experts (MoE) model engineered for highest compute efficiency and accuracy in multi-agent applications and specialized agentic systems. It is optimized to run many collaborating agents per application on a single GPU, delivering high accuracy for reasoning, tool use, and instruction following.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nvidia-nemotron-3-super-120b-a12b"},{"api":"chat","id":"deepinfra/google/gemma-4-31B-it:flex","object":"model","created":1775148486,"updated":1787843415,"owned_by":"system","input_price":1.04e-7,"output_price":3.04e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.04e-7,"output_price":3.04e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemma is a family of open models built by Google DeepMind. Gemma 4 models are multimodal, handling text and image input and generating text output.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"google","model_canonical_name":"gemma-4-31b-it"},{"api":"chat","id":"sference/deepseek-v4.1-flash","object":"model","created":1789201097,"updated":1789201097,"owned_by":"system","input_price":5e-7,"cached_price":5e-8,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"cached_price":5e-8,"output_price":0.0000015}],"max_output_tokens":393216,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4.1 Flash is a 552B parameter Mixture of Experts model built on a new Causal Encoder Decoder architecture that activates only 8B parameters for input and 16B for output, giving frontier level intelligence at a fraction of the cost. It is the smallest model in the new DeepSeek architecture family and the first Flash with native vision and multimodal understanding. V4.1 Flash beats DeepSeek V4 Pro on performance, cost, speed and task completion time and scores 88.1 on CyberGym. Its KV cache is compressed to 890 bytes per token, four times smaller than V4 Flash and over 400 times smaller than V1, cutting the cache hit costs that dominate agent workloads. 1M token context window, 384K max output, thinking and non thinking modes, tool calling, structured JSON output and prompt caching. Served in the EU via Sference at $0.50 input, $0.05 cache read and $1.50 output per 1M tokens, with no data retention and no training on user data. Open weights are published on Hugging Face. Ideal for EU hosted coding agents, long context RAG, high throughput batch processing and multimodal assistants.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4.1-flash"},{"api":"chat","id":"sference/glm-5.3-flash","object":"model","created":1787761346,"updated":1788857717,"owned_by":"system","input_price":2e-7,"cached_price":7e-8,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":7e-8,"output_price":6e-7}],"max_output_tokens":262144,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while reducing compute overhead.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp4","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3-flash"},{"api":"chat","id":"sference/deepseek-v4-flash-0731","object":"model","created":1785478908,"updated":1788857757,"owned_by":"system","input_price":2.8e-7,"cached_price":7e-8,"output_price":5.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.8e-7,"cached_price":7e-8,"output_price":5.6e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash is an efficiency-focused MoE model with 284B total parameters (13B active) and a 1M-token context window. It's tuned for fast inference and high-throughput use cases while still holding up on reasoning and coding tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0731"},{"api":"chat","id":"sference/glm-5.2","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":0.0000012,"cached_price":2.6e-7,"output_price":0.0000042,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000012,"cached_price":2.6e-7,"output_price":0.0000042}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.2 is the latest model in the GLM series from Z.ai, continuing its focus on coding and long horizon agentic work. Building on GLM-5.1, it plans, executes, and iterates autonomously on extended engineering grade tasks. Served via Sference.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp4","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2","retires":1790812800},{"api":"chat","id":"sference/kimi-k3","object":"model","created":1784215858,"updated":1788857693,"owned_by":"system","input_price":0.000003,"cached_price":4.5e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"cached_price":4.5e-7,"output_price":0.000015}],"max_output_tokens":262144,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K3 is Moonshot AI's flagship reasoning model with a 1M token context window, strong agentic tool use, and long horizon task execution. Served via Sference.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"fp4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k3"},{"api":"chat","id":"sference/glm-5.3","object":"model","created":1787529600,"updated":1788114567,"owned_by":"system","input_price":0.0000012,"cached_price":2.6e-7,"output_price":0.0000042,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000012,"cached_price":2.6e-7,"output_price":0.0000042}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM 5.3 is the latest Z.ai flagship model for long horizon coding, reasoning, and agentic workflows. It builds on GLM 5.2 with a 1M token context window, stronger engineering task performance, and multiple thinking effort levels. Released under the MIT license. Served via Sference.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","quantization":"fp4","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3"},{"api":"chat","id":"scaleway/glm-5.2","object":"model","created":1781568000,"updated":1788116466,"owned_by":"system","input_price":0.00000209,"output_price":0.00000638,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000209,"output_price":0.00000638}],"max_output_tokens":32768,"context_window":256000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.2 is Z.ai's flagship model for long-horizon coding, agentic engineering, and sustained multi-step execution. 256K context during Scaleway preview (1M native). EU-hosted by Scaleway.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"scaleway/deepseek-v4-flash-0731","object":"model","created":1785456000,"updated":1788116569,"owned_by":"system","input_price":4.6e-7,"cached_price":9e-8,"output_price":9.3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.6e-7,"cached_price":9e-8,"output_price":9.3e-7}],"max_output_tokens":32768,"context_window":256000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash 0731 is DeepSeek's fast, cost-efficient frontier MoE model for coding, reasoning, and agent workflows. 256K context during Scaleway preview (1M native). EU-hosted by Scaleway.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0731"},{"api":"chat","id":"scaleway/gpt-oss-120b","object":"model","created":1754352000,"updated":1788116534,"owned_by":"system","input_price":1.7e-7,"output_price":7e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.7e-7,"output_price":7e-7}],"max_output_tokens":32768,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"gpt-oss-120b is OpenAI's open-weight flagship reasoning MoE model with configurable reasoning effort, native tool use, and structured output. EU-hosted by Scaleway.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"mxfp4","open_weights":true,"model_lab":"openai","model_canonical_name":"gpt-oss-120b"},{"api":"chat","id":"azure/gpt-5@francecentral","object":"model","created":1754587413,"updated":1787843415,"owned_by":"system","input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy in high-stakes use cases. It supports test-time routing features and advanced prompt understanding, including user-specified intent like \"think hard about this.\" Improvements include reductions in hallucination, sycophancy, and better performance in coding, writing, and health-related tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5"},{"api":"chat","id":"azure/gpt-4.1-nano@uksouth","object":"model","created":1744651369,"updated":1787843415,"owned_by":"system","input_price":1.1e-7,"cached_price":2.75e-8,"output_price":4.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.1e-7,"cached_price":2.75e-8,"output_price":4.4e-7}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"uk","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-nano"},{"api":"chat","id":"azure/gpt-5.2@eastus2","object":"model","created":1765389775,"updated":1787843415,"owned_by":"system","input_price":0.00000175,"cached_price":1.75e-7,"output_price":0.000014,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":1.75e-7,"output_price":0.000014}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly to simple queries while spending more depth on complex tasks.\n\nBuilt for broad task coverage, GPT-5.2 delivers consistent gains across math, coding, sciende, and tool calling workloads, with more coherent long-form answers and improved tool-use reliability.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.2"},{"api":"chat","id":"azure/gpt-5@germanywestcentral","object":"model","created":1754587413,"updated":1788971419,"owned_by":"system","input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy in high-stakes use cases. It supports test-time routing features and advanced prompt understanding, including user-specified intent like \"think hard about this.\" Improvements include reductions in hallucination, sycophancy, and better performance in coding, writing, and health-related tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5"},{"api":"chat","id":"azure/o4-mini@westus3","object":"model","created":1744820942,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"azure/gpt-5.5@swedencentral","object":"model","created":1777051893,"updated":1790847020,"owned_by":"system","input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.000033,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.000033},{"prompt_tokens_threshold":272000,"input_price":0.000011,"cached_price":0.0000011,"output_price":0.0000495}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.5 is OpenAI's frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large-scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.5"},{"api":"chat","id":"azure/gpt-5.6-luna@francecentral","object":"model","created":1783590864,"updated":1790845727,"owned_by":"system","input_price":2.2e-7,"caching_price":2.75e-7,"cached_price":2.2e-8,"output_price":0.00000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.2e-7,"caching_price":2.75e-7,"cached_price":2.2e-8,"output_price":0.00000132},{"prompt_tokens_threshold":272000,"input_price":4.4e-7,"caching_price":5.5e-7,"cached_price":4.4e-8,"output_price":0.00000198}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for its price tier.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-luna"},{"api":"chat","id":"azure/gpt-6-luna@westeurope","object":"model","created":1790109419,"updated":1790845763,"owned_by":"system","input_price":1.2e-7,"caching_price":1.5e-7,"cached_price":1.2e-8,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.2e-7,"caching_price":1.5e-7,"cached_price":1.2e-8,"output_price":6e-7},{"prompt_tokens_threshold":272000,"input_price":2.4e-7,"caching_price":3e-7,"cached_price":2.4e-8,"output_price":9e-7}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic tasks, and at higher reasoning effort it can take on complex software engineering and computer-use tasks that previously called for a Sol-tier model.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-luna"},{"api":"chat","id":"azure/gpt-6-sol@westeurope","object":"model","created":1790109419,"updated":1790845771,"owned_by":"system","input_price":0.0000024,"caching_price":0.000003,"cached_price":2.4e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000024,"caching_price":0.000003,"cached_price":2.4e-7,"output_price":0.000012},{"prompt_tokens_threshold":272000,"input_price":0.0000048,"caching_price":0.000006,"cached_price":4.8e-7,"output_price":0.000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional work, agentic coding, business workflow automation, and computer use, and is particularly strong at long-horizon software engineering tasks in real codebases. It approaches Astra-level factual reliability at a much lower cost.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-sol"},{"api":"chat","id":"azure/gpt-4.1-nano@eastus2","object":"model","created":1744651369,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":2.5e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":2.5e-8,"output_price":4e-7}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-nano"},{"api":"chat","id":"azure/gpt-5-mini@westeurope","object":"model","created":1754587407,"updated":1788971546,"owned_by":"system","input_price":2.75e-7,"cached_price":2.75e-8,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.75e-7,"cached_price":2.75e-8,"output_price":0.0000022}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-mini"},{"api":"chat","id":"azure/gpt-6-luna@germanywestcentral","object":"model","created":1790109419,"updated":1790845762,"owned_by":"system","input_price":1.2e-7,"caching_price":1.5e-7,"cached_price":1.2e-8,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.2e-7,"caching_price":1.5e-7,"cached_price":1.2e-8,"output_price":6e-7},{"prompt_tokens_threshold":272000,"input_price":2.4e-7,"caching_price":3e-7,"cached_price":2.4e-8,"output_price":9e-7}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic tasks, and at higher reasoning effort it can take on complex software engineering and computer-use tasks that previously called for a Sol-tier model.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-luna"},{"api":"chat","id":"azure/gpt-5-nano@swedencentral","object":"model","created":1754587402,"updated":1787843415,"owned_by":"system","input_price":5.5e-8,"cached_price":5.5e-9,"output_price":4.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5.5e-8,"cached_price":5.5e-9,"output_price":4.4e-7}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 Nano is our fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-nano"},{"api":"chat","id":"azure/o4-mini@francecentral","object":"model","created":1744820942,"updated":1787843415,"owned_by":"system","input_price":0.00000121,"cached_price":3.025e-7,"output_price":0.00000484,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000121,"cached_price":3.025e-7,"output_price":0.00000484}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"azure/o4-mini@westeurope","object":"model","created":1744820942,"updated":1788971546,"owned_by":"system","input_price":0.00000121,"cached_price":3.025e-7,"output_price":0.00000484,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000121,"cached_price":3.025e-7,"output_price":0.00000484}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"azure/gpt-5.6-sol@eastus2","object":"model","created":1783590850,"updated":1790845734,"owned_by":"system","input_price":0.000004,"caching_price":0.000005,"cached_price":4e-7,"output_price":0.00002,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000004,"caching_price":0.000005,"cached_price":4e-7,"output_price":0.00002},{"prompt_tokens_threshold":272000,"input_price":0.000008,"caching_price":0.00001,"cached_price":8e-7,"output_price":0.00003}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks and long-horizon problem solving.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-sol"},{"api":"chat","id":"azure/gpt-5.1@eastus2","object":"model","created":1763060305,"updated":1787843415,"owned_by":"system","input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning to allocate computation dynamically, responding quickly to simple queries while spending more depth on complex tasks. The model produces clearer, more grounded explanations with reduced jargon, making it easier to follow even on technical or multi-step problems.\n\nBuilt for broad task coverage, GPT-5.1 delivers consistent gains across math, coding, and structured analysis workloads, with more coherent long-form answers and improved tool-use reliability. It also features refined conversational alignment, enabling warmer, more intuitive responses without compromising precision. GPT-5.1 serves as the primary full-capability successor to GPT-5","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.1"},{"api":"chat","id":"azure/gpt-5@westeurope","object":"model","created":1754587413,"updated":1788971546,"owned_by":"system","input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy in high-stakes use cases. It supports test-time routing features and advanced prompt understanding, including user-specified intent like \"think hard about this.\" Improvements include reductions in hallucination, sycophancy, and better performance in coding, writing, and health-related tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5"},{"api":"chat","id":"azure/gpt-5.6-luna@germanywestcentral","object":"model","created":1783590864,"updated":1790845729,"owned_by":"system","input_price":2.2e-7,"caching_price":2.75e-7,"cached_price":2.2e-8,"output_price":0.00000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.2e-7,"caching_price":2.75e-7,"cached_price":2.2e-8,"output_price":0.00000132},{"prompt_tokens_threshold":272000,"input_price":4.4e-7,"caching_price":5.5e-7,"cached_price":4.4e-8,"output_price":0.00000198}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for its price tier.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-luna"},{"api":"chat","id":"azure/gpt-5-nano@francecentral","object":"model","created":1754587402,"updated":1787843415,"owned_by":"system","input_price":5.5e-8,"cached_price":5.5e-9,"output_price":4.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5.5e-8,"cached_price":5.5e-9,"output_price":4.4e-7}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 Nano is our fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-nano"},{"api":"chat","id":"azure/gpt-4o-mini@westeurope","object":"model","created":1721260800,"updated":1788971546,"owned_by":"system","input_price":1.65e-7,"cached_price":8.25e-8,"output_price":6.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.65e-7,"cached_price":8.25e-8,"output_price":6.6e-7}],"max_output_tokens":16000,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4o mini is a fast, affordable small model for lightweight tasks. Supports vision, tools, and caching.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4o-mini"},{"api":"chat","id":"azure/gpt-4.1-mini@westus3","object":"model","created":1744651381,"updated":1787843415,"owned_by":"system","input_price":4e-7,"cached_price":1e-7,"output_price":0.0000016,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"cached_price":1e-7,"output_price":0.0000016}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-mini"},{"api":"chat","id":"azure/gpt-6-sol@eastus2","object":"model","created":1790140750,"updated":1790845766,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001},{"prompt_tokens_threshold":272000,"input_price":0.000004,"caching_price":0.000005,"cached_price":4e-7,"output_price":0.000015}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional work, agentic coding, business workflow automation, and computer use, and is particularly strong at long-horizon software engineering tasks in real codebases. It approaches Astra-level factual reliability at a much lower cost.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-sol"},{"api":"chat","id":"azure/gpt-5-mini@germanywestcentral","object":"model","created":1754587407,"updated":1788971419,"owned_by":"system","input_price":2.75e-7,"cached_price":2.75e-8,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.75e-7,"cached_price":2.75e-8,"output_price":0.0000022}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-mini"},{"api":"chat","id":"azure/gpt-5.4@germanywestcentral","object":"model","created":1772734352,"updated":1790847020,"owned_by":"system","input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165},{"prompt_tokens_threshold":272000,"input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.00002475}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT 5.4 is OpenAI's frontier model designed for complex professional workloads, with strong reasoning, high reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"azure/gpt-5.6-terra@francecentral","object":"model","created":1783590857,"updated":1790845749,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.0000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.0000132},{"prompt_tokens_threshold":272000,"input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.0000198}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic tasks where capability and cost need to be balanced, offering strong performance at roughly half the cost of Sol.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-terra"},{"api":"chat","id":"azure/gpt-5.4@westeurope","object":"model","created":1772734352,"updated":1790847020,"owned_by":"system","input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165},{"prompt_tokens_threshold":272000,"input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.00002475}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT 5.4 is OpenAI's frontier model designed for complex professional workloads, with strong reasoning, high reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"azure/gpt-5-nano@westeurope","object":"model","created":1754587402,"updated":1788971546,"owned_by":"system","input_price":5.5e-8,"cached_price":5.5e-9,"output_price":4.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5.5e-8,"cached_price":5.5e-9,"output_price":4.4e-7}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 Nano is our fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-nano"},{"api":"chat","id":"azure/gpt-5.6-luna@westeurope","object":"model","created":1783590864,"updated":1790845730,"owned_by":"system","input_price":2.2e-7,"caching_price":2.75e-7,"cached_price":2.2e-8,"output_price":0.00000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.2e-7,"caching_price":2.75e-7,"cached_price":2.2e-8,"output_price":0.00000132},{"prompt_tokens_threshold":272000,"input_price":4.4e-7,"caching_price":5.5e-7,"cached_price":4.4e-8,"output_price":0.00000198}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for its price tier.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-luna"},{"api":"chat","id":"azure/gpt-5-mini@eastus2","object":"model","created":1754587407,"updated":1787843415,"owned_by":"system","input_price":2.5e-7,"cached_price":2.5e-8,"output_price":0.000002,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"cached_price":2.5e-8,"output_price":0.000002}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-mini"},{"api":"chat","id":"azure/gpt-5-nano@germanywestcentral","object":"model","created":1754587402,"updated":1788971419,"owned_by":"system","input_price":5.5e-8,"cached_price":5.5e-9,"output_price":4.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5.5e-8,"cached_price":5.5e-9,"output_price":4.4e-7}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 Nano is our fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-nano"},{"api":"chat","id":"azure/gpt-5.6-sol@swedencentral","object":"model","created":1783590850,"updated":1790845743,"owned_by":"system","input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.000022},{"prompt_tokens_threshold":272000,"input_price":0.0000088,"caching_price":0.000011,"cached_price":8.8e-7,"output_price":0.000033}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks and long-horizon problem solving.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-sol"},{"api":"chat","id":"azure/gpt-5-mini@swedencentral","object":"model","created":1754587407,"updated":1787843415,"owned_by":"system","input_price":2.75e-7,"cached_price":2.75e-8,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.75e-7,"cached_price":2.75e-8,"output_price":0.0000022}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-mini"},{"api":"chat","id":"azure/gpt-5-mini@uksouth","object":"model","created":1754587407,"updated":1787843415,"owned_by":"system","input_price":2.75e-7,"cached_price":2.75e-8,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.75e-7,"cached_price":2.75e-8,"output_price":0.0000022}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"uk","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-mini"},{"api":"chat","id":"azure/gpt-5.4-mini@eastus2","object":"model","created":1773748178,"updated":1787843415,"owned_by":"system","input_price":7.5e-7,"cached_price":7.5e-8,"output_price":0.0000045,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.5e-7,"cached_price":7.5e-8,"output_price":0.0000045}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding, and tool use, while reducing latency and cost for large-scale deployments.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4-mini"},{"api":"chat","id":"azure/gpt-5.5@eastus2","object":"model","created":1777051893,"updated":1787843415,"owned_by":"system","input_price":0.000005,"cached_price":5e-7,"output_price":0.00003,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"cached_price":5e-7,"output_price":0.00003},{"prompt_tokens_threshold":272000,"input_price":0.00001,"cached_price":0.000001,"output_price":0.000045}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.5 is OpenAI's frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large-scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.5"},{"api":"chat","id":"azure/gpt-5.4","object":"model","created":1772734352,"updated":1787843415,"owned_by":"system","input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.000015},{"prompt_tokens_threshold":272000,"input_price":0.000005,"cached_price":5e-7,"output_price":0.0000225}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT 5.4 is OpenAI's frontier model designed for complex professional workloads, with strong reasoning, high reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"azure/gpt-5.6-terra@westeurope","object":"model","created":1783590857,"updated":1790845752,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.0000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.0000132},{"prompt_tokens_threshold":272000,"input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.0000198}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic tasks where capability and cost need to be balanced, offering strong performance at roughly half the cost of Sol.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-terra"},{"api":"chat","id":"azure/gpt-6-sol@francecentral","object":"model","created":1790109419,"updated":1790845768,"owned_by":"system","input_price":0.0000024,"caching_price":0.000003,"cached_price":2.4e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000024,"caching_price":0.000003,"cached_price":2.4e-7,"output_price":0.000012},{"prompt_tokens_threshold":272000,"input_price":0.0000048,"caching_price":0.000006,"cached_price":4.8e-7,"output_price":0.000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional work, agentic coding, business workflow automation, and computer use, and is particularly strong at long-horizon software engineering tasks in real codebases. It approaches Astra-level factual reliability at a much lower cost.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-sol"},{"api":"chat","id":"azure/o4-mini@swedencentral","object":"model","created":1744820942,"updated":1787843415,"owned_by":"system","input_price":0.00000121,"cached_price":3.025e-7,"output_price":0.00000484,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000121,"cached_price":3.025e-7,"output_price":0.00000484}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"azure/gpt-6-sol@swedencentral","object":"model","created":1790109419,"updated":1790845772,"owned_by":"system","input_price":0.0000024,"caching_price":0.000003,"cached_price":2.4e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000024,"caching_price":0.000003,"cached_price":2.4e-7,"output_price":0.000012},{"prompt_tokens_threshold":272000,"input_price":0.0000048,"caching_price":0.000006,"cached_price":4.8e-7,"output_price":0.000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional work, agentic coding, business workflow automation, and computer use, and is particularly strong at long-horizon software engineering tasks in real codebases. It approaches Astra-level factual reliability at a much lower cost.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-sol"},{"api":"chat","id":"azure/gpt-5.1@swedencentral","object":"model","created":1763060305,"updated":1787843415,"owned_by":"system","input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning to allocate computation dynamically, responding quickly to simple queries while spending more depth on complex tasks. The model produces clearer, more grounded explanations with reduced jargon, making it easier to follow even on technical or multi-step problems.\n\nBuilt for broad task coverage, GPT-5.1 delivers consistent gains across math, coding, and structured analysis workloads, with more coherent long-form answers and improved tool-use reliability. It also features refined conversational alignment, enabling warmer, more intuitive responses without compromising precision. GPT-5.1 serves as the primary full-capability successor to GPT-5","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.1"},{"api":"chat","id":"azure/gpt-5.4@swedencentral","object":"model","created":1772734352,"updated":1790847020,"owned_by":"system","input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165},{"prompt_tokens_threshold":272000,"input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.00002475}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT 5.4 is OpenAI's frontier model designed for complex professional workloads, with strong reasoning, high reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"azure/gpt-5.1@francecentral","object":"model","created":1763060305,"updated":1787843415,"owned_by":"system","input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning to allocate computation dynamically, responding quickly to simple queries while spending more depth on complex tasks. The model produces clearer, more grounded explanations with reduced jargon, making it easier to follow even on technical or multi-step problems.\n\nBuilt for broad task coverage, GPT-5.1 delivers consistent gains across math, coding, and structured analysis workloads, with more coherent long-form answers and improved tool-use reliability. It also features refined conversational alignment, enabling warmer, more intuitive responses without compromising precision. GPT-5.1 serves as the primary full-capability successor to GPT-5","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.1"},{"api":"chat","id":"azure/gpt-6-luna@swedencentral","object":"model","created":1790109419,"updated":1790845765,"owned_by":"system","input_price":1.2e-7,"caching_price":1.5e-7,"cached_price":1.2e-8,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.2e-7,"caching_price":1.5e-7,"cached_price":1.2e-8,"output_price":6e-7},{"prompt_tokens_threshold":272000,"input_price":2.4e-7,"caching_price":3e-7,"cached_price":2.4e-8,"output_price":9e-7}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic tasks, and at higher reasoning effort it can take on complex software engineering and computer-use tasks that previously called for a Sol-tier model.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-luna"},{"api":"chat","id":"azure/gpt-4.1@westeurope","object":"model","created":1744651385,"updated":1788971546,"owned_by":"system","input_price":0.0000022,"cached_price":5.5e-7,"output_price":0.0000088,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"cached_price":5.5e-7,"output_price":0.0000088}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1"},{"api":"chat","id":"azure/gpt-5.6-sol@westeurope","object":"model","created":1783590850,"updated":1790845739,"owned_by":"system","input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.000022},{"prompt_tokens_threshold":272000,"input_price":0.0000088,"caching_price":0.000011,"cached_price":8.8e-7,"output_price":0.000033}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks and long-horizon problem solving.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-sol"},{"api":"chat","id":"azure/gpt-4.1@swedencentral","object":"model","created":1744651385,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"cached_price":5.5e-7,"output_price":0.0000088,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"cached_price":5.5e-7,"output_price":0.0000088}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1"},{"api":"chat","id":"azure/gpt-5.4@francecentral","object":"model","created":1772734352,"updated":1790847020,"owned_by":"system","input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000275,"cached_price":2.75e-7,"output_price":0.0000165},{"prompt_tokens_threshold":272000,"input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.00002475}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT 5.4 is OpenAI's frontier model designed for complex professional workloads, with strong reasoning, high reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"azure/gpt-5","object":"model","created":1754587413,"updated":1787843415,"owned_by":"system","input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy in high-stakes use cases. It supports test-time routing features and advanced prompt understanding, including user-specified intent like \"think hard about this.\" Improvements include reductions in hallucination, sycophancy, and better performance in coding, writing, and health-related tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5"},{"api":"chat","id":"azure/gpt-4.1-nano@francecentral","object":"model","created":1744651369,"updated":1787843415,"owned_by":"system","input_price":1.1e-7,"cached_price":2.75e-8,"output_price":4.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.1e-7,"cached_price":2.75e-8,"output_price":4.4e-7}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-nano"},{"api":"chat","id":"azure/gpt-4.1@uksouth","object":"model","created":1744651385,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"cached_price":5.5e-7,"output_price":0.0000088,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"cached_price":5.5e-7,"output_price":0.0000088}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"uk","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1"},{"api":"chat","id":"azure/gpt-4.1@francecentral","object":"model","created":1744651385,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"cached_price":5.5e-7,"output_price":0.0000088,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"cached_price":5.5e-7,"output_price":0.0000088}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1"},{"api":"chat","id":"azure/gpt-5.3-codex@eastus2","object":"model","created":1771959164,"updated":1787843415,"owned_by":"system","input_price":0.00000175,"cached_price":1.75e-7,"output_price":0.000014,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":1.75e-7,"output_price":0.000014}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.3-codex"},{"api":"chat","id":"azure/gpt-4.1-nano@westeurope","object":"model","created":1744651369,"updated":1788971546,"owned_by":"system","input_price":1.1e-7,"cached_price":2.75e-8,"output_price":4.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.1e-7,"cached_price":2.75e-8,"output_price":4.4e-7}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-nano"},{"api":"chat","id":"azure/gpt-5@swedencentral","object":"model","created":1754587413,"updated":1787843415,"owned_by":"system","input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy in high-stakes use cases. It supports test-time routing features and advanced prompt understanding, including user-specified intent like \"think hard about this.\" Improvements include reductions in hallucination, sycophancy, and better performance in coding, writing, and health-related tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5"},{"api":"chat","id":"azure/o4-mini@eastus2","object":"model","created":1744820942,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"cached_price":2.75e-7,"output_price":0.0000044}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"azure/gpt-4o-mini@germanywestcentral","object":"model","created":1721260800,"updated":1788971419,"owned_by":"system","input_price":1.65e-7,"cached_price":8.25e-8,"output_price":6.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.65e-7,"cached_price":8.25e-8,"output_price":6.6e-7}],"max_output_tokens":16000,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4o mini is a fast, affordable small model for lightweight tasks. Supports vision, tools, and caching.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4o-mini"},{"api":"chat","id":"azure/gpt-4.1-mini@eastus2","object":"model","created":1744651381,"updated":1787843415,"owned_by":"system","input_price":4e-7,"cached_price":1e-7,"output_price":0.0000016,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"cached_price":1e-7,"output_price":0.0000016}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-mini"},{"api":"chat","id":"azure/gpt-5@uksouth","object":"model","created":1754587413,"updated":1787843415,"owned_by":"system","input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001375,"cached_price":1.375e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy in high-stakes use cases. It supports test-time routing features and advanced prompt understanding, including user-specified intent like \"think hard about this.\" Improvements include reductions in hallucination, sycophancy, and better performance in coding, writing, and health-related tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"uk","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5"},{"api":"chat","id":"azure/gpt-5.6-terra@eastus2","object":"model","created":1783590857,"updated":1790845747,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.000012},{"prompt_tokens_threshold":272000,"input_price":0.000004,"caching_price":0.000005,"cached_price":4e-7,"output_price":0.000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic tasks where capability and cost need to be balanced, offering strong performance at roughly half the cost of Sol.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-terra"},{"api":"chat","id":"azure/gpt-6-astra@eastus2","object":"model","created":1788555830,"updated":1790845756,"owned_by":"system","input_price":0.00001,"caching_price":0.0000125,"cached_price":0.000001,"output_price":0.00005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00001,"caching_price":0.0000125,"cached_price":0.000001,"output_price":0.00005},{"prompt_tokens_threshold":272000,"input_price":0.00002,"caching_price":0.000025,"cached_price":0.000002,"output_price":0.000075}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Astra is OpenAI's most capable model, built for the hardest end-to-end work. It is suited for complex reasoning, coding, computer use, research, and document creation, with reasoning effort levels from low up to xhigh and max.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-astra"},{"api":"chat","id":"azure/gpt-5.2-codex@eastus2","object":"model","created":1768409315,"updated":1787843415,"owned_by":"system","input_price":0.00000175,"cached_price":1.75e-7,"output_price":0.000014,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":1.75e-7,"output_price":0.000014}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"OpenAI's most intelligent coding model optimized for long-horizon, agentic coding tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.2-codex"},{"api":"chat","id":"azure/gpt-5.4@eastus2","object":"model","created":1772734352,"updated":1787843415,"owned_by":"system","input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":2.5e-7,"output_price":0.000015},{"prompt_tokens_threshold":272000,"input_price":0.000005,"cached_price":5e-7,"output_price":0.0000225}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT 5.4 is OpenAI's frontier model designed for complex professional workloads, with strong reasoning, high reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4"},{"api":"chat","id":"azure/gpt-4o-mini@swedencentral","object":"model","created":1721260800,"updated":1787843415,"owned_by":"system","input_price":1.65e-7,"cached_price":8.25e-8,"output_price":6.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.65e-7,"cached_price":8.25e-8,"output_price":6.6e-7}],"max_output_tokens":16000,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4o mini is a fast, affordable small model for lightweight tasks. Supports vision, tools, and caching.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4o-mini"},{"api":"chat","id":"azure/gpt-5.6-sol@francecentral","object":"model","created":1783590850,"updated":1790845737,"owned_by":"system","input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.000022},{"prompt_tokens_threshold":272000,"input_price":0.0000088,"caching_price":0.000011,"cached_price":8.8e-7,"output_price":0.000033}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks and long-horizon problem solving.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-sol"},{"api":"chat","id":"azure/gpt-5-mini@francecentral","object":"model","created":1754587407,"updated":1787843415,"owned_by":"system","input_price":2.75e-7,"cached_price":2.75e-8,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.75e-7,"cached_price":2.75e-8,"output_price":0.0000022}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-mini"},{"api":"chat","id":"azure/gpt-6-luna@francecentral","object":"model","created":1790109419,"updated":1790845761,"owned_by":"system","input_price":1.2e-7,"caching_price":1.5e-7,"cached_price":1.2e-8,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.2e-7,"caching_price":1.5e-7,"cached_price":1.2e-8,"output_price":6e-7},{"prompt_tokens_threshold":272000,"input_price":2.4e-7,"caching_price":3e-7,"cached_price":2.4e-8,"output_price":9e-7}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic tasks, and at higher reasoning effort it can take on complex software engineering and computer-use tasks that previously called for a Sol-tier model.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-luna"},{"api":"chat","id":"azure/gpt-5-mini","object":"model","created":1754587407,"updated":1787843415,"owned_by":"system","input_price":2.5e-7,"cached_price":2.5e-8,"output_price":0.000002,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"cached_price":2.5e-8,"output_price":0.000002}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-mini"},{"api":"chat","id":"azure/gpt-6.1-sol@westeurope","object":"model","created":1790640000,"updated":1790845777,"owned_by":"system","input_price":0.0000024,"caching_price":0.000003,"cached_price":1.2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000024,"caching_price":0.000003,"cached_price":1.2e-7,"output_price":0.000012},{"prompt_tokens_threshold":272000,"input_price":0.0000048,"caching_price":0.000006,"cached_price":2.4e-7,"output_price":0.000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6.1 Sol delivers near-Astra performance at a lower cost for complex coding, computer use, and professional work. It succeeds GPT-6 Sol with the same 1M context and 128K output, cheaper cached input, and reasoning effort from low to max.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6.1-sol"},{"api":"chat","id":"azure/gpt-4.1@eastus2","object":"model","created":1744651385,"updated":1787843415,"owned_by":"system","input_price":0.000002,"cached_price":5e-7,"output_price":0.000008,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"cached_price":5e-7,"output_price":0.000008}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1"},{"api":"chat","id":"azure/gpt-6.1-sol@germanywestcentral","object":"model","created":1790640000,"updated":1790845776,"owned_by":"system","input_price":0.0000024,"caching_price":0.000003,"cached_price":1.2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000024,"caching_price":0.000003,"cached_price":1.2e-7,"output_price":0.000012},{"prompt_tokens_threshold":272000,"input_price":0.0000048,"caching_price":0.000006,"cached_price":2.4e-7,"output_price":0.000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6.1 Sol delivers near-Astra performance at a lower cost for complex coding, computer use, and professional work. It succeeds GPT-6 Sol with the same 1M context and 128K output, cheaper cached input, and reasoning effort from low to max.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6.1-sol"},{"api":"chat","id":"azure/gpt-5.5@westeurope","object":"model","created":1777051893,"updated":1790847020,"owned_by":"system","input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.000033,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.000033},{"prompt_tokens_threshold":272000,"input_price":0.000011,"cached_price":0.0000011,"output_price":0.0000495}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.5 is OpenAI's frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large-scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.5"},{"api":"chat","id":"azure/gpt-5.1","object":"model","created":1763060305,"updated":1787843415,"owned_by":"system","input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning to allocate computation dynamically, responding quickly to simple queries while spending more depth on complex tasks. The model produces clearer, more grounded explanations with reduced jargon, making it easier to follow even on technical or multi-step problems.\n\nBuilt for broad task coverage, GPT-5.1 delivers consistent gains across math, coding, and structured analysis workloads, with more coherent long-form answers and improved tool-use reliability. It also features refined conversational alignment, enabling warmer, more intuitive responses without compromising precision. GPT-5.1 serves as the primary full-capability successor to GPT-5","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.1"},{"api":"chat","id":"azure/o4-mini@germanywestcentral","object":"model","created":1744820942,"updated":1788971419,"owned_by":"system","input_price":0.00000121,"cached_price":3.025e-7,"output_price":0.00000484,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000121,"cached_price":3.025e-7,"output_price":0.00000484}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"o4-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"o4-mini"},{"api":"chat","id":"azure/gpt-5.6-terra@germanywestcentral","object":"model","created":1783590857,"updated":1790845751,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.0000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.0000132},{"prompt_tokens_threshold":272000,"input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.0000198}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic tasks where capability and cost need to be balanced, offering strong performance at roughly half the cost of Sol.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-terra"},{"api":"chat","id":"azure/gpt-4.1-mini@francecentral","object":"model","created":1744651381,"updated":1787843415,"owned_by":"system","input_price":4.4e-7,"cached_price":1.1e-7,"output_price":0.00000176,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.4e-7,"cached_price":1.1e-7,"output_price":0.00000176}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-mini"},{"api":"chat","id":"azure/gpt-5.6-luna@swedencentral","object":"model","created":1783590864,"updated":1790845731,"owned_by":"system","input_price":2.2e-7,"caching_price":2.75e-7,"cached_price":2.2e-8,"output_price":0.00000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.2e-7,"caching_price":2.75e-7,"cached_price":2.2e-8,"output_price":0.00000132},{"prompt_tokens_threshold":272000,"input_price":4.4e-7,"caching_price":5.5e-7,"cached_price":4.4e-8,"output_price":0.00000198}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for its price tier.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-luna"},{"api":"chat","id":"azure/gpt-5.6-luna@eastus2","object":"model","created":1783590864,"updated":1790845726,"owned_by":"system","input_price":2e-7,"caching_price":2.5e-7,"cached_price":2e-8,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"caching_price":2.5e-7,"cached_price":2e-8,"output_price":0.0000012},{"prompt_tokens_threshold":272000,"input_price":4e-7,"caching_price":5e-7,"cached_price":4e-8,"output_price":0.0000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for its price tier.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-luna"},{"api":"chat","id":"azure/gpt-4.1-nano@germanywestcentral","object":"model","created":1744651369,"updated":1788971419,"owned_by":"system","input_price":1.1e-7,"cached_price":2.75e-8,"output_price":4.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.1e-7,"cached_price":2.75e-8,"output_price":4.4e-7}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-nano"},{"api":"chat","id":"azure/gpt-4.1-nano@westus3","object":"model","created":1744651369,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":2.5e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":2.5e-8,"output_price":4e-7}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-nano"},{"api":"chat","id":"azure/gpt-5.5@germanywestcentral","object":"model","created":1777051893,"updated":1790847020,"owned_by":"system","input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.000033,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"cached_price":5.5e-7,"output_price":0.000033},{"prompt_tokens_threshold":272000,"input_price":0.000011,"cached_price":0.0000011,"output_price":0.0000495}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.5 is OpenAI's frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token context window (922K input, 128K output) with support for text and image inputs, enabling large-scale reasoning, coding, and multimodal workflows within a single system.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.5"},{"api":"chat","id":"azure/gpt-6-luna@eastus2","object":"model","created":1790140750,"updated":1790845759,"owned_by":"system","input_price":1e-7,"caching_price":1.25e-7,"cached_price":1e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.25e-7,"cached_price":1e-8,"output_price":5e-7},{"prompt_tokens_threshold":272000,"input_price":2e-7,"caching_price":2.5e-7,"cached_price":2e-8,"output_price":7.5e-7}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic tasks, and at higher reasoning effort it can take on complex software engineering and computer-use tasks that previously called for a Sol-tier model.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-luna"},{"api":"chat","id":"azure/gpt-5-nano","object":"model","created":1754587402,"updated":1787843415,"owned_by":"system","input_price":5e-8,"cached_price":5e-9,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"cached_price":5e-9,"output_price":4e-7}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 Nano is our fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-nano"},{"api":"chat","id":"azure/gpt-4.1","object":"model","created":1744651385,"updated":1787843415,"owned_by":"system","input_price":0.000002,"cached_price":5e-7,"output_price":0.000008,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"cached_price":5e-7,"output_price":0.000008}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1"},{"api":"chat","id":"azure/gpt-6.1-sol@swedencentral","object":"model","created":1790640000,"updated":1790845778,"owned_by":"system","input_price":0.0000024,"caching_price":0.000003,"cached_price":1.2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000024,"caching_price":0.000003,"cached_price":1.2e-7,"output_price":0.000012},{"prompt_tokens_threshold":272000,"input_price":0.0000048,"caching_price":0.000006,"cached_price":2.4e-7,"output_price":0.000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6.1 Sol delivers near-Astra performance at a lower cost for complex coding, computer use, and professional work. It succeeds GPT-6 Sol with the same 1M context and 128K output, cheaper cached input, and reasoning effort from low to max.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6.1-sol"},{"api":"chat","id":"azure/gpt-5.6-sol@germanywestcentral","object":"model","created":1783590850,"updated":1790845738,"owned_by":"system","input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.000022},{"prompt_tokens_threshold":272000,"input_price":0.0000088,"caching_price":0.000011,"cached_price":8.8e-7,"output_price":0.000033}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks and long-horizon problem solving.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-sol"},{"api":"chat","id":"azure/gpt-4.1-nano@swedencentral","object":"model","created":1744651369,"updated":1787843415,"owned_by":"system","input_price":1.1e-7,"cached_price":2.75e-8,"output_price":4.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.1e-7,"cached_price":2.75e-8,"output_price":4.4e-7}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-nano"},{"api":"chat","id":"azure/gpt-4.1-nano","object":"model","created":1744651369,"updated":1787843415,"owned_by":"system","input_price":1e-7,"cached_price":2.5e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"cached_price":2.5e-8,"output_price":4e-7}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-nano"},{"api":"chat","id":"azure/gpt-6-sol@germanywestcentral","object":"model","created":1790109419,"updated":1790845769,"owned_by":"system","input_price":0.0000024,"caching_price":0.000003,"cached_price":2.4e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000024,"caching_price":0.000003,"cached_price":2.4e-7,"output_price":0.000012},{"prompt_tokens_threshold":272000,"input_price":0.0000048,"caching_price":0.000006,"cached_price":4.8e-7,"output_price":0.000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional work, agentic coding, business workflow automation, and computer use, and is particularly strong at long-horizon software engineering tasks in real codebases. It approaches Astra-level factual reliability at a much lower cost.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6-sol"},{"api":"chat","id":"azure/gpt-4.1@westus3","object":"model","created":1744651385,"updated":1787843415,"owned_by":"system","input_price":0.000002,"cached_price":5e-7,"output_price":0.000008,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"cached_price":5e-7,"output_price":0.000008}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1"},{"api":"chat","id":"azure/gpt-6.1-sol@eastus2","object":"model","created":1790640000,"updated":1790845773,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"cached_price":1e-7,"output_price":0.00001},{"prompt_tokens_threshold":272000,"input_price":0.000004,"caching_price":0.000005,"cached_price":2e-7,"output_price":0.000015}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6.1 Sol delivers near-Astra performance at a lower cost for complex coding, computer use, and professional work. It succeeds GPT-6 Sol with the same 1M context and 128K output, cheaper cached input, and reasoning effort from low to max.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6.1-sol"},{"api":"chat","id":"azure/gpt-6.1-sol@francecentral","object":"model","created":1790640000,"updated":1790845774,"owned_by":"system","input_price":0.0000024,"caching_price":0.000003,"cached_price":1.2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000024,"caching_price":0.000003,"cached_price":1.2e-7,"output_price":0.000012},{"prompt_tokens_threshold":272000,"input_price":0.0000048,"caching_price":0.000006,"cached_price":2.4e-7,"output_price":0.000018}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-6.1 Sol delivers near-Astra performance at a lower cost for complex coding, computer use, and professional work. It succeeds GPT-6 Sol with the same 1M context and 128K output, cheaper cached input, and reasoning effort from low to max.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-6.1-sol"},{"api":"chat","id":"azure/gpt-5-nano@eastus2","object":"model","created":1754587402,"updated":1787843415,"owned_by":"system","input_price":5e-8,"cached_price":5e-9,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"cached_price":5e-9,"output_price":4e-7}],"max_output_tokens":100000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 Nano is our fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5-nano"},{"api":"chat","id":"azure/gpt-5@eastus2","object":"model","created":1754587413,"updated":1787843415,"owned_by":"system","input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"cached_price":1.25e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy in high-stakes use cases. It supports test-time routing features and advanced prompt understanding, including user-specified intent like \"think hard about this.\" Improvements include reductions in hallucination, sycophancy, and better performance in coding, writing, and health-related tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5"},{"api":"chat","id":"azure/gpt-4o-mini@eastus2","object":"model","created":1721260800,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":7.5e-8,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":7.5e-8,"output_price":6e-7}],"max_output_tokens":16000,"context_window":128000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4o mini is a fast, affordable small model for lightweight tasks. Supports vision, tools, and caching.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4o-mini"},{"api":"chat","id":"azure/gpt-5.4-mini","object":"model","created":1773748178,"updated":1787843415,"owned_by":"system","input_price":7.5e-7,"cached_price":7.5e-8,"output_price":0.0000045,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.5e-7,"cached_price":7.5e-8,"output_price":0.0000045}],"max_output_tokens":128000,"context_window":400000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding, and tool use, while reducing latency and cost for large-scale deployments.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.4-mini"},{"api":"chat","id":"azure/gpt-5.6-terra@swedencentral","object":"model","created":1783590857,"updated":1790845755,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.0000132,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.0000132},{"prompt_tokens_threshold":272000,"input_price":0.0000044,"caching_price":0.0000055,"cached_price":4.4e-7,"output_price":0.0000198}],"max_output_tokens":128000,"context_window":1050000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":true,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic tasks where capability and cost need to be balanced, offering strong performance at roughly half the cost of Sol.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-5.6-terra"},{"api":"chat","id":"azure/gpt-4.1@germanywestcentral","object":"model","created":1744651385,"updated":1788971419,"owned_by":"system","input_price":0.0000022,"cached_price":5.5e-7,"output_price":0.0000088,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"cached_price":5.5e-7,"output_price":0.0000088}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1"},{"api":"chat","id":"azure/gpt-4.1-mini@uksouth","object":"model","created":1744651381,"updated":1787843415,"owned_by":"system","input_price":4.4e-7,"cached_price":1.1e-7,"output_price":0.00000176,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.4e-7,"cached_price":1.1e-7,"output_price":0.00000176}],"max_output_tokens":32768,"context_window":1047576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"uk","open_weights":false,"model_lab":"openai","model_canonical_name":"gpt-4.1-mini"},{"api":"chat","id":"vertex/gemini-3.8-flash@eu","object":"model","created":1788307200,"updated":1788365777,"owned_by":"system","input_price":8.25e-7,"cached_price":8.25e-8,"output_price":0.000004125,"discount_percentage":50,"pricing":[{"prompt_tokens_threshold":0,"input_price":8.25e-7,"cached_price":8.25e-8,"output_price":0.000004125}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.8 Flash is Google's most intelligent workhorse model, delivering significant improvements over 3.7 Flash across software engineering, agentic tasks, and critical multi step reasoning in specialized domains.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.8-flash"},{"api":"chat","id":"vertex/claude-opus-4-8","object":"model","created":1779905091,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 4.8 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.7, it delivers stronger performance on complex, multi-step tasks and more reliable agentic execution across extended workflows. It is especially effective for asynchronous agent pipelines where tasks unfold over time - large codebases, multi-stage debugging, and end-to-end project orchestration.\n\nBeyond coding, Opus 4.8 brings improved knowledge work capabilities - from drafting documents and building presentations to analyzing data. It maintains coherence across very long outputs and extended sessions, making it a strong default for tasks that require persistence, judgment, and follow-through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-8"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite@us-south1","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"vertex/claude-opus-4-7@us","object":"model","created":1776351100,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on complex, multi-step tasks and more reliable agentic execution across extended workflows. It is especially effective for asynchronous agent pipelines where tasks unfold over time - large codebases, multi-stage debugging, and end-to-end project orchestration.\n\nBeyond coding, Opus 4.7 brings improved knowledge work capabilities - from drafting documents and building presentations to analyzing data. It maintains coherence across very long outputs and extended sessions, making it a strong default for tasks that require persistence, judgment, and follow-through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-7"},{"api":"chat","id":"vertex/gemini-2.5-flash-image","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/claude-sonnet-5@eu","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"vertex/kimi-k2","object":"model","created":1752263252,"updated":1787843415,"owned_by":"system","input_price":6e-7,"caching_price":0.0000025,"cached_price":6e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"caching_price":0.0000025,"cached_price":6e-8,"output_price":0.0000025}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2 Thinking is an open-source model that operates as a \"thinking agent,\" reasoning step-by-step while using tools to achieve state-of-the-art performance on various benchmarks. It is capable of executing up to 200-300 sequential tool calls without human intervention, allowing it to solve complex problems across a wide range of tasks. The model uses Quantization-Aware Training (QAT) to support INT4 inference, which provides a roughly 2x improvement in generation speed.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2"},{"api":"chat","id":"vertex/gemini-3.1-flash-image:flex","object":"model","created":1781754065,"updated":1787843415,"owned_by":"system","input_price":2.5e-7,"caching_price":0,"cached_price":0,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"caching_price":0,"cached_price":0,"output_price":0.0000015}],"max_output_tokens":32768,"context_window":131072,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Flash Image is optimized for image understanding and generation and offers a balance of price and performance.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-flash-image"},{"api":"chat","id":"vertex/claude-opus-4-5@europe-west1","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite@europe-west8","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@europe-southwest1","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/gemini-3.7-flash","object":"model","created":1786641818,"updated":1787843415,"owned_by":"system","input_price":7.5e-7,"cached_price":7.5e-8,"output_price":0.00000375,"discount_percentage":50,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.5e-7,"cached_price":7.5e-8,"output_price":0.00000375}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.7 Flash is the high efficiency agentic workhorse of the Gemini 3 family. It delivers Pro level agentic capabilities with major leaps in code generation and terminal execution, bridging deep reasoning Pro models and high throughput Flash Lite models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.7-flash"},{"api":"chat","id":"vertex/claude-sonnet-4-6@europe-west1","object":"model","created":1771342990,"updated":1788220122,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3.3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3.3e-7,"output_price":0.0000165}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with memory, polished document creation, and confident computer use for web QA and workflow automation","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"vertex/gemini-3.5-flash-lite","object":"model","created":1784646726,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":3e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":3e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash-Lite is a high efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi agent workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash-lite"},{"api":"chat","id":"vertex/claude-sonnet-5","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"caching_5m_price":0.0000025,"caching_1h_price":0.000004,"cached_price":2e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"vertex/claude-haiku-4-5@europe-west1","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"caching_price":0.000001375,"cached_price":1.1e-7,"output_price":0.0000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"caching_price":0.000001375,"caching_5m_price":0.000001375,"caching_1h_price":0.0000022,"cached_price":1.1e-7,"output_price":0.0000055}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@us-west1","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/gemini-2.5-flash-image@us-west4","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/claude-opus-5-5@us","object":"model","created":1790098200,"updated":1790668360,"owned_by":"system","input_price":0.0000044,"caching_price":0.0000055,"cached_price":2.2e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"caching_price":0.0000055,"caching_5m_price":0.0000055,"caching_1h_price":0.0000088,"cached_price":2.2e-7,"output_price":0.000022}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 5.5 is the first model in Anthropic's Claude 5.5 family and the most capable Opus to date, built for long running autonomous agents, large scale coding, and frontier knowledge work. It improves on Opus 5 in planning, sustained multi step execution, and judgment across extended sessions, at a lower price than Opus 5.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5-5"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@us-south1","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/claude-opus-4-6@europe-west1","object":"model","created":1770219050,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.6 is Anthropic's most powerful model yet and the best coding model in the world.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-6"},{"api":"chat","id":"vertex/gemini-3.8-flash","object":"model","created":1788307200,"updated":1788365728,"owned_by":"system","input_price":7.5e-7,"cached_price":7.5e-8,"output_price":0.00000375,"discount_percentage":50,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.5e-7,"cached_price":7.5e-8,"output_price":0.00000375}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.8 Flash is Google's most intelligent workhorse model, delivering significant improvements over 3.7 Flash across software engineering, agentic tasks, and critical multi step reasoning in specialized domains.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.8-flash"},{"api":"chat","id":"vertex/claude-sonnet-4-5@europe-west1","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3.3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3.3e-7,"output_price":0.0000165},{"prompt_tokens_threshold":200000,"input_price":0.0000066,"caching_price":0.00000825,"caching_5m_price":0.00000825,"caching_1h_price":0.0000132,"cached_price":6.6e-7,"output_price":0.00002475}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@europe-west1","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/gemini-2.5-flash@us-central1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/gemini-3-pro-image:flex","object":"model","created":1781754054,"updated":1787843415,"owned_by":"system","input_price":0.000001,"caching_price":0.0000045,"cached_price":1e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"caching_price":0.0000045,"cached_price":1e-7,"output_price":0.000006}],"max_output_tokens":32768,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3 Pro Image, or Gemini 3 Pro (with Nano Banana), is designed to tackle the most challenging image generation by incorporating state-of-the-art reasoning capabilities. It's the best model for complex and multi-turn image generation and editing, having improved accuracy and enhanced image quality.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3-pro-image"},{"api":"chat","id":"vertex/gemini-2.5-flash@us-west1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/gemini-3.5-flash-lite@eu","object":"model","created":1784646726,"updated":1787843415,"owned_by":"system","input_price":3.3e-7,"cached_price":3.3e-8,"output_price":0.00000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.3e-7,"cached_price":3.3e-8,"output_price":0.00000275}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash-Lite is a high efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi agent workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash-lite"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite@europe-north1","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@us-central1","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/claude-opus-5-5","object":"model","created":1790098200,"updated":1790668360,"owned_by":"system","input_price":0.000004,"caching_price":0.000005,"cached_price":2e-7,"output_price":0.00002,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000004,"caching_price":0.000005,"caching_5m_price":0.000005,"caching_1h_price":0.000008,"cached_price":2e-7,"output_price":0.00002}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 5.5 is the first model in Anthropic's Claude 5.5 family and the most capable Opus to date, built for long running autonomous agents, large scale coding, and frontier knowledge work. It improves on Opus 5 in planning, sustained multi step execution, and judgment across extended sessions, at a lower price than Opus 5.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5-5"},{"api":"chat","id":"vertex/gemini-3.1-flash-image","object":"model","created":1781754065,"updated":1788220122,"owned_by":"system","input_price":5e-7,"caching_price":0,"cached_price":0,"output_price":0.000002,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"caching_price":0,"cached_price":0,"output_price":0.000002}],"max_output_tokens":32768,"context_window":131072,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Flash Image is optimized for image understanding and generation and offers a balance of price and performance.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-flash-image"},{"api":"chat","id":"vertex/gemini-2.5-pro@europe-west1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite@europe-west4","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"vertex/claude-fable-5@eu","object":"model","created":1781007515,"updated":1787843415,"owned_by":"system","input_price":0.000011,"caching_price":0.00001375,"cached_price":0.0000011,"output_price":0.000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000011,"caching_price":0.00001375,"caching_5m_price":0.00001375,"caching_1h_price":0.000022,"cached_price":0.0000011,"output_price":0.000055}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Fable 5 is Anthropic's most capable generally available model for autonomous knowledge work and coding. It is a Mythos-class model with robust safeguards. It can handle long-running, complex, and asynchronous tasks where previous models would have needed more frequent check-ins.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-fable-5"},{"api":"chat","id":"vertex/gemini-3-pro-preview","object":"model","created":1771509627,"updated":1788220514,"owned_by":"system","input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012},{"prompt_tokens_threshold":200000,"input_price":0.000004,"caching_price":0.000009,"cached_price":4e-7,"output_price":0.000018}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Pro is the next iteration in the Gemini 3 series of models, a suite of highly capable, natively multimodal reasoning models. As of this model card’s date of publication, Gemini 3.1 Pro is Google’s most advanced model for complex tasks. Geminin 3.1 Pro can comprehend vast datasets and challenging problems from massively multimodal information sources, including text, audio, images, video, and entire code repositories.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-pro-preview"},{"api":"chat","id":"vertex/claude-opus-4-6","object":"model","created":1770219050,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.6 is Anthropic's most powerful model yet and the best coding model in the world.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-6"},{"api":"chat","id":"vertex/gemini-3.1-pro-preview","object":"model","created":1771509627,"updated":1788220514,"owned_by":"system","input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012},{"prompt_tokens_threshold":200000,"input_price":0.000004,"caching_price":0.000009,"cached_price":4e-7,"output_price":0.000018}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Pro is the next iteration in the Gemini 3 series of models, a suite of highly capable, natively multimodal reasoning models. As of this model card’s date of publication, Gemini 3.1 Pro is Google’s most advanced model for complex tasks. Geminin 3.1 Pro can comprehend vast datasets and challenging problems from massively multimodal information sources, including text, audio, images, video, and entire code repositories.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-pro-preview"},{"api":"chat","id":"vertex/gemini-2.5-flash@europe-west4","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/gemini-3.6-flash","object":"model","created":1784646733,"updated":1787843415,"owned_by":"system","input_price":0.0000015,"cached_price":1.5e-7,"output_price":0.000007,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000015,"cached_price":1.5e-7,"output_price":0.000007}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.6 Flash is the latest Flash model in the Gemini series, delivering frontier level intelligence at high speed and low cost. Built for the agentic era, it excels at sub agent deployment, multi step workflows, and long horizon tasks at scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.6-flash"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"vertex/deepseek-v3.2","object":"model","created":1764594642,"updated":1787843415,"owned_by":"system","input_price":5.6e-7,"caching_price":0.00000168,"cached_price":5.6e-8,"output_price":0.00000168,"pricing":[{"prompt_tokens_threshold":0,"input_price":5.6e-7,"caching_price":0.00000168,"cached_price":5.6e-8,"output_price":0.00000168}],"max_output_tokens":65535,"context_window":163840,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek-V3.2 is a model that harmonizes high computational efficiency with superior reasoning and agent performance. DeepSeek's approach is built upon three key technical breakthroughs: DeepSeek Sparse Attention (DSA), scalable reinforcement learning framework, and large scale agentic task synthesis pipeline.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v3.2"},{"api":"chat","id":"vertex/gemini-3-pro-image","object":"model","created":1781754054,"updated":1788220122,"owned_by":"system","input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000045,"cached_price":2e-7,"output_price":0.000012}],"max_output_tokens":32768,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3 Pro Image, or Gemini 3 Pro (with Nano Banana), is designed to tackle the most challenging image generation by incorporating state-of-the-art reasoning capabilities. It's the best model for complex and multi-turn image generation and editing, having improved accuracy and enhanced image quality.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3-pro-image"},{"api":"chat","id":"vertex/gemini-2.5-pro@europe-north1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite@us-east5","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"vertex/gemini-3.5-flash","object":"model","created":1779193800,"updated":1787843415,"owned_by":"system","input_price":0.0000015,"caching_price":0.000001583,"cached_price":1.5e-7,"output_price":0.000009,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000015,"caching_price":0.000001583,"cached_price":1.5e-7,"output_price":0.000009}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash provides sustained frontier-level intelligence optimized for real-world tasks at a higher speed and lower cost. Designed for the agentic era, it excels at sub-agent deployment, multi-step workflows, and long-horizon tasks at scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash"},{"api":"chat","id":"vertex/gemini-3.5-flash@eu","object":"model","created":1779193800,"updated":1787843415,"owned_by":"system","input_price":0.00000165,"caching_price":0.0000017413,"cached_price":1.65e-7,"output_price":0.0000099,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000165,"caching_price":0.0000017413,"cached_price":1.65e-7,"output_price":0.0000099}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash provides sustained frontier-level intelligence optimized for real-world tasks at a higher speed and lower cost. Designed for the agentic era, it excels at sub-agent deployment, multi-step workflows, and long-horizon tasks at scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash"},{"api":"chat","id":"vertex/gemini-3.7-flash@eu","object":"model","created":1786641818,"updated":1787843415,"owned_by":"system","input_price":8.25e-7,"cached_price":8.25e-8,"output_price":0.000004125,"discount_percentage":50,"pricing":[{"prompt_tokens_threshold":0,"input_price":8.25e-7,"cached_price":8.25e-8,"output_price":0.000004125}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.7 Flash is the high efficiency agentic workhorse of the Gemini 3 family. It delivers Pro level agentic capabilities with major leaps in code generation and terminal execution, bridging deep reasoning Pro models and high throughput Flash Lite models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.7-flash"},{"api":"chat","id":"vertex/claude-opus-4-8@eu","object":"model","created":1779905091,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 4.8 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.7, it delivers stronger performance on complex, multi-step tasks and more reliable agentic execution across extended workflows. It is especially effective for asynchronous agent pipelines where tasks unfold over time - large codebases, multi-stage debugging, and end-to-end project orchestration.\n\nBeyond coding, Opus 4.8 brings improved knowledge work capabilities - from drafting documents and building presentations to analyzing data. It maintains coherence across very long outputs and extended sessions, making it a strong default for tasks that require persistence, judgment, and follow-through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-8"},{"api":"chat","id":"vertex/claude-opus-4-5@us-east5","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"vertex/claude-fable-5.1@eu","object":"model","created":1788220800,"updated":1788288205,"owned_by":"system","input_price":0.000011,"caching_price":0.00001375,"cached_price":2.75e-7,"output_price":0.000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000011,"caching_price":0.00001375,"caching_5m_price":0.00001375,"caching_1h_price":0.000022,"cached_price":2.75e-7,"output_price":0.000055}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Fable 5.1 is Anthropic's most capable model, improving on Claude Fable 5 across agentic coding, long running agentic workflows, and knowledge work such as large code refactors, front end and visual code generation, and finance and analysis tasks. It ships a 1M token context window by default with 128K max output, is more token efficient and concise than Fable 5, and supports per message effort control. It is a Mythos class model with robust safeguards.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-fable-5.1"},{"api":"chat","id":"vertex/claude-sonnet-4-5@us-east5","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3.3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3.3e-7,"output_price":0.0000165},{"prompt_tokens_threshold":200000,"input_price":0.0000066,"caching_price":0.00000825,"caching_5m_price":0.00000825,"caching_1h_price":0.0000132,"cached_price":6.6e-7,"output_price":0.00002475}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite@us-east1","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"vertex/gemini-2.5-pro@us-west1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/gemini-2.5-pro","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/claude-sonnet-4-6@us-east5","object":"model","created":1771342990,"updated":1788220122,"owned_by":"system","input_price":0.0000033,"caching_price":0.000004125,"cached_price":3.3e-7,"output_price":0.0000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000033,"caching_price":0.000004125,"caching_5m_price":0.000004125,"caching_1h_price":0.0000066,"cached_price":3.3e-7,"output_price":0.0000165}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with memory, polished document creation, and confident computer use for web QA and workflow automation","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"vertex/claude-opus-4","object":"model","created":1747933971,"updated":1787843415,"owned_by":"system","input_price":0.000015,"caching_price":0.00001875,"cached_price":0.0000015,"output_price":0.000075,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000015,"caching_price":0.00001875,"caching_5m_price":0.00001875,"caching_1h_price":0.00003,"cached_price":0.0000015,"output_price":0.000075}],"max_output_tokens":32000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4"},{"api":"chat","id":"vertex/claude-sonnet-5-5","object":"model","created":1790553600,"updated":1790668471,"owned_by":"system","input_price":0.000002,"caching_price":0.0000025,"cached_price":2e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"caching_price":0.0000025,"caching_5m_price":0.0000025,"caching_1h_price":0.000004,"cached_price":2e-7,"output_price":0.00001}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5.5 is a step-change improvement over Sonnet 5. It performs great with well-scoped everyday tasks like building features, fixing bugs, and creating polished documents, slides, and spreadsheets. Sonnet 5.5 also improves on communication and writes more clearly than Sonnet 5.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5-5"},{"api":"chat","id":"vertex/claude-haiku-4-5@us-east5","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.0000011,"caching_price":0.000001375,"cached_price":1.1e-7,"output_price":0.0000055,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000011,"caching_price":0.000001375,"caching_5m_price":0.000001375,"caching_1h_price":0.0000022,"cached_price":1.1e-7,"output_price":0.0000055}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"vertex/gemini-3.1-flash-lite@us","object":"model","created":1778168828,"updated":1788220122,"owned_by":"system","input_price":2.75e-7,"caching_price":9.1663e-8,"cached_price":2.75e-8,"output_price":0.00000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.75e-7,"caching_price":9.1663e-8,"cached_price":2.75e-8,"output_price":0.00000165}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Flash Lite Preview is the most cost-efficient model in the Gemini family, optimized for high-volume, low-latency tasks. It delivers fast responses with solid quality for everyday use cases including summarization, classification, and simple reasoning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-flash-lite"},{"api":"chat","id":"vertex/gemini-3.1-pro-preview:flex","object":"model","created":1771509627,"updated":1788220514,"owned_by":"system","input_price":0.000001,"caching_price":0.00000275,"cached_price":1e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"caching_price":0.00000275,"cached_price":1e-7,"output_price":0.000006},{"prompt_tokens_threshold":200000,"input_price":0.000002,"caching_price":0.0000055,"cached_price":2e-7,"output_price":0.000009}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Pro is the next iteration in the Gemini 3 series of models, a suite of highly capable, natively multimodal reasoning models. As of this model card’s date of publication, Gemini 3.1 Pro is Google’s most advanced model for complex tasks. Geminin 3.1 Pro can comprehend vast datasets and challenging problems from massively multimodal information sources, including text, audio, images, video, and entire code repositories.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-pro-preview"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@us-east5","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/claude-sonnet-5-5@us","object":"model","created":1790553600,"updated":1790668471,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5.5 is a step-change improvement over Sonnet 5. It performs great with well-scoped everyday tasks like building features, fixing bugs, and creating polished documents, slides, and spreadsheets. Sonnet 5.5 also improves on communication and writes more clearly than Sonnet 5.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5-5"},{"api":"chat","id":"vertex/claude-opus-4-8@us","object":"model","created":1779905091,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 4.8 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.7, it delivers stronger performance on complex, multi-step tasks and more reliable agentic execution across extended workflows. It is especially effective for asynchronous agent pipelines where tasks unfold over time - large codebases, multi-stage debugging, and end-to-end project orchestration.\n\nBeyond coding, Opus 4.8 brings improved knowledge work capabilities - from drafting documents and building presentations to analyzing data. It maintains coherence across very long outputs and extended sessions, making it a strong default for tasks that require persistence, judgment, and follow-through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-8"},{"api":"chat","id":"vertex/claude-opus-5@eu","object":"model","created":1784912544,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 5 is the most capable model in Anthropic's Opus family, built for long running, asynchronous agents operating with full autonomy. It advances the coding and agentic strengths of Opus 4.8 with stronger planning, more reliable execution across extended multi step workflows, and better judgment on when to act and when to escalate. It excels in asynchronous agent pipelines spanning large codebases, multi stage debugging, and end to end project orchestration. Beyond coding, Opus 5 delivers frontier knowledge work, from drafting documents and building presentations to deep data analysis, and maintains coherence across very long outputs and extended sessions, making it the default choice for work that requires persistence, judgment, and follow through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5"},{"api":"chat","id":"vertex/gemini-2.5-flash@europe-west8","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@europe-west4","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/claude-opus-4-1","object":"model","created":1754411591,"updated":1787843415,"owned_by":"system","input_price":0.000015,"caching_price":0.00001875,"cached_price":0.0000015,"output_price":0.000075,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000015,"caching_price":0.00001875,"caching_5m_price":0.00001875,"caching_1h_price":0.00003,"cached_price":0.0000015,"output_price":0.000075}],"max_output_tokens":32000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-1"},{"api":"chat","id":"vertex/gemini-2.5-pro@us-south1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/claude-sonnet-4@us-east5","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"vertex/gemini-3.5-flash:flex","object":"model","created":1779193800,"updated":1787843415,"owned_by":"system","input_price":7.5e-7,"caching_price":0.000001583,"cached_price":7.5e-8,"output_price":0.0000045,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.5e-7,"caching_price":0.000001583,"cached_price":7.5e-8,"output_price":0.0000045}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash provides sustained frontier-level intelligence optimized for real-world tasks at a higher speed and lower cost. Designed for the agentic era, it excels at sub-agent deployment, multi-step workflows, and long-horizon tasks at scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash"},{"api":"chat","id":"vertex/gemini-3.1-flash-lite:flex","object":"model","created":1778168828,"updated":1787843415,"owned_by":"system","input_price":1.25e-7,"caching_price":4.1665e-8,"cached_price":1.25e-8,"output_price":7.5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.25e-7,"caching_price":4.1665e-8,"cached_price":1.25e-8,"output_price":7.5e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Flash Lite Preview is the most cost-efficient model in the Gemini family, optimized for high-volume, low-latency tasks. It delivers fast responses with solid quality for everyday use cases including summarization, classification, and simple reasoning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-flash-lite"},{"api":"chat","id":"vertex/gemini-3.5-flash-lite:flex","object":"model","created":1784646726,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":1.5e-8,"output_price":0.00000125,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":1.5e-8,"output_price":0.00000125}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash-Lite is a high efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi agent workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash-lite"},{"api":"chat","id":"vertex/claude-opus-4-7","object":"model","created":1776351100,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on complex, multi-step tasks and more reliable agentic execution across extended workflows. It is especially effective for asynchronous agent pipelines where tasks unfold over time - large codebases, multi-stage debugging, and end-to-end project orchestration.\n\nBeyond coding, Opus 4.7 brings improved knowledge work capabilities - from drafting documents and building presentations to analyzing data. It maintains coherence across very long outputs and extended sessions, making it a strong default for tasks that require persistence, judgment, and follow-through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-7"},{"api":"chat","id":"vertex/claude-opus-4-1@us-east5","object":"model","created":1754411591,"updated":1787843415,"owned_by":"system","input_price":0.000015,"caching_price":0.00001875,"cached_price":0.0000015,"output_price":0.000075,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000015,"caching_price":0.00001875,"caching_5m_price":0.00001875,"caching_1h_price":0.00003,"cached_price":0.0000015,"output_price":0.000075}],"max_output_tokens":32000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-1"},{"api":"chat","id":"vertex/gemini-3.6-flash:flex","object":"model","created":1784646733,"updated":1787843415,"owned_by":"system","input_price":3.75e-7,"cached_price":3.75e-8,"output_price":0.000001875,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.75e-7,"cached_price":3.75e-8,"output_price":0.000001875}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.6 Flash is the latest Flash model in the Gemini series, delivering frontier level intelligence at high speed and low cost. Built for the agentic era, it excels at sub agent deployment, multi step workflows, and long horizon tasks at scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.6-flash"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@europe-west8","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/claude-sonnet-5-5@eu","object":"model","created":1790553600,"updated":1790668471,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5.5 is a step-change improvement over Sonnet 5. It performs great with well-scoped everyday tasks like building features, fixing bugs, and creating polished documents, slides, and spreadsheets. Sonnet 5.5 also improves on communication and writes more clearly than Sonnet 5.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5-5"},{"api":"chat","id":"vertex/claude-fable-5.1","object":"model","created":1788220800,"updated":1788288205,"owned_by":"system","input_price":0.00001,"caching_price":0.0000125,"cached_price":2.5e-7,"output_price":0.00005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00001,"caching_price":0.0000125,"caching_5m_price":0.0000125,"caching_1h_price":0.00002,"cached_price":2.5e-7,"output_price":0.00005}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Fable 5.1 is Anthropic's most capable model, improving on Claude Fable 5 across agentic coding, long running agentic workflows, and knowledge work such as large code refactors, front end and visual code generation, and finance and analysis tasks. It ships a 1M token context window by default with 128K max output, is more token efficient and concise than Fable 5, and supports per message effort control. It is a Mythos class model with robust safeguards.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-fable-5.1"},{"api":"chat","id":"vertex/gemini-2.5-flash@us-south1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite@us-central1","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"vertex/claude-sonnet-4","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"vertex/gemini-2.5-pro@us-east5","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/gemini-3-flash-preview","object":"model","created":1765987078,"updated":1787843415,"owned_by":"system","input_price":5e-7,"caching_price":0.000001,"cached_price":5e-8,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"caching_price":0.000001,"cached_price":5e-8,"output_price":0.000003}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3 Flash Preview is designed to deliver strong agentic capabilities (near-Pro level) at substantial speed and value. Making it perfect for engaging multi-turn chats, and collaborating back and forth with your coding agent without getting out of flow. Compared to 2.5 Flash it delivers significant improvements across the board.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3-flash-preview"},{"api":"chat","id":"vertex/claude-fable-5","object":"model","created":1781007515,"updated":1787843415,"owned_by":"system","input_price":0.00001,"caching_price":0.0000125,"cached_price":0.000001,"output_price":0.00005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00001,"caching_price":0.0000125,"caching_5m_price":0.0000125,"caching_1h_price":0.00002,"cached_price":0.000001,"output_price":0.00005}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Fable 5 is Anthropic's most capable generally available model for autonomous knowledge work and coding. It is a Mythos-class model with robust safeguards. It can handle long-running, complex, and asynchronous tasks where previous models would have needed more frequent check-ins.","data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-fable-5"},{"api":"chat","id":"vertex/gemini-2.5-flash@us-east5","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/claude-sonnet-4-6","object":"model","created":1771342990,"updated":1788220122,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with memory, polished document creation, and confident computer use for web QA and workflow automation","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-6"},{"api":"chat","id":"vertex/claude-opus-5-5@eu","object":"model","created":1790098200,"updated":1790668360,"owned_by":"system","input_price":0.0000044,"caching_price":0.0000055,"cached_price":2.2e-7,"output_price":0.000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000044,"caching_price":0.0000055,"caching_5m_price":0.0000055,"caching_1h_price":0.0000088,"cached_price":2.2e-7,"output_price":0.000022}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Opus 5.5 is the first model in Anthropic's Claude 5.5 family and the most capable Opus to date, built for long running autonomous agents, large scale coding, and frontier knowledge work. It improves on Opus 5 in planning, sustained multi step execution, and judgment across extended sessions, at a lower price than Opus 5.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5-5"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite@europe-west1","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@europe-north1","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/gemini-3-flash-preview:flex","object":"model","created":1765987078,"updated":1787843415,"owned_by":"system","input_price":2.5e-7,"caching_price":5e-7,"cached_price":2.5e-8,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"caching_price":5e-7,"cached_price":2.5e-8,"output_price":0.0000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3 Flash Preview is designed to deliver strong agentic capabilities (near-Pro level) at substantial speed and value. Making it perfect for engaging multi-turn chats, and collaborating back and forth with your coding agent without getting out of flow. Compared to 2.5 Flash it delivers significant improvements across the board.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3-flash-preview"},{"api":"chat","id":"vertex/gemini-2.5-pro@europe-west8","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/gemini-2.5-flash@us-east1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/gemini-2.5-pro@europe-central2","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/gemini-2.5-pro@us-east1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/gemini-2.5-flash","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@us-east4","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/claude-sonnet-4@europe-west1","object":"model","created":1747933971,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@us-east1","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/gemini-2.5-flash@europe-west1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/claude-opus-4@us-east5","object":"model","created":1747933971,"updated":1787843415,"owned_by":"system","input_price":0.000015,"caching_price":0.00001875,"cached_price":0.0000015,"output_price":0.000075,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000015,"caching_price":0.00001875,"caching_5m_price":0.00001875,"caching_1h_price":0.00003,"cached_price":0.0000015,"output_price":0.000075}],"max_output_tokens":32000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4"},{"api":"chat","id":"vertex/gemini-2.5-pro@europe-west4","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/gemini-3.1-flash-lite@eu","object":"model","created":1778168828,"updated":1788220122,"owned_by":"system","input_price":2.75e-7,"caching_price":9.1663e-8,"cached_price":2.75e-8,"output_price":0.00000165,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.75e-7,"caching_price":9.1663e-8,"cached_price":2.75e-8,"output_price":0.00000165}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Flash Lite Preview is the most cost-efficient model in the Gemini family, optimized for high-volume, low-latency tasks. It delivers fast responses with solid quality for everyday use cases including summarization, classification, and simple reasoning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-flash-lite"},{"api":"chat","id":"vertex/gemini-3.5-flash@us","object":"model","created":1779193800,"updated":1787843415,"owned_by":"system","input_price":0.00000165,"caching_price":0.0000017413,"cached_price":1.65e-7,"output_price":0.0000099,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000165,"caching_price":0.0000017413,"cached_price":1.65e-7,"output_price":0.0000099}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.5 Flash provides sustained frontier-level intelligence optimized for real-world tasks at a higher speed and lower cost. Designed for the agentic era, it excels at sub-agent deployment, multi-step workflows, and long-horizon tasks at scale.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.5-flash"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite@europe-central2","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"vertex/claude-opus-4-7@eu","object":"model","created":1776351100,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on complex, multi-step tasks and more reliable agentic execution across extended workflows. It is especially effective for asynchronous agent pipelines where tasks unfold over time - large codebases, multi-stage debugging, and end-to-end project orchestration.\n\nBeyond coding, Opus 4.7 brings improved knowledge work capabilities - from drafting documents and building presentations to analyzing data. It maintains coherence across very long outputs and extended sessions, making it a strong default for tasks that require persistence, judgment, and follow-through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-7"},{"api":"chat","id":"vertex/gemini-2.5-flash@europe-central2","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/gemini-2.5-flash-image@europe-central2","object":"model","created":1759870431,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":0.0000025,"cached_price":3e-7,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":true,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-image","retires":1790899200},{"api":"chat","id":"vertex/gemini-3.1-flash-lite","object":"model","created":1778168828,"updated":1788220122,"owned_by":"system","input_price":2.5e-7,"caching_price":8.333e-8,"cached_price":2.5e-8,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"caching_price":8.333e-8,"cached_price":2.5e-8,"output_price":0.0000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 3.1 Flash Lite Preview is the most cost-efficient model in the Gemini family, optimized for high-volume, low-latency tasks. It delivers fast responses with solid quality for everyday use cases including summarization, classification, and simple reasoning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-3.1-flash-lite"},{"api":"chat","id":"vertex/claude-opus-4-6@us-east5","object":"model","created":1770219050,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.6 is Anthropic's most powerful model yet and the best coding model in the world.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-6"},{"api":"chat","id":"vertex/gemini-2.5-flash@europe-north1","object":"model","created":1750172488,"updated":1787843415,"owned_by":"system","input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"caching_price":5.5e-7,"cached_price":7.5e-8,"output_price":0.0000025}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash"},{"api":"chat","id":"vertex/claude-opus-5","object":"model","created":1784912544,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 5 is the most capable model in Anthropic's Opus family, built for long running, asynchronous agents operating with full autonomy. It advances the coding and agentic strengths of Opus 4.8 with stronger planning, more reliable execution across extended multi step workflows, and better judgment on when to act and when to escalate. It excels in asynchronous agent pipelines spanning large codebases, multi stage debugging, and end to end project orchestration. Beyond coding, Opus 5 delivers frontier knowledge work, from drafting documents and building presentations to deep data analysis, and maintains coherence across very long outputs and extended sessions, making it the default choice for work that requires persistence, judgment, and follow through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5"},{"api":"chat","id":"vertex/claude-opus-5@us","object":"model","created":1784912544,"updated":1787843415,"owned_by":"system","input_price":0.0000055,"caching_price":0.000006875,"cached_price":5.5e-7,"output_price":0.0000275,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000055,"caching_price":0.000006875,"caching_5m_price":0.000006875,"caching_1h_price":0.000011,"cached_price":5.5e-7,"output_price":0.0000275}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Opus 5 is the most capable model in Anthropic's Opus family, built for long running, asynchronous agents operating with full autonomy. It advances the coding and agentic strengths of Opus 4.8 with stronger planning, more reliable execution across extended multi step workflows, and better judgment on when to act and when to escalate. It excels in asynchronous agent pipelines spanning large codebases, multi stage debugging, and end to end project orchestration. Beyond coding, Opus 5 delivers frontier knowledge work, from drafting documents and building presentations to deep data analysis, and maintains coherence across very long outputs and extended sessions, making it the default choice for work that requires persistence, judgment, and follow through.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-5"},{"api":"chat","id":"vertex/claude-haiku-4-5","object":"model","created":1760547638,"updated":1787843415,"owned_by":"system","input_price":0.000001,"caching_price":0.00000125,"cached_price":1e-7,"output_price":0.000005,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"caching_price":0.00000125,"caching_5m_price":0.00000125,"caching_1h_price":0.000002,"cached_price":1e-7,"output_price":0.000005}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost-effective AI model, offering near-frontier intelligence. It matches the coding and computer-use performance of older flagship models like Sonnet 4, but runs at a fraction of the price.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-haiku-4-5"},{"api":"chat","id":"vertex/claude-sonnet-4-5","object":"model","created":1759161676,"updated":1788220514,"owned_by":"system","input_price":0.000003,"caching_price":0.00000375,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.00000375,"caching_5m_price":0.00000375,"caching_1h_price":0.000006,"cached_price":3e-7,"output_price":0.000015},{"prompt_tokens_threshold":200000,"input_price":0.000006,"caching_price":0.0000075,"caching_5m_price":0.0000075,"caching_1h_price":0.000012,"cached_price":6e-7,"output_price":0.0000225}],"max_output_tokens":64000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-4-5"},{"api":"chat","id":"vertex/claude-opus-4-5","object":"model","created":1764010580,"updated":1787843415,"owned_by":"system","input_price":0.000005,"caching_price":0.00000625,"cached_price":5e-7,"output_price":0.000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000005,"caching_price":0.00000625,"caching_5m_price":0.00000625,"caching_1h_price":0.00001,"cached_price":5e-7,"output_price":0.000025}],"max_output_tokens":64000,"context_window":200000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":false,"description":"Claude Opus 4.5 is Anthropic's flagship AI model, celebrated as an industry leader for complex coding, autonomous agent tasks, and deep analysis. It combines extreme reasoning capabilities with high token efficiency, making multi-day engineering projects and routine office tasks drastically faster and more cost-effective","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-opus-4-5"},{"api":"chat","id":"vertex/claude-sonnet-5@us","object":"model","created":1782843083,"updated":1787843415,"owned_by":"system","input_price":0.0000022,"caching_price":0.00000275,"cached_price":2.2e-7,"output_price":0.000011,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000022,"caching_price":0.00000275,"caching_5m_price":0.00000275,"caching_1h_price":0.0000044,"cached_price":2.2e-7,"output_price":0.000011}],"max_output_tokens":128000,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Claude Sonnet 5 is the latest model in the Sonnet family and an upgrade to Sonnet 4.6, with gains in agentic coding and professional work. It delivers top-tier intelligence at Sonnet pricing, well suited to production agents, high-volume pipelines, and everyday professional work like document drafting, spreadsheet analysis, and presentations.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"anthropic","model_canonical_name":"claude-sonnet-5"},{"api":"chat","id":"vertex/gemini-2.5-pro@us-central1","object":"model","created":1750169544,"updated":1788220514,"owned_by":"system","input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"caching_price":0.000002375,"cached_price":3.1e-7,"output_price":0.00001},{"prompt_tokens_threshold":200000,"input_price":0.0000025,"caching_price":0.00000475,"cached_price":6.2e-7,"output_price":0.000015}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":true,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-pro"},{"api":"chat","id":"vertex/gemini-2.5-flash-lite@us-west1","object":"model","created":1753200276,"updated":1787843415,"owned_by":"system","input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"caching_price":1.8333e-7,"cached_price":1e-8,"output_price":4e-7}],"max_output_tokens":65535,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Google's smallest and most cost effective model, built for at scale usage.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":false,"model_lab":"google","model_canonical_name":"gemini-2.5-flash-lite"},{"api":"chat","id":"tensorx/qwen3.8-2.4t-a95b","object":"model","created":1787842260,"updated":1787842260,"owned_by":"system","input_price":0.0000025,"cached_price":6.3e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":6.3e-7,"output_price":0.000006}],"max_output_tokens":262144,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of Qwen3.8 Max, with 95 billion active parameters out of 2.4 trillion total. It is suited for coding, research, complex reasoning, and agentic workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3.8-2.4t-a95b"},{"api":"chat","id":"tensorx/deepseek-v4-flash-0731","object":"model","created":1785478908,"updated":1787843415,"owned_by":"system","input_price":2.5e-7,"cached_price":6e-8,"output_price":3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"cached_price":6e-8,"output_price":3e-7}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash 0731 is the July 31 update of DeepSeek's efficiency-optimized MoE model (284B total, 13B active parameters) with a 1M-token context window. It is built for fast inference and high-throughput workloads while keeping strong reasoning and coding performance. Reasoning is toggleable and off by default.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","quantization":"fp4","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0731"},{"api":"chat","id":"tensorx/kimi-k3","object":"model","created":1784215858,"updated":1787843415,"owned_by":"system","input_price":0.000003,"cached_price":7.5e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"cached_price":7.5e-7,"output_price":0.000015}],"max_output_tokens":262144,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K3 is Moonshot AI's flagship reasoning model with a 1M token context window, vision input, strong agentic tool use, and long horizon task execution. Served via TensorX.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","quantization":"mxfp4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k3"},{"api":"chat","id":"tensorx/deepseek-v4-pro-0424","object":"model","created":1777000679,"updated":1787843415,"owned_by":"system","input_price":0.00000175,"cached_price":4.4e-7,"output_price":0.0000035,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":4.4e-7,"output_price":0.0000035}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Pro is a large scale Mixture of Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M token context window. It is built for advanced reasoning, coding, and long horizon agent workflows, with strong performance across knowledge, math, and software engineering benchmarks. Built on the same architecture as DeepSeek V4 Flash, it adds a hybrid attention system for efficient long context processing. Reasoning efforts high and xhigh are supported, with xhigh mapping to max reasoning. It is well suited for complex workloads such as full codebase analysis, multi step automation, and large scale information synthesis, where both capability and efficiency are critical.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro-0424"},{"api":"chat","id":"tensorx/kimi-k2.7-code","object":"model","created":1781266361,"updated":1787843415,"owned_by":"system","input_price":0.00000125,"cached_price":3.1e-7,"output_price":0.0000045,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000125,"cached_price":3.1e-7,"output_price":0.0000045}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.7 Code is a coding focused model in Moonshot AI Kimi K2 family, built to complete end to end programming tasks reliably over long contexts. It uses a native multimodal mixture of experts architecture that accepts text and image input, and it always operates in a thinking mode, preserving full reasoning content across multi turn conversations. With a 256K token context window, it targets long horizon coding, agentic task decomposition, and multi turn dialogue. The model activates 32B parameters out of roughly 1T total.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","quantization":"int4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.7-code"},{"api":"chat","id":"tensorx/deepseek-v4.1-flash","object":"model","created":1789201097,"updated":1790082980,"owned_by":"system","input_price":5e-7,"cached_price":1.3e-7,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"cached_price":1.3e-7,"output_price":0.0000015}],"max_output_tokens":393216,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4.1 Flash is a 552B parameter Mixture of Experts model built on a new Causal Encoder Decoder architecture that activates only 8B parameters for input and 16B for output, giving frontier level intelligence at a fraction of the cost. It is the smallest model in the new DeepSeek architecture family and the first Flash with native vision and multimodal understanding. V4.1 Flash beats DeepSeek V4 Pro on performance, cost, speed and task completion time. 1M token context window, 384K max output, thinking and non thinking modes, tool calling, structured JSON output and prompt caching. Served in the EU via TensorX at $0.50 input, $0.13 cache read and $1.50 output per 1M tokens, with no data retention and no training on user data. Open weights are published on Hugging Face.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4.1-flash"},{"api":"chat","id":"tensorx/glm-5.3","object":"model","created":1787077232,"updated":1787940598,"owned_by":"system","input_price":0.00000175,"cached_price":4.4e-7,"output_price":0.0000045,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":4.4e-7,"output_price":0.0000045}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM 5.3 is the latest Z.ai flagship model for long horizon coding, reasoning, and agentic workflows. It builds on GLM 5.2 with a 1M token context window, stronger engineering task performance, and multiple thinking effort levels. Released under the MIT license, it is designed for sustained work over large repositories, complex tool use, and multi step technical tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3"},{"api":"chat","id":"tensorx/qwen3.8","object":"model","created":1787842260,"updated":1787842260,"owned_by":"system","input_price":0.0000025,"cached_price":6.3e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000025,"cached_price":6.3e-7,"output_price":0.000006}],"max_output_tokens":262144,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of Qwen3.8 Max, with 95 billion active parameters out of 2.4 trillion total. It is suited for coding, research, complex reasoning, and agentic workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3.8-2.4t-a95b"},{"api":"chat","id":"tensorx/minimax-m3","object":"model","created":1780245374,"updated":1787843415,"owned_by":"system","input_price":4e-7,"cached_price":1e-7,"output_price":0.000002,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"cached_price":1e-7,"output_price":0.000002}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiniMax M3 is a frontier multimodal model with a 1M token context window built on MiniMax Sparse Attention (MSA). It delivers frontier level performance on coding and agentic tasks, outperforming GPT 5.5 and Gemini 3.1 Pro on SWE Bench Pro and approaching Claude Opus 4.7. It natively handles image and video input and is the first open weight model to combine frontier coding, ultra long context, and native multimodality.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","quantization":"fp8","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m3"},{"api":"chat","id":"tensorx/glm-5.3-flash","object":"model","created":1787761346,"updated":1788282815,"owned_by":"system","input_price":2e-7,"cached_price":5e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":5e-8,"output_price":5e-7}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while reducing compute overhead.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3-flash"},{"api":"chat","id":"tensorx/qwen3.8-flash-next","object":"model","created":1787702400,"updated":1788283802,"owned_by":"system","input_price":2e-7,"cached_price":5e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":5e-8,"output_price":5e-7}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen3.8 Flash Next is a fast, efficiency optimized model from Qwen with a 256K token context window. It supports vision input, reasoning, and strong function calling with tool choice, making it well suited for high throughput coding assistants and agentic workflows where speed and cost matter.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3.8-flash-next"},{"api":"chat","id":"tensorx/deepseek-v4-pro-0813","object":"model","created":1786579200,"updated":1788276423,"owned_by":"system","input_price":0.00000175,"cached_price":4.4e-7,"output_price":0.0000035,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":4.4e-7,"output_price":0.0000035}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Pro (0813) is DeepSeek's flagship model for advanced reasoning, coding, and agentic workflows. This snapshot brings a 1M token context window with strong function calling and tool choice support, designed for long horizon tasks over large codebases and complex multi step tool use.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro-0813"},{"api":"chat","id":"tensorx/glm-5.2","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":0.0000015,"cached_price":3.8e-7,"output_price":0.0000045,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000015,"cached_price":3.8e-7,"output_price":0.0000045}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM 5.2 is the latest Z.ai flagship model for long horizon coding, reasoning, and agentic workflows. It improves on GLM 5.1 with a 1M token context window, stronger engineering task performance, multiple thinking effort levels, and architectural changes that reduce long context inference cost while improving speculative decoding. Released under the MIT license, it is designed for sustained work over large repositories, complex tool use, and multi step technical tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"No training on user data","geolocation":"eu","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"tencent/hy3","object":"model","created":1783296000,"updated":1790594486,"owned_by":"system","input_price":1.32e-7,"cached_price":3.3e-8,"output_price":5.28e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.32e-7,"cached_price":3.3e-8,"output_price":5.28e-7}],"max_output_tokens":131072,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Hy3 is Tencent's open-weight (Apache 2.0) MoE model with a 256K context window, optimized for agentic workflows, coding and productivity tasks with strong token efficiency.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","quantization":"fp8","open_weights":true,"model_lab":"tencent","model_canonical_name":"hy3"},{"api":"chat","id":"tencent/glm-5.3","object":"model","created":1787011200,"updated":1790594486,"owned_by":"system","input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.3 is Zhipu's latest flagship model with a 1M token context window, built for deep reasoning, complex agent and coding tasks and long-running workflows.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3"},{"api":"chat","id":"tencent/mimo-v2.6-pro","object":"model","created":1790035200,"updated":1790594486,"owned_by":"system","input_price":4.35e-7,"cached_price":3.6e-9,"output_price":8.7e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.35e-7,"cached_price":3.6e-9,"output_price":8.7e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiMo-V2.6-Pro is Xiaomi's flagship multimodal model with a 1M token context window, supporting text, image and video input, deep thinking, function calling and structured output.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.6-pro"},{"api":"chat","id":"tencent/kimi-k2.7-code-highspeed","object":"model","created":1782172800,"updated":1790594486,"owned_by":"system","input_price":0.0000019,"cached_price":3.8e-7,"output_price":0.000008,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000019,"cached_price":3.8e-7,"output_price":0.000008}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.7 Code HighSpeed is the higher-throughput variant of Kimi K2.7 Code, Moonshot's coding model with a 256K context window that follows instructions reliably across long agentic sessions.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.7-code-highspeed"},{"api":"chat","id":"tencent/minimax-m3","object":"model","created":1780245374,"updated":1790594486,"owned_by":"system","input_price":3e-7,"cached_price":6e-8,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":6e-8,"output_price":0.0000012}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"MiniMax-M3 is MiniMax's latest model, achieving frontier capabilities in professional tasks such as programming and agents. It uses the new attention architecture MSA (MiniMax Sparse Attention), supports up to 1M ultra-long context, and is a native multimodal model supporting image and video input.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m3"},{"api":"chat","id":"tencent/mimo-v2.5-pro","object":"model","created":1776816000,"updated":1790594486,"owned_by":"system","input_price":4.35e-7,"cached_price":3.6e-9,"output_price":8.7e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":4.35e-7,"cached_price":3.6e-9,"output_price":8.7e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiMo-V2.5-Pro is Xiaomi's previous-generation flagship multimodal model with a 1M token context window, deep thinking, function calling and structured output.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.5-pro"},{"api":"chat","id":"tencent/hy4-preview","object":"model","created":1787875200,"updated":1790594486,"owned_by":"system","input_price":8.34e-7,"cached_price":4.2e-8,"output_price":0.000002501,"pricing":[{"prompt_tokens_threshold":0,"input_price":8.34e-7,"cached_price":4.2e-8,"output_price":0.000002501}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Hy4 preview is Tencent's flagship class MoE model with 770B total parameters and 49B activated, supporting a 1M token context window. It is optimized for agent, coding, and productivity scenarios, with strong task decomposition, context tracking, instruction following, and long chain execution. It is suited for coding agents, complex tool use, and multi step agentic workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"tencent","model_canonical_name":"hy4-preview"},{"api":"chat","id":"tencent/glm-5.2","object":"model","created":1781631930,"updated":1790594486,"owned_by":"system","input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.2 is Zhipu's latest flagship model, excelling in coding and agent tasks.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"tencent/kimi-k2.7-code","object":"model","created":1781266361,"updated":1790594486,"owned_by":"system","input_price":9.5e-7,"cached_price":1.9e-7,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":9.5e-7,"cached_price":1.9e-7,"output_price":0.000004}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi's most intelligent coding model to date. It follows instructions more reliably across long contexts and achieves higher success rates on coding tasks. Supports text, image, and video inputs, as well as reasoning, chat, and agent workflows.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.7-code"},{"api":"chat","id":"tencent/deepseek-v4.1-flash","object":"model","created":1788998400,"updated":1790594486,"owned_by":"system","input_price":3e-7,"cached_price":6e-9,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":6e-9,"output_price":0.0000012}],"max_output_tokens":393216,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek-V4.1-Flash is DeepSeek's efficient production model with a 1M token context window, configurable thinking mode and strong tool calling for high-concurrency, low-latency workloads.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4.1-flash"},{"api":"chat","id":"tencent/kimi-k3","object":"model","created":1784160000,"updated":1790594486,"owned_by":"system","input_price":0.000003,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"cached_price":3e-7,"output_price":0.000015}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K3 is Moonshot's flagship open-weight 2.8T parameter MoE model with native vision, a 1M token context window and frontier long-horizon coding, knowledge work and reasoning capabilities.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k3"},{"api":"chat","id":"tencent/mimo-v2.6-flash","object":"model","created":1790035200,"updated":1790594486,"owned_by":"system","input_price":1.4e-7,"cached_price":2.8e-9,"output_price":2.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":2.8e-9,"output_price":2.8e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"MiMo-V2.6-Flash is Xiaomi's fast, cost-efficient multimodal model with a 1M token context window, supporting text, image and video input, deep thinking, function calling and structured output.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"xiaomi","model_canonical_name":"mimo-v2.6-flash"},{"api":"chat","id":"tencent/glm-5.3-flash","object":"model","created":1787702400,"updated":1790594486,"owned_by":"system","input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.3-Flash is Zhipu's lightweight multimodal model with a 1M token context window, supporting image, video and file inputs for fast reasoning, daily conversation and content creation.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3-flash"},{"api":"chat","id":"tencent/glm-5.3-flashx","object":"model","created":1789689600,"updated":1790594486,"owned_by":"system","input_price":3.7e-7,"cached_price":7.5e-8,"output_price":0.00000125,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.7e-7,"cached_price":7.5e-8,"output_price":0.00000125}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.3-FlashX is the enhanced, high-speed (~200 tokens/s) variant of GLM-5.3-Flash with a 1M token context window and image, video and file understanding.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"sg","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3-flashx"},{"api":"chat","id":"lyceum/glm-5.3","object":"model","created":1787011200,"updated":1789540005,"owned_by":"system","input_price":0.00000175,"cached_price":4.4e-7,"output_price":0.0000045,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":4.4e-7,"output_price":0.0000045}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"ZAI's newest GLM-5.3 flagship MoE model with strong bilingual reasoning, a 1M token context window, long-context understanding, and tool use. Served by Lyceum in the EU.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3"},{"api":"chat","id":"lyceum/deepseek-v4-pro","object":"model","created":1776988800,"updated":1789540005,"owned_by":"system","input_price":0.00000175,"cached_price":4.4e-7,"output_price":0.0000035,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000175,"cached_price":4.4e-7,"output_price":0.0000035}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Pro is DeepSeek's flagship Mixture of Experts model (1.6T total, 49B active parameters) with a 1M token context window, built for advanced reasoning, coding, and long-horizon agentic workflows with strong tool use. Served by Lyceum in the EU.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro"},{"api":"chat","id":"lyceum/deepseek-v4.1-flash","object":"model","created":1776988800,"updated":1790776917,"owned_by":"system","input_price":5e-7,"cached_price":1.3e-7,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-7,"cached_price":1.3e-7,"output_price":0.0000015}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on output from a 552B-parameter backbone, an asymmetric split that keeps per-token compute low relative to the model's total size. Image understanding is native to the architecture, with visual and text embeddings trained jointly from the start of pre-training rather than added afterward as in the earlier experimental V4 Flash Vision Exp.\n\nIt is suited for coding, terminal, and computer-use agents, along with long-horizon tasks that must run to completion across many steps and long-context analysis. Compressed KV caching cuts cache memory to roughly a quarter of the previous Flash generation, significantly reducing costs on agentic workloads. DeepSeek positions it as the cost-efficient tier of the V4.1 family and reports that it exceeds V4 Pro on performance, speed, and task completion time.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4.1-flash"},{"api":"chat","id":"lyceum/glm-5.3-flash","object":"model","created":1787702400,"updated":1789540005,"owned_by":"system","input_price":2e-7,"cached_price":5e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":5e-8,"output_price":5e-7}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"ZAI's fast GLM-5.3 variant with hybrid reasoning, vision input, a 1M token context window, and tool use at a fraction of the flagship price. Served by Lyceum in the EU.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3-flash"},{"api":"chat","id":"lyceum/kimi-k3","object":"model","created":1784160000,"updated":1789540005,"owned_by":"system","input_price":0.000003,"cached_price":7.5e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"cached_price":7.5e-7,"output_price":0.000015}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Moonshot AI's Kimi K3 - flagship 2.8T MoE reasoning model with a 1M token context window, vision input, long-context understanding, strong tool use, and agentic capabilities. Served by Lyceum in the EU.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k3"},{"api":"chat","id":"lyceum/deepseek-v4-flash-0731","object":"model","created":1785456000,"updated":1789540005,"owned_by":"system","input_price":2.5e-7,"cached_price":6e-8,"output_price":3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.5e-7,"cached_price":6e-8,"output_price":3e-7}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash 0731 is the July 31 update of DeepSeek's efficiency-optimized MoE model (284B total, 13B active parameters) with a 1M token context window, built for fast inference and high-throughput workloads while keeping strong reasoning, coding and tool use. Served by Lyceum in the EU.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0731"},{"api":"chat","id":"moonshot/kimi-k2.6","object":"model","created":1776699402,"updated":1787843415,"owned_by":"system","input_price":9.5e-7,"cached_price":1.6e-7,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":9.5e-7,"cached_price":1.6e-7,"output_price":0.000004}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and can convert prompts and visual inputs into production-ready interfaces. Its agent swarm architecture scales to hundreds of parallel sub-agents for autonomous task decomposition - delivering documents, websites, and spreadsheets in a single run without human oversight.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"int4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.6"},{"api":"chat","id":"moonshot/kimi-k3","object":"model","created":1784215858,"updated":1787843415,"owned_by":"system","input_price":0.000003,"caching_price":0.000015,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"caching_price":0.000015,"cached_price":3e-7,"output_price":0.000015}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K3 is Kimi’s most capable model to date, with 2.8 trillion parameters. Built on Kimi Delta Attention, a hybrid linear attention mechanism, and Attention Residuals, it offers native visual understanding and a 1M-token context window for frontier intelligence scenarios such as software engineering, knowledge work, and deep reasoning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"mxfp4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k3"},{"api":"chat","id":"moonshot/kimi-k2.7-code","object":"model","created":1781266361,"updated":1787843415,"owned_by":"system","input_price":9.5e-7,"cached_price":1.9e-7,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":9.5e-7,"cached_price":1.9e-7,"output_price":0.000004}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts architecture that accepts text and image input, and it always operates in a thinking mode, preserving full reasoning content across multi-turn conversations. With a 256K-token context window, it targets long-horizon coding, agentic task decomposition, and multi-turn dialogue. The model activates 32B parameters out of roughly 1T total.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"int4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.7-code"},{"api":"chat","id":"moonshot/kimi-k2.5","object":"model","created":1769487076,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":1e-7,"output_price":0.000003,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":1e-7,"output_price":0.000003}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed visual and text tokens, it delivers strong performance in general reasoning, visual coding, and agentic tool-calling.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"int4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.5"},{"api":"chat","id":"poolside/laguna-m.1","object":"model","created":1781014014,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":0,"context_window":32768,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Poolside Laguna M.1 — a mid-size code-focused LLM from Poolside, served via an OpenAI-compatible API.","data_retention":true,"data_retention_days":30,"data_used_for_training":true,"privacy_comments":"Governed by Poolside terms of service.","geolocation":"global","open_weights":false,"model_lab":"poolside","model_canonical_name":"laguna-m.1","retires":1789948800},{"api":"chat","id":"poolside/laguna-xs.2","object":"model","created":1781014014,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":0,"context_window":32768,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Poolside Laguna XS.2 — a small, fast code-focused LLM from Poolside, served via an OpenAI-compatible API.","data_retention":true,"data_retention_days":30,"data_used_for_training":true,"privacy_comments":"Governed by Poolside terms of service.","geolocation":"global","open_weights":false,"model_lab":"poolside","model_canonical_name":"laguna-xs.2","retires":1789948800},{"api":"chat","id":"runware/qwen3.5-4b","object":"model","created":1772409600,"updated":1788116823,"owned_by":"system","input_price":5e-8,"cached_price":5e-9,"output_price":7e-8,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"cached_price":5e-9,"output_price":7e-8}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Qwen3.5 4B is a lightweight hybrid-reasoning model from the Qwen3.5 family, optimized for high-throughput, cost-sensitive workloads with a 262K token context window, tool calling, and structured output support. Served via Runware.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"qwen","model_canonical_name":"qwen3.5-4b"},{"api":"chat","id":"runware/gpt-oss-120b","object":"model","created":1754352000,"updated":1788117229,"owned_by":"system","input_price":3.2e-8,"cached_price":3.2e-8,"output_price":1.4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.2e-8,"cached_price":3.2e-8,"output_price":1.4e-7}],"max_output_tokens":131072,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"gpt-oss-120b is OpenAI's open-weight (Apache 2.0) flagship reasoning model, a 117B-parameter MoE with configurable reasoning effort, native tool use, and strong math and coding performance. Served via Runware.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"openai","model_canonical_name":"gpt-oss-120b"},{"api":"chat","id":"runware/glm-5.3","object":"model","created":1787077232,"updated":1788783572,"owned_by":"system","input_price":0.0000012,"cached_price":2e-7,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000012,"cached_price":2e-7,"output_price":0.000004}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves on GLM-5.2 in coding and in the balance between performance and token efficiency.\n\nReasoning is always on and cannot be disabled. Reasoning efforts low, high, and max are supported; max is the default.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3"},{"api":"chat","id":"runware/deepseek-v4.1-flash","object":"model","created":1789487766,"updated":1789487766,"owned_by":"system","input_price":1.5e-7,"cached_price":1e-8,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":1e-8,"output_price":6e-7}],"max_output_tokens":384000,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on output from a 552B-parameter backbone, an asymmetric split that keeps per-token compute low relative to the model's total size. Image understanding is native to the architecture, with visual and text embeddings trained jointly from the start of pre-training rather than added afterward as in the earlier experimental V4 Flash Vision Exp.\n\nIt is suited for coding, terminal, and computer-use agents, along with long-horizon tasks that must run to completion across many steps and long-context analysis. Compressed KV caching cuts cache memory to roughly a quarter of the previous Flash generation, significantly reducing costs on agentic workloads. DeepSeek positions it as the cost-efficient tier of the V4.1 family and reports that it exceeds V4 Pro on performance, speed, and task completion time.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4.1-flash"},{"api":"chat","id":"runware/qwen3.5-9b","object":"model","created":1772409600,"updated":1788116825,"owned_by":"system","input_price":9e-8,"cached_price":9e-9,"output_price":1.3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":9e-8,"cached_price":9e-9,"output_price":1.3e-7}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Qwen3.5 9B is a compact hybrid-reasoning model from the Qwen3.5 family with strong coding, math, and multilingual performance for its size, a 262K token context window, tool calling, and structured output support. Served via Runware.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"qwen","model_canonical_name":"qwen3.5-9b"},{"api":"chat","id":"runware/glm-5.3-flash","object":"model","created":1787875200,"updated":1789539761,"owned_by":"system","input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM 5.3 Flash is Z.ai's lower-cost model from the GLM 5.3 family for coding and long-horizon agent workflows, with reasoning and efficient long-context serving. Image input is not supported on this deployment. Served via Runware.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3-flash"},{"api":"chat","id":"runware/deepseek-v4-flash-0731","object":"model","created":1785456000,"updated":1788117189,"owned_by":"system","input_price":7.6e-8,"cached_price":1.4e-8,"output_price":1.53e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":7.6e-8,"cached_price":1.4e-8,"output_price":1.53e-7}],"max_output_tokens":384000,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash (0731) is DeepSeek's fast, cost-focused frontier model for coding, reasoning, and agent workflows, with thinking and non-thinking modes, a 1M token context window, and up to 384K output tokens. Served via Runware.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0731"},{"api":"chat","id":"runware/glm-5.2","object":"model","created":1781568000,"updated":1788117205,"owned_by":"system","input_price":8e-7,"cached_price":1.6e-7,"output_price":0.00000255,"pricing":[{"prompt_tokens_threshold":0,"input_price":8e-7,"cached_price":1.6e-7,"output_price":0.00000255}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM 5.2 is Z.ai's flagship model for long-horizon coding, agentic engineering, and sustained multi-step execution, with a 1M token context window, 128K max output, multiple thinking modes, function calling, and structured output. Released under the MIT license. Served via Runware.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"nvidia/muse-glimmer-30b","object":"model","created":1786302394,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":20480,"context_window":131072,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Meta Muse Glimmer 30B is a 30B parameter multimodal model from Meta accepting text and image input and returning text output, with a 131K token context window. Supports chain of thought reasoning and function calling, suited to multimodal question answering, document and image understanding, and agentic workflows. Open weights.","data_retention":true,"data_retention_days":30,"data_used_for_training":true,"privacy_comments":"NVIDIA API Trial Terms of Service apply.","geolocation":"global","open_weights":true,"model_lab":"meta","model_canonical_name":"muse-glimmer-30b"},{"api":"chat","id":"nvidia/nemotron-3.5-content-safety","object":"model","created":1781510650,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":8192,"context_window":131072,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting text and image input and returning text output: a safe/unsafe classification for the user prompt and the response, safety category labels, and an optional reasoning trace. It covers 12 languages with a context window of up to 128K tokens. Suited for prompt and response moderation, content classification, safety pipelines, and enterprise AI guardrails with policy enforcement, and includes a togglable reasoning mode. Part of the NVIDIA Nemotron family of open models for agentic AI.","data_retention":true,"data_retention_days":30,"data_used_for_training":true,"privacy_comments":"NVIDIA API Trial Terms of Service apply.","geolocation":"global","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3.5-content-safety"},{"api":"chat","id":"nvidia/nemotron-3-super-120b-a12b","object":"model","created":1773245239,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid Mamba-Transformer MoE model activating just 12B parameters, with multi-token prediction for high throughput. 1M token context for long-horizon agent coherence, cross-document reasoning, and multi-step planning. Strong on AIME 2025, TerminalBench, and SWE-Bench Verified. Part of the NVIDIA Nemotron family.","data_retention":true,"data_retention_days":30,"data_used_for_training":true,"privacy_comments":"NVIDIA API Trial Terms of Service apply.","geolocation":"global","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3-super-120b-a12b"},{"api":"chat","id":"nvidia/nemotron-3-ultra-550b-a55b","object":"model","created":1780551208,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3 Ultra is an open frontier reasoning and orchestration model with 55B active parameters out of 550B total (hybrid Transformer-Mamba MoE). Text input/output with up to 1M context, built for long-running agentic workflows: agent orchestration, coding agents, deep research, and complex enterprise tasks. Strong at multi-step reasoning and planning with high-throughput inference. Part of the NVIDIA Nemotron family.","data_retention":true,"data_retention_days":30,"data_used_for_training":true,"privacy_comments":"NVIDIA API Trial Terms of Service apply.","geolocation":"global","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3-ultra-550b-a55b"},{"api":"chat","id":"nvidia/nemotron-3.5-lightning-30b-a3b","object":"model","created":1786471648,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":65536,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3.5 Lightning 30B-A3B is a hybrid Mamba-2 + MoE + Attention model with 30B total and 3B active parameters, pre-trained on over 20T tokens with an NVFP4 recipe and Multi-Token Prediction for fast generation. Up to 1M token context for long-running autonomous agents, sub-agent workhorse deployments, and agentic workflows. Supports reasoning and tool calling. English and coding languages plus Spanish, French, German, Italian, and Japanese. Open weights under the OpenMDW License Agreement v1.1. Part of the NVIDIA Nemotron family.","data_retention":true,"data_retention_days":30,"data_used_for_training":true,"privacy_comments":"NVIDIA API Trial Terms of Service apply.","geolocation":"global","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3.5-lightning-30b-a3b"},{"api":"chat","id":"nvidia/nemotron-3-nano-30b-a3b","object":"model","created":1765731275,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3 Nano 30B-A3B is a small language MoE model offering high compute efficiency and accuracy for building specialized agentic AI systems. Fully open weights, datasets, and recipes. Text in/out, up to 256K context.","data_retention":true,"data_retention_days":30,"data_used_for_training":true,"privacy_comments":"NVIDIA API Trial Terms of Service apply.","geolocation":"global","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3-nano-30b-a3b"},{"api":"chat","id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning","object":"model","created":1781014014,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":20480,"context_window":131072,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"NVIDIA Nemotron 3 Nano Omni 30B-A3B (reasoning) is a multimodal MoE model unifying video, audio, image, and text understanding for enterprise Q\u0026A, summarization, transcription, OCR, GUI automation, and document intelligence. Supports chain-of-thought reasoning, tool calling, and up to 256k context.","data_retention":true,"data_retention_days":30,"data_used_for_training":true,"privacy_comments":"NVIDIA API Trial Terms of Service apply.","geolocation":"global","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3-nano-omni-30b-a3b-reasoning"},{"api":"chat","id":"fireworks/glm-5.2-fast","object":"model","created":1783970937,"updated":1787843415,"owned_by":"system","input_price":0.0000021,"cached_price":2.1e-7,"output_price":0.0000066,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000021,"cached_price":2.1e-7,"output_price":0.0000066}],"max_output_tokens":131072,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-5.2 introduces a robust 1M-token context and advanced, multi-effort coding capabilities to significantly enhance performance on long-horizon tasks. Its new IndexShare architecture and improved MTP layer simultaneously boost efficiency by reducing per-token FLOPs and increasing speculative decoding lengths. A 743B-parameter model in Zhipu AI's GLM series, designed to plan, execute, and iterate autonomously on extended, engineering-grade tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2-fast"},{"api":"chat","id":"fireworks/nemotron-lightning-3.5-30b-a3b","object":"model","created":1786826050,"updated":1787843415,"owned_by":"system","input_price":5e-8,"cached_price":1e-8,"output_price":2e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"cached_price":1e-8,"output_price":2e-7}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Nemotron-Lightning-3.5-30B-A3B is a 30B-parameter Mixture-of-Experts language model (3B active) from NVIDIA's Nemotron-H family, built on a hybrid Mamba-Transformer architecture for efficient long-context inference. Like other models in the family, it responds to queries by first generating a reasoning trace and then concluding with a final response, with reasoning behavior configurable through a flag in the chat template. It includes a multi-token prediction (MTP) speculative decoding head for low-latency serving.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-lightning-3.5-30b-a3b"},{"api":"chat","id":"fireworks/minimax-m3","object":"model","created":1780245374,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":6e-8,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":6e-8,"output_price":0.0000012}],"max_output_tokens":512000,"context_window":512000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"MiniMax-M3 is a native multimodal model with 512K context running ~428B parameters and ~23B activated parameters. It brings native multimodality, enabling deeper semantic fusion across text, image, and video. M3 also introduces MiniMax Sparse Attention (MSA) to improve long context efficiency, achieving frontier-level performance across long-horizon agentic benchmarks, excelling in both coding and cowork.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m3"},{"api":"chat","id":"fireworks/deepseek-v4-flash-0731","object":"model","created":1785478908,"updated":1788857412,"owned_by":"system","input_price":2.2e-7,"cached_price":7e-9,"output_price":6.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.2e-7,"cached_price":7e-9,"output_price":6.6e-7}],"max_output_tokens":131072,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash 0731 is the official release of DeepSeek V4 Flash, superseding the preview version, with substantially enhanced agentic capabilities. A 304B parameter open-source Mixture-of-Experts model with speculative decoding, 1M token context, optimized for fast, cost-efficient inference with strong reasoning, coding, and function calling performance.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0731"},{"api":"chat","id":"fireworks/muse-glimmer-30b","object":"model","created":1786302394,"updated":1787843415,"owned_by":"system","input_price":3.5e-7,"cached_price":4e-8,"output_price":0.0000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.5e-7,"cached_price":4e-8,"output_price":0.0000015}],"max_output_tokens":131072,"context_window":131072,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Muse Glimmer 30B is a dense causal language model distilled from Muse Spark and purpose-built for autonomous agentic work. It combines multi-step reasoning, reliable schema-based tool calling, and failure recovery with multimodal understanding via a ~1.8B ViT-G/14 perception encoder, supporting interleaved text and image input, a 131K+ context window, and selectable reasoning strength (low through xhigh). Trained on data from over 100 languages, Muse Glimmer performs strongly for its size class on agentic benchmarks including MCP Atlas, DeepSearch QA, Gaia2 and SWE-Bench Pro, and is released under Apache 2.0.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"meta","model_canonical_name":"muse-glimmer-30b"},{"api":"chat","id":"fireworks/kimi-k2.6","object":"model","created":1776699402,"updated":1787843415,"owned_by":"system","input_price":9.5e-7,"cached_price":1.6e-7,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":9.5e-7,"cached_price":1.6e-7,"output_price":0.000004}],"max_output_tokens":32768,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Kimi K2.6 is an open-source, native multimodal agentic model that advances practical capabilities in long-horizon coding, coding-driven design, proactive autonomous execution, and swarm-based task orchestration. It features a 1028B Mixture-of-Experts architecture and supports vision, function calling, and agentic paradigms.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.6"},{"api":"chat","id":"fireworks/gpt-oss-20b","object":"model","created":1754414229,"updated":1787843415,"owned_by":"system","input_price":7e-8,"cached_price":3.5e-8,"output_price":3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":7e-8,"cached_price":3.5e-8,"output_price":3e-7}],"max_output_tokens":65536,"context_window":131072,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GPT-OSS-20B is OpenAI's lightweight open-weight model with 21.5 billion total parameters (3.6 billion active, Mixture-of-Experts). Optimized for lower latency and local or specialized use cases including agentic tasks, chain-of-thought reasoning, and web browsing. Released under Apache 2.0 license.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"openai","model_canonical_name":"gpt-oss-20b"},{"api":"chat","id":"fireworks/kimi-k2.7-code","object":"model","created":1781266361,"updated":1787843415,"owned_by":"system","input_price":9.5e-7,"cached_price":1.9e-7,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":9.5e-7,"cached_price":1.9e-7,"output_price":0.000004}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Kimi K2.7 Code is a coding-focused agentic model built upon Kimi K2.6. With substantial improvements on real-world long-horizon coding tasks, it strengthens end-to-end task completion across complex software engineering workflows while improving token efficiency, reducing thinking-token usage by approximately 30% compared with Kimi K2.6.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2.7-code"},{"api":"chat","id":"fireworks/glm-5.2","object":"model","created":1781631930,"updated":1787843415,"owned_by":"system","input_price":0.0000014,"cached_price":1.4e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"cached_price":1.4e-7,"output_price":0.0000044}],"max_output_tokens":131072,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-5.2 introduces a robust 1M-token context and advanced, multi-effort coding capabilities to significantly enhance performance on long-horizon tasks. Its new IndexShare architecture and improved MTP layer simultaneously boost efficiency by reducing per-token FLOPs and increasing speculative decoding lengths. A 743B-parameter model in Zhipu AI's GLM series, designed to plan, execute, and iterate autonomously on extended, engineering-grade tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.2"},{"api":"chat","id":"fireworks/glm-5.1","object":"model","created":1775578025,"updated":1787843415,"owned_by":"system","input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044}],"max_output_tokens":25344,"context_window":202000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM 5.1 is Z.ai's next generation flagship model built for agentic engineering, with stronger coding capabilities and sustained performance over long horizon tasks with hundreds of iteration rounds. It's a 754B parameter MoE model.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.1"},{"api":"chat","id":"fireworks/nemotron-3-ultra-nvfp4","object":"model","created":1781342122,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":1.2e-7,"output_price":0.0000024,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":1.2e-7,"output_price":0.0000024}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Nemotron-3-Ultra-550B-A55B-NVFP4 is a frontier-scale large language model (LLM) trained by NVIDIA, designed to deliver strong agentic, reasoning, and conversational capabilities. It is optimized for the most demanding workloads, including complex multi-step agents, long-context analysis, and high-accuracy reasoning over code, math, and science. The model employs a hybrid Latent Mixture-of-Experts (LatentMoE) architecture, utilizing interleaved Mamba-2 and MoE layers, along with select Attention layers. Like the Super model, the Ultra model incorporates Multi-Token Prediction (MTP) layers for faster text generation and improved quality, and it is trained using an NVFP4 pre-training recipe to maximize compute efficiency. The model has 55B active parameters and 550B parameters in total.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"nvidia","model_canonical_name":"nemotron-3-ultra-nvfp4"},{"api":"chat","id":"fireworks/glm-5.3-flash","object":"model","created":1787761346,"updated":1788283811,"owned_by":"system","input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.3-Flash is an efficiency optimized model from Z.ai suited for fast coding and long horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long context behavior while reducing compute overhead. Served via Fireworks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3-flash"},{"api":"chat","id":"fireworks/qwen3.7-plus","object":"model","created":1780491783,"updated":1787843415,"owned_by":"system","input_price":4e-7,"cached_price":8e-8,"output_price":0.0000016,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"cached_price":8e-8,"output_price":0.0000016}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its vision-language abilities while retaining full-stack, agent-level intelligence for coding, tool use, and productivity workflows. Its distinguishing trait is multi-modal interactive hybrid agent capability: it can perceive real-world scenes, read screens and interact with GUIs, generate code from visual references, and perform end-to-end navigation within mobile apps.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.7-plus","retires":1788480000},{"api":"chat","id":"fireworks/gpt-oss-120b","object":"model","created":1754414231,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":1.5e-8,"output_price":6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":1.5e-8,"output_price":6e-7}],"max_output_tokens":65536,"context_window":131072,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GPT-OSS-120B is OpenAI's open-weight large language model with 116 billion total parameters (5.1 billion active, Mixture-of-Experts). Designed for high-performance reasoning, agentic tasks, function calling, and general-purpose applications. Supports configurable reasoning depth and chain-of-thought. Released under Apache 2.0 license.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"openai","model_canonical_name":"gpt-oss-120b"},{"api":"chat","id":"fireworks/qwen3.8-max","object":"model","created":1785731612,"updated":1787843415,"owned_by":"system","input_price":0.000002,"cached_price":2.5e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"cached_price":2.5e-7,"output_price":0.000006}],"max_output_tokens":131072,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Built on the architectural foundation of Qwen3.5, Qwen3.8 delivers substantial gains across coding, professional work, research, and long-horizon agentic tasks. Beyond answering harder questions, Qwen3.8 is designed to carry complex, multi-step tasks through to completion with greater reliability.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.8-max"},{"api":"chat","id":"fireworks/deepseek-v4-flash-vision-exp","object":"model","created":1788134400,"updated":1788857986,"owned_by":"system","input_price":2.2e-7,"cached_price":7e-9,"output_price":6.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.2e-7,"cached_price":7e-9,"output_price":6.6e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash Vision Exp is the first experimental multimodal model in the DeepSeek V4 family. It builds on DeepSeek V4 Flash with added visual modules and continued training for visual understanding, with substantially improved multimodal agent capabilities while remaining comparable on text only agent tasks. 305B MoE, 1M token context. Served via Fireworks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"us","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-vision-exp"},{"api":"chat","id":"fireworks/kimi-k3","object":"model","created":1784215858,"updated":1787843415,"owned_by":"system","input_price":0.000003,"cached_price":3e-7,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"cached_price":3e-7,"output_price":0.000015}],"max_output_tokens":131072,"context_window":1000000,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Kimi’s most capable flagship model to date, with 2.8 trillion parameters. It is built on Kimi Delta Attention (KDA), a hybrid linear attention mechanism, and Attention Residuals, with native visual understanding and a 1M-token context window. It is the world’s first open-source model in the 3-trillion-parameter class, designed for frontier intelligence scenarios including long-horizon coding, knowledge work, and reasoning.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp4","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k3"},{"api":"chat","id":"fireworks/deepseek-v4-pro-0813","object":"model","created":1786549364,"updated":1787843415,"owned_by":"system","input_price":0.00000132,"cached_price":4.4e-8,"output_price":0.00000396,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000132,"cached_price":4.4e-8,"output_price":0.00000396}],"max_output_tokens":131072,"context_window":1000000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek-V4-Pro-0813 is the official release of DeepSeek-V4-Pro, superseding the preview version, with greatly enhanced agentic capabilities and performance improvements that are especially pronounced in production environments. It is built on the DeepSeek-V4-Pro (Preview) model structure, with a DSpark speculative decoding module attached.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-pro-0813"},{"api":"chat","id":"fireworks/deepseek-v4.1-flash","object":"model","created":1789116795,"updated":1789116795,"owned_by":"system","input_price":2.2e-7,"cached_price":7e-9,"output_price":6.6e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.2e-7,"cached_price":7e-9,"output_price":6.6e-7}],"max_output_tokens":393216,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4.1 Flash is a 552B parameter Mixture of Experts model built on a new Causal Encoder Decoder architecture that activates only 8B parameters for input and 16B for output, giving frontier level intelligence at a fraction of the cost. It is the smallest model in the new DeepSeek architecture family and the first Flash with native vision and multimodal understanding. V4.1 Flash beats DeepSeek V4 Pro on performance, cost, speed and task completion time and scores 88.1 on CyberGym. Its KV cache is compressed to 890 bytes per token, four times smaller than V4 Flash and over 400 times smaller than V1, cutting the cache hit costs that dominate agent workloads. 1M token context window, 384K max output, thinking and non thinking modes, tool calling, structured JSON output and prompt caching. Served on Fireworks at $0.22 input, $0.007 cache read and $0.66 output per 1M tokens. Open weights are published on Hugging Face. Ideal for coding agents, long context RAG, high throughput batch processing and multimodal assistants.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4.1-flash"},{"api":"chat","id":"fireworks/glm-5.3","object":"model","created":1787077232,"updated":1788026445,"owned_by":"system","input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.0000014,"cached_price":2.6e-7,"output_price":0.0000044}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM 5.3 is the latest Z.ai flagship model for long horizon coding, reasoning, and agentic workflows. It builds on GLM 5.2 with a 1M token context window, stronger engineering task performance, and multiple thinking effort levels. Released under the MIT license. Served via Fireworks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3"},{"api":"chat","id":"novita/sao10k/l3-8b-stheno-v3.2","object":"model","created":1734535928,"updated":1787843415,"owned_by":"system","input_price":5e-8,"cached_price":5e-8,"output_price":5e-8,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"cached_price":5e-8,"output_price":5e-8}],"max_output_tokens":0,"context_window":8192,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Sao10K/L3-8B-Stheno-v3.2 is a highly skilled actor that excels at fully immersing itself in any role assigned.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"sao10k","model_canonical_name":"l3-8b-stheno-v3.2"},{"api":"chat","id":"novita/minimax/minimax-m2.7","object":"model","created":1773836697,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":6e-8,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":6e-8,"output_price":0.0000012}],"max_output_tokens":128000,"context_window":200000,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent collaboration, enabling it to plan, execute, and refine complex tasks across dynamic environments.\n\nTrained for production-grade performance, M2.7 handles workflows such as live debugging, root cause analysis, financial modeling, and full document generation across Word, Excel, and PowerPoint. It delivers strong results on benchmarks including 56.2% on SWE-Pro and 57.0% on Terminal Bench 2, while achieving a 1495 ELO on GDPval-AA, setting a new standard for multi-agent systems operating in real-world digital workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.7"},{"api":"chat","id":"novita/meta-llama/llama-3.3-70b-instruct","object":"model","created":1733506137,"updated":1787843415,"owned_by":"system","input_price":3.9e-7,"cached_price":3.9e-7,"output_price":3.9e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.9e-7,"cached_price":3.9e-7,"output_price":3.9e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model is optimized for multilingual dialogue use cases and outperforms many of the available open source and closed chat models on common industry benchmarks.\n\nSupported languages: English, German, French, Italian, Portuguese, Hindi, Spanish, and Thai.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"meta","model_canonical_name":"llama-3.3-70b-instruct"},{"api":"chat","id":"novita/minimax/minimax-m2.7-highspeed","object":"model","created":1773836697,"updated":1787843415,"owned_by":"system","input_price":6e-7,"output_price":0.0000024,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"output_price":0.0000024}],"max_output_tokens":0,"context_window":204800,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"MiniMax M2.7-highspeed is an accelerated SOTA model engineered for scenarios demanding extreme efficiency. It perfectly inherits the core intelligence and robust digital workspace capabilities of the standard M2.7. In real-world software engineering, M2.7 excels by independently driving end-to-end project delivery while efficiently handling advanced tasks such as log analysis, bug troubleshooting, code security, and machine learning. In the professional workspace, it boasts the highest open-sour","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"minimax","model_canonical_name":"minimax-m2.7-highspeed"},{"api":"chat","id":"novita/sao10k/l31-70b-euryale-v2.2","object":"model","created":1724803200,"updated":1787843415,"owned_by":"system","input_price":0.00000148,"cached_price":0.00000148,"output_price":0.00000148,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000148,"cached_price":0.00000148,"output_price":0.00000148}],"max_output_tokens":0,"context_window":16000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from Sao10k. It is the successor of Euryale L3 70B v2.1.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"sao10k","model_canonical_name":"l31-70b-euryale-v2.2"},{"api":"chat","id":"novita/stepfun/step-3.7-flash","object":"model","created":1779985069,"updated":1787843415,"owned_by":"system","input_price":2e-7,"cached_price":4e-8,"output_price":0.00000115,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":4e-8,"output_price":0.00000115}],"max_output_tokens":256000,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Step 3.7 Flash is a fast reasoning model from StepFun with a 256K context window, supporting function calling and structured output for high volume agentic workloads.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"stepfun","model_canonical_name":"step-3.7-flash"},{"api":"chat","id":"novita/deepseek/deepseek_v3","object":"model","created":1735241320,"updated":1787843415,"owned_by":"system","input_price":8.9e-7,"cached_price":8.9e-7,"output_price":8.9e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":8.9e-7,"cached_price":8.9e-7,"output_price":8.9e-7}],"max_output_tokens":0,"context_window":64000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations reveal that the model outperforms other open-source models and rivals leading closed-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek_v3"},{"api":"chat","id":"novita/gryphe/mythomax-l2-13b","object":"model","created":1688256000,"updated":1787843415,"owned_by":"system","input_price":9e-8,"cached_price":9e-8,"output_price":9e-8,"pricing":[{"prompt_tokens_threshold":0,"input_price":9e-8,"cached_price":9e-8,"output_price":9e-8}],"max_output_tokens":0,"context_window":4096,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"The idea behind this merge is that each layer is composed of several tensors, which are in turn responsible for specific functions. Using MythoLogic-L2's robust understanding as its input and Huginn's extensive writing capability as its output seems to have resulted in a model that exceeds at both, confirming my theory. (More details to be released at a later time).","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp16","open_weights":true,"model_lab":"gryphe","model_canonical_name":"mythomax-l2-13b"},{"api":"chat","id":"novita/meta-llama/llama-3.1-8b-instruct","object":"model","created":1721692800,"updated":1787843415,"owned_by":"system","input_price":5e-8,"cached_price":5e-8,"output_price":5e-8,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"cached_price":5e-8,"output_price":5e-8}],"max_output_tokens":0,"context_window":16384,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Meta's latest class of models, Llama 3.1, launched with a variety of sizes and configurations. The 8B instruct-tuned version is particularly fast and efficient. It has demonstrated strong performance in human evaluations, outperforming several leading closed-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"meta","model_canonical_name":"llama-3.1-8b-instruct"},{"api":"chat","id":"novita/meta-llama/llama-4-maverick-17b-128e-instruct-fp8","object":"model","created":1743881822,"updated":1787843415,"owned_by":"system","input_price":2e-7,"cached_price":2e-7,"output_price":8.5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":2e-7,"output_price":8.5e-7}],"max_output_tokens":1048576,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"meta","model_canonical_name":"llama-4-maverick-17b-128e-instruct"},{"api":"chat","id":"novita/inclusionai/ring-2.6-1t","object":"model","created":1778247440,"updated":1787843415,"owned_by":"system","input_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"output_price":0.0000025}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Inclusion AI ring-2.6-1t","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"inclusionai","model_canonical_name":"ring-2.6-1t"},{"api":"chat","id":"novita/inclusionai/ling-3.0-tiny","object":"model","created":1785941340,"updated":1787843415,"owned_by":"system","input_price":0,"cached_price":0,"output_price":0,"pricing":[{"prompt_tokens_threshold":0,"input_price":0,"cached_price":0,"output_price":0}],"max_output_tokens":32768,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Ling-3.0-tiny is an efficient 7.9B parameter MoE model from inclusionAI with only 1.3B active parameters per token. Built for responsive agents, reliable instruction following and multi turn conversation, with a 256K context window, native function calling, prompt caching and switchable Thinking and Instant modes.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"inclusionai","model_canonical_name":"ling-3.0-tiny"},{"api":"chat","id":"novita/sao10k/l3-8b-lunaris","object":"model","created":1723507200,"updated":1787843415,"owned_by":"system","input_price":5e-8,"cached_price":5e-8,"output_price":5e-8,"pricing":[{"prompt_tokens_threshold":0,"input_price":5e-8,"cached_price":5e-8,"output_price":5e-8}],"max_output_tokens":0,"context_window":8192,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"A generalist / roleplaying model merge based on Llama 3.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"sao10k","model_canonical_name":"l3-8b-lunaris"},{"api":"chat","id":"novita/deepseek/deepseek-v3.2","object":"model","created":1764594642,"updated":1787843415,"owned_by":"system","input_price":2.69e-7,"cached_price":1.345e-7,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2.69e-7,"cached_price":1.345e-7,"output_price":4e-7}],"max_output_tokens":65536,"context_window":163840,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek-V3.2 is a next-generation foundation model designed to unify high computational efficiency with state-of-the-art reasoning and agentic performance. Built upon DeepSeek Sparse Attention (DSA) for efficient long-context reasoning, a scalable reinforcement learning framework reaching frontier-level performance, and a large-scale agentic task synthesis pipeline for reliable tool-use and multi-step decision-making.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v3.2"},{"api":"chat","id":"novita/glm-5.3-flash","object":"model","created":1787702400,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":3e-8,"output_price":5e-7}],"max_output_tokens":131072,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"GLM-5.3-Flash is a native multimodal model from Z.ai, suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while reducing compute overhead. Accepts text, image and video input.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.3-flash"},{"api":"chat","id":"novita/zai-org/glm-5.1","object":"model","created":1775578025,"updated":1787843415,"owned_by":"system","input_price":0.00000138,"output_price":0.0000044,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000138,"output_price":0.0000044}],"max_output_tokens":0,"context_window":204800,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-5.1 is Z.AI's latest flagship model, meticulously engineered for long-horizon autonomous tasks. Capable of working continuously on a single assignment for up to 8 hours, it autonomously manages the entire workflow—from initial planning and execution to iterative optimization and the delivery of production-grade results. While its general capabilities and coding proficiency are on par with Claude Opus 4.6, GLM-5.1 excels particularly in sustained execution, complex engineering optimization, a","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5.1"},{"api":"chat","id":"novita/deepseek/deepseek-r1","object":"model","created":1737381095,"updated":1787843415,"owned_by":"system","input_price":0.000004,"cached_price":0.000004,"output_price":0.000004,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000004,"cached_price":0.000004,"output_price":0.000004}],"max_output_tokens":0,"context_window":64000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek R1 is the latest open-source model released by the DeepSeek team, featuring impressive reasoning capabilities, particularly achieving performance comparable to OpenAI's o1 model in mathematics, coding, and reasoning tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-r1"},{"api":"chat","id":"novita/qwen/qwen3-235b-a22b-fp8","object":"model","created":1745875757,"updated":1787843415,"owned_by":"system","input_price":2e-7,"cached_price":2e-7,"output_price":8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":2e-7,"cached_price":2e-7,"output_price":8e-7}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-235b-a22b"},{"api":"chat","id":"novita/qwen/qwen-2.5-72b-instruct","object":"model","created":1726704000,"updated":1787843415,"owned_by":"system","input_price":3.8e-7,"cached_price":3.8e-7,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":3.8e-7,"cached_price":3.8e-7,"output_price":4e-7}],"max_output_tokens":0,"context_window":32000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Qwen2.5 is the latest series of Qwen large language models. For Qwen2.5, we release a number of base language models and instruction-tuned language models ranging from 0.5 to 72 billion parameters.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen-2.5-72b-instruct"},{"api":"chat","id":"novita/deepseek/deepseek-v3-0324","object":"model","created":1742824755,"updated":1787843415,"owned_by":"system","input_price":4e-7,"cached_price":4e-7,"output_price":0.0000013,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"cached_price":4e-7,"output_price":0.0000013}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek R1 is the latest open-source model released by the DeepSeek team, featuring impressive reasoning capabilities, particularly achieving performance comparable to OpenAI's o1 model in mathematics, coding, and reasoning tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v3-0324"},{"api":"chat","id":"novita/zai-org/glm-4.6","object":"model","created":1759235576,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":6e-7,"output_price":0.0000022,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":6e-7,"output_price":0.0000022}],"max_output_tokens":131072,"context_window":204800,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-4.6 is Z AI’s latest flagship model, designed to push agentic and coding performance further. It expands the context window from 128K to 200K tokens, improves reasoning and tool-use capabilities, and delivers stronger results in coding benchmarks and real-world development workflows. GLM-4.6 demonstrates refined writing quality, more capable agent behavior, and higher token efficiency (≈15% fewer tokens vs. GLM-4.5).\n\nEvaluations show clear gains over GLM-4.5 across reasoning, agents, and coding, reaching near parity with Claude Sonnet 4 in practical tasks while outperforming other open-source baselines. GLM-4.6 is available through the Z.ai API platform, OpenRouter, coding agents (Claude Code, Roo Code, Cline, Kilo Code), and soon as downloadable weights on HuggingFace and ModelScope.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-4.6"},{"api":"chat","id":"novita/google/gemma-4-26b-a4b-it","object":"model","created":1775227989,"updated":1787843415,"owned_by":"system","input_price":1.3e-7,"cached_price":1.3e-7,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.3e-7,"cached_price":1.3e-7,"output_price":4e-7}],"max_output_tokens":131072,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Gemma 4 26B A4B is a Mixture of Experts model from Google DeepMind built for scalable performance with a 256K context window. Supports text and image input, native thinking mode, function calling, structured output, and 140+ languages. Ideal for long context RAG, agentic workflows, coding, and multimodal understanding.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"google","model_canonical_name":"gemma-4-26b-a4b-it"},{"api":"chat","id":"novita/deepseek/deepseek-v4-flash-0731","object":"model","created":1785478908,"updated":1787843415,"owned_by":"system","input_price":1.4e-7,"cached_price":2.8e-8,"output_price":2.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":2.8e-8,"output_price":2.8e-7}],"max_output_tokens":393216,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4 Flash 0731 is the official release of DeepSeek V4 Flash, superseding the preview version, with substantially enhanced agentic capabilities. A sparse MoE model with 13B active parameters out of 284B total, 1M token context window, suited for coding, reasoning, and agent workflows with function calling and structured output support.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0731"},{"api":"chat","id":"novita/deepseek-v4.1-flash","object":"model","created":1789037733,"updated":1789037733,"owned_by":"system","input_price":3e-7,"cached_price":6e-9,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":6e-9,"output_price":0.0000012}],"max_output_tokens":393216,"context_window":1048576,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek V4.1 Flash is a 552B parameter Mixture of Experts model built on a new Causal Encoder Decoder architecture that activates only 8B parameters for input and 16B for output, giving frontier level intelligence at a fraction of the cost. It is the smallest model in the new DeepSeek architecture family and the first Flash with native vision and multimodal understanding. V4.1 Flash beats DeepSeek V4 Pro on performance, cost, speed and task completion time and scores 88.1 on CyberGym. Its KV cache is compressed to 890 bytes per token, four times smaller than V4 Flash and over 400 times smaller than V1, cutting the cache hit costs that dominate agent workloads. 1M token context window, 384K max output, thinking and non thinking modes, tool calling, structured JSON output and prompt caching. Served on Novita at $0.30 input, $0.006 cache read and $1.20 output per 1M tokens. Open weights are published on Hugging Face. Ideal for coding agents, long context RAG, high throughput batch processing and multimodal assistants.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4.1-flash"},{"api":"chat","id":"novita/deepseek/deepseek-r1-distill-llama-70b","object":"model","created":1737663169,"updated":1787843415,"owned_by":"system","input_price":8e-7,"cached_price":8e-7,"output_price":8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":8e-7,"cached_price":8e-7,"output_price":8e-7}],"max_output_tokens":0,"context_window":32000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek R1 Distill LLama 70B","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-r1-distill-llama-70b"},{"api":"chat","id":"novita/deepseek/deepseek-v4-flash-0424","object":"model","created":1777000666,"updated":1787843415,"owned_by":"system","input_price":1.4e-7,"output_price":2.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"output_price":2.8e-7}],"max_output_tokens":0,"context_window":1048576,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek-V4-Flash is a lightweight model meticulously designed by DeepSeek to deliver the ultimate combination of lightning-fast response times and unmatched cost-effectiveness. Engineered with fewer parameters and significantly lower activation overhead, V4-Flash provides an exceptionally fast and economical API service. At its core, V4-Flash demonstrates outstanding reasoning capabilities that closely rival the V4-Pro model. While featuring a slightly streamlined repository of world knowledge,","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0424"},{"api":"chat","id":"novita/kwaipilot/kat-coder-pro","object":"model","created":1774649310,"updated":1787843415,"owned_by":"system","input_price":3e-7,"output_price":0.0000012,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"output_price":0.0000012}],"max_output_tokens":0,"context_window":256000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"KAT-Coder-Pro V2 by KwaiKAT is a non-reasoning model optimized for agentic coding. It delivers strong performance on reasoning-style tasks while requiring significantly fewer output tokens than peer models. With the 1210 release, it achieved a score of 64 on the Artificial Analysis Intelligence Index, placing it in the global Top 10 and ranking first among all non-reasoning models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"kwaipilot","model_canonical_name":"kat-coder-pro"},{"api":"chat","id":"novita/inclusionai/ling-2.6-1t","object":"model","created":1776948238,"updated":1787843415,"owned_by":"system","input_price":3e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"output_price":0.0000025}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Inclusion AI ling-2.6-1t","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"inclusionai","model_canonical_name":"ling-2.6-1t"},{"api":"chat","id":"novita/qwen/qwen3.8-max","object":"model","created":1785731612,"updated":1787843415,"owned_by":"system","input_price":0.000002,"cached_price":2.5e-7,"output_price":0.000006,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"cached_price":2.5e-7,"output_price":0.000006}],"max_output_tokens":131072,"context_window":1000000,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"A 2.4T-parameter MoE flagship for coding and knowledge work. Programs autonomously for days to deliver complete projects end to end, with native visual understanding across planning, execution, and verification, plus deep semantic analysis of ultra-long documents and video.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"alibaba","model_canonical_name":"qwen3.8-max"},{"api":"chat","id":"novita/qwen/qwen3.5-397b-a17b","object":"model","created":1771223018,"updated":1787843415,"owned_by":"system","input_price":6e-7,"cached_price":6e-7,"output_price":0.0000036,"pricing":[{"prompt_tokens_threshold":0,"input_price":6e-7,"cached_price":6e-7,"output_price":0.0000036}],"max_output_tokens":65536,"context_window":262144,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"The Qwen3.5 series 397B-A17B native vision-language model is based on a hybrid architecture design that integrates linear attention mechanisms with sparse Mixture-of-Experts (MoE), achieving higher inference efficiency. Across a variety of tasks—including language understanding, logical reasoning, code generation, agentic tasks, image understanding, video understanding, and graphical user interface (GUI) interaction—it demonstrates exceptional performance comparable to current top-tier frontier models. Possessing robust code generation and agentic capabilities, it exhibits strong generalization across various agent scenarios.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3.5-397b-a17b"},{"api":"chat","id":"novita/moonshotai/kimi-k2-instruct","object":"model","created":1752263252,"updated":1787843415,"owned_by":"system","input_price":5.7e-7,"cached_price":5.7e-7,"output_price":0.0000023,"pricing":[{"prompt_tokens_threshold":0,"input_price":5.7e-7,"cached_price":5.7e-7,"output_price":0.0000023}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Kimi K2 Instruct","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"moonshot","model_canonical_name":"kimi-k2-instruct"},{"api":"chat","id":"novita/mistralai/mistral-nemo","object":"model","created":1721347200,"updated":1787843415,"owned_by":"system","input_price":1.7e-7,"cached_price":1.7e-7,"output_price":1.7e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.7e-7,"cached_price":1.7e-7,"output_price":1.7e-7}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese, Korean, Arabic, and Hindi. It supports function calling and is released under the Apache 2.0 license.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"mistral","model_canonical_name":"mistral-nemo"},{"api":"chat","id":"novita/tencent/hy3","object":"model","created":1783344048,"updated":1787843415,"owned_by":"system","input_price":1.4e-7,"cached_price":3.5e-8,"output_price":5.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":3.5e-8,"output_price":5.8e-7}],"max_output_tokens":262144,"context_window":262144,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Hy3 is built for real world business scenarios with a 295B total, 21B active MoE architecture, native 256K context support, and three reasoning modes. It enhances coding, long form comprehension, multi turn dialogue, and agentic task execution, balancing reliability, efficiency, and cost across high frequency interactions and complex workflows.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"tencent","model_canonical_name":"hy3"},{"api":"chat","id":"novita/deepseek/deepseek-v3-turbo","object":"model","created":1735241320,"updated":1787843415,"owned_by":"system","input_price":4e-7,"cached_price":4e-7,"output_price":0.0000013,"pricing":[{"prompt_tokens_threshold":0,"input_price":4e-7,"cached_price":4e-7,"output_price":0.0000013}],"max_output_tokens":0,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek R1 is the latest open-source model released by the DeepSeek team, featuring impressive reasoning capabilities, particularly achieving performance comparable to OpenAI's o1 model in mathematics, coding, and reasoning tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v3-turbo"},{"api":"chat","id":"novita/deepseek/deepseek-r1-turbo","object":"model","created":1737381095,"updated":1787843415,"owned_by":"system","input_price":7e-7,"cached_price":7e-7,"output_price":0.0000025,"pricing":[{"prompt_tokens_threshold":0,"input_price":7e-7,"cached_price":7e-7,"output_price":0.0000025}],"max_output_tokens":0,"context_window":64000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"DeepSeek R1 is the latest open-source model released by the DeepSeek team, featuring impressive reasoning capabilities, particularly achieving performance comparable to OpenAI's o1 model in mathematics, coding, and reasoning tasks.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-r1-turbo"},{"api":"chat","id":"novita/baichuan/baichuan-m2-32b","object":"model","created":1754870400,"updated":1787843415,"owned_by":"system","input_price":7e-8,"output_price":7e-8,"pricing":[{"prompt_tokens_threshold":0,"input_price":7e-8,"output_price":7e-8}],"max_output_tokens":0,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Baichuan-M2 is a medically-enhanced reasoning model specifically designed for real-world medical reasoning tasks. We begin with real-world medical questions and conduct reinforcement learning training based on a large-scale verifier system. While maintaining the model's general capabilities, the medical effectiveness of Baichuan-M2 has achieved breakthrough improvements.  Baichuan-M2 is currently the world's best open-source medical model. On the HealthBench Benchmark, it surpasses all open-sour","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"baichuan","model_canonical_name":"baichuan-m2-32b"},{"api":"chat","id":"novita/inclusionai/ling-2.6-flash","object":"model","created":1776795886,"updated":1787843415,"owned_by":"system","input_price":1e-7,"output_price":3e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1e-7,"output_price":3e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Inclusion AI ling-2.6-flash","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"inclusionai","model_canonical_name":"ling-2.6-flash"},{"api":"chat","id":"novita/GLM-5","object":"model","created":1770829182,"updated":1787843415,"owned_by":"system","input_price":0.000001,"cached_price":2e-7,"output_price":0.0000032,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"cached_price":2e-7,"output_price":0.0000032}],"max_output_tokens":131072,"context_window":202800,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"GLM-5 is an open-source foundation model engineered for complex system engineering and long-horizon Agent tasks, delivering reliable productivity for top-tier programmers. Transcending the boundary from \"writing code\" to \"building systems,\" it moves beyond traditional snippet generation to offer senior-architect-level planning and execution capabilities. By rejecting the \"frontend-heavy, logic-light\" approach, GLM-5 demonstrates exceptional reasoning and self-healing abilities in backend refactoring, complex algorithm implementation, and deep debugging—autonomously analyzing logs and iteratively fixing persistent bugs until the system runs. As the first open-source model featuring Opus-class style and system engineering depth, GLM-5 provides extreme logic density alongside the freedom of local deployment and high cost-effectiveness, making it the ideal choice for large-scale backend development and automated Agent construction.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"zai","model_canonical_name":"glm-5"},{"api":"chat","id":"novita/microsoft/wizardlm-2-8x22b","object":"model","created":1713225600,"updated":1787843415,"owned_by":"system","input_price":6.2e-7,"cached_price":6.2e-7,"output_price":6.2e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":6.2e-7,"cached_price":6.2e-7,"output_price":6.2e-7}],"max_output_tokens":0,"context_window":65535,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"microsoft","model_canonical_name":"wizardlm-2-8x22b"},{"api":"chat","id":"perplexity/sonar-reasoning-pro","object":"model","created":1741313308,"updated":1787843415,"owned_by":"system","input_price":0.000002,"cached_price":0.000002,"output_price":0.000008,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000002,"cached_price":0.000002,"output_price":0.000008}],"max_output_tokens":8192,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Premier reasoning offering powered by DeepSeek R1 with Chain of Thought (CoT).","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"perplexity","model_canonical_name":"sonar-reasoning-pro"},{"api":"chat","id":"perplexity/sonar-pro","object":"model","created":1741312423,"updated":1787843415,"owned_by":"system","input_price":0.000003,"cached_price":0.000003,"output_price":0.000015,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000003,"cached_price":0.000003,"output_price":0.000015}],"max_output_tokens":8192,"context_window":204800,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Premier search offering with search grounding, supporting advanced queries and follow-ups.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"perplexity","model_canonical_name":"sonar-pro"},{"api":"chat","id":"perplexity/sonar","object":"model","created":1738013808,"updated":1787843415,"owned_by":"system","input_price":0.000001,"cached_price":0.000001,"output_price":0.000001,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.000001,"cached_price":0.000001,"output_price":0.000001}],"max_output_tokens":8192,"context_window":131072,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":false,"supports_role_developer":false,"supports_web_search":true,"supports_output_json_object":false,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Lightweight offering with search grounding, quicker and cheaper than Sonar Pro.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":false,"model_lab":"perplexity","model_canonical_name":"sonar"},{"api":"chat","id":"parasail/deepseek-v4-flash-0424","object":"model","created":1777000666,"updated":1787843415,"owned_by":"system","input_price":1.4e-7,"cached_price":7e-8,"output_price":2.8e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.4e-7,"cached_price":7e-8,"output_price":2.8e-7}],"max_output_tokens":8192,"context_window":1048576,"supports_caching":true,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"DeepSeek-V4-Flash","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"deepseek","model_canonical_name":"deepseek-v4-flash-0424"},{"api":"chat","id":"parasail/parasail-qwen25-vl-72b-instruct","object":"model","created":1738410311,"updated":1787843415,"owned_by":"system","input_price":7e-7,"cached_price":7e-7,"output_price":7e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":7e-7,"cached_price":7e-7,"output_price":7e-7}],"max_output_tokens":8192,"context_window":32768,"supports_caching":false,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen 2.5 VL 72B Instruct","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"parasail-qwen25-vl-72b-instruct"},{"api":"chat","id":"parasail/google/gemma-4-26B-A4B-it","object":"model","created":1775227989,"updated":1787843415,"owned_by":"system","input_price":1.3e-7,"cached_price":5e-8,"output_price":4e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.3e-7,"cached_price":5e-8,"output_price":4e-7}],"max_output_tokens":0,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemma 4 26B A4B is a Mixture of Experts model from Google DeepMind built for scalable performance with a 256K context window. Supports text and image input, native thinking mode, function calling, and 140+ languages. Ideal for long context RAG, agentic workflows, coding, and multimodal understanding.","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"bf16","open_weights":true,"model_lab":"google","model_canonical_name":"gemma-4-26b-a4b-it"},{"api":"chat","id":"parasail/parasail-qwen3-235b-a22b-instruct-2507","object":"model","created":1753119555,"updated":1787843415,"owned_by":"system","input_price":1.5e-7,"cached_price":1.5e-7,"output_price":8.5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":1.5e-7,"cached_price":1.5e-7,"output_price":8.5e-7}],"max_output_tokens":8192,"context_window":262144,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Qwen 3 235B a22b Instruct","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"alibaba","model_canonical_name":"qwen3-235b-a22b-instruct-2507"},{"api":"chat","id":"parasail/gemma3-27b-it","object":"model","created":1741756359,"updated":1787843415,"owned_by":"system","input_price":3e-7,"cached_price":3e-7,"output_price":5e-7,"pricing":[{"prompt_tokens_threshold":0,"input_price":3e-7,"cached_price":3e-7,"output_price":5e-7}],"max_output_tokens":8192,"context_window":128000,"supports_caching":false,"supports_vision":false,"supports_computer_use":false,"supports_reasoning":false,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":true,"supports_output_json_schema":true,"supports_system_mid_conversation":true,"description":"Gemma 3 1B is the smallest of the new Gemma 3 family. It handles context windows up to 32k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities, including structured outputs and function calling. Note: Gemma 3 1B is not multimodal. For the smallest multimodal Gemma 3 model, please see [Gemma 3 4B](google/gemma-3-4b-it)","data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","quantization":"fp8","open_weights":true,"model_lab":"google","model_canonical_name":"parasail-gemma3-27b-it"},{"api":"chat","id":"thinkingmachines/inkling","object":"model","created":1784191574,"updated":1787843415,"owned_by":"system","input_price":0.00000187,"cached_price":3.74e-7,"output_price":0.00000468,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000187,"cached_price":3.74e-7,"output_price":0.00000468}],"max_output_tokens":32768,"context_window":65536,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Inkling is a large MoE hybrid reasoning model from Thinking Machines with audio and vision input support and a 64K context window.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"thinkingmachines","model_canonical_name":"inkling"},{"api":"chat","id":"thinkingmachines/inkling-256k","object":"model","created":1784191574,"updated":1787843415,"owned_by":"system","input_price":0.00000187,"cached_price":3.74e-7,"output_price":0.00000468,"discount_percentage":50,"pricing":[{"prompt_tokens_threshold":0,"input_price":0.00000187,"cached_price":3.74e-7,"output_price":0.00000468}],"max_output_tokens":32768,"context_window":262144,"supports_caching":true,"supports_vision":true,"supports_computer_use":false,"supports_reasoning":true,"supports_image_generation":false,"supports_tool_calling":true,"supports_role_developer":false,"supports_web_search":false,"supports_output_json_object":false,"supports_output_json_schema":false,"supports_system_mid_conversation":true,"description":"Inkling 256K is the extended context variant of Inkling, a large MoE hybrid reasoning model from Thinking Machines with audio and vision input support and a 256K context window.","data_retention":true,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global","open_weights":true,"model_lab":"thinkingmachines","model_canonical_name":"inkling-256k"},{"api":"embedding","id":"tensorx/qwen/qwen3-embedding-8b","object":"model","created":1784900065,"updated":1786007553,"owned_by":"system","pricing":{"input_text":2e-8},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"azure/openai/text-embedding-3-large@francecentral","object":"model","created":1775148772,"updated":1786007553,"owned_by":"system","pricing":{"input_text":1.3e-7},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"eu"},{"api":"embedding","id":"azure/openai/text-embedding-3-small@francecentral","object":"model","created":1775148772,"updated":1786007553,"owned_by":"system","pricing":{"input_text":2e-8},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"eu"},{"api":"embedding","id":"vertex/google/gemini-embedding-001","object":"model","created":1759252334,"updated":1786007553,"owned_by":"system","pricing":{"input_text":1.5e-7},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"vertex/google/text-embedding-005","object":"model","created":1759252334,"updated":1786007553,"owned_by":"system","pricing":{"input_text":2.5e-8},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"vertex/google/gemini-embedding-001@europe-west1","object":"model","created":1775122296,"updated":1786007553,"owned_by":"system","pricing":{"input_text":1.5e-7},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"vertex/google/gemini-embedding-001@europe-west8","object":"model","created":1775122296,"updated":1786007553,"owned_by":"system","pricing":{"input_text":1.5e-7},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"vertex/google/gemini-embedding-2-preview","object":"model","created":1775148772,"updated":1786007553,"owned_by":"system","pricing":{"input_text":2e-7,"input_audio":0.0000064,"input_image":4.651162791e-7,"input_video":0.0000119696969697,"input_document":4.651162791e-7},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"vertex/google/gemini-embedding-001@europe-central2","object":"model","created":1775122296,"updated":1786007553,"owned_by":"system","pricing":{"input_text":1.5e-7},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"vertex/google/gemini-embedding-001@europe-north1","object":"model","created":1775122296,"updated":1786007553,"owned_by":"system","pricing":{"input_text":1.5e-7},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"vertex/google/gemini-embedding-001@europe-west4","object":"model","created":1775122296,"updated":1786007553,"owned_by":"system","pricing":{"input_text":1.5e-7},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"vertex/google/text-multilingual-embedding-002","object":"model","created":1759252334,"updated":1786007553,"owned_by":"system","pricing":{"input_text":2.5e-8},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"openai/text-embedding-3-large","object":"model","created":1759252334,"updated":1786007553,"owned_by":"system","pricing":{"input_text":1.3e-7},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"openai/text-embedding-3-small","object":"model","created":1759252334,"updated":1786007553,"owned_by":"system","pricing":{"input_text":2e-8},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"embedding","id":"nebius/Qwen/Qwen3-Embedding-8B","object":"model","created":1779231494,"updated":1786007553,"owned_by":"system","pricing":{"input_text":1e-8},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"image","id":"openai/gpt-image-2.5-flare","object":"model","created":1788970906,"updated":1788971247,"owned_by":"system","pricing":{"input_text":0.000005,"input_image":0.000008,"output_text":0,"output_image":0.00003},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"image","id":"openai/gpt-image-2.5-sunburst","object":"model","created":1788970906,"updated":1788971247,"owned_by":"system","pricing":{"input_text":0.000005,"input_image":0.000008,"output_text":0,"output_image":0.00003},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"image","id":"openai/gpt-image-2","object":"model","created":1788970906,"updated":1788971247,"owned_by":"system","pricing":{"input_text":0.000005,"input_image":0.000008,"output_text":0,"output_image":0.00003},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"image","id":"azure/openai/gpt-image-1.5","object":"model","created":1771862400,"updated":1771862400,"owned_by":"system","pricing":{"input_text":0.000005,"input_image":0.000008,"output_text":0.00001,"output_image":0.000032},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"image","id":"azure/mai-image-2.5","object":"model","created":1785233246,"updated":1785233246,"owned_by":"system","pricing":{"input_text":0.000005,"input_image":0.000008,"output_text":0.000108,"output_image":0.000108},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"image","id":"azure/openai/gpt-image-1","object":"model","created":1771862400,"updated":1771862400,"owned_by":"system","pricing":{"input_text":0.000005,"input_image":0.00001,"output_text":0,"output_image":0.00004},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"image","id":"azure/openai/gpt-image-2","object":"model","created":1776868154,"updated":1776868154,"owned_by":"system","pricing":{"input_text":0.000005,"input_image":0.000008,"output_text":0,"output_image":0.00003},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"image","id":"vertex/google/gemini-2.5-flash-image","object":"model","created":1764439884,"updated":1764439884,"owned_by":"system","pricing":{"input_text":3e-7,"input_image":3e-7,"output_text":0.00003,"output_image":0.00003},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"image","id":"vertex/google/gemini-3-pro-image","object":"model","created":1764439884,"updated":1764439884,"owned_by":"system","pricing":{"input_text":0.000002,"input_image":0.000002,"output_text":0.000012,"output_image":0.00012},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"image","id":"vertex/google/gemini-3.1-flash-image","object":"model","created":1773673224,"updated":1773673224,"owned_by":"system","pricing":{"input_text":5e-7,"input_image":5e-7,"output_text":0.000003,"output_image":0.00006},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"transcription","id":"deepinfra/openai/whisper-large-v3","object":"model","created":1790949213,"updated":1790949213,"owned_by":"system","pricing":{"input_second":0.0000075},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"deepinfra/openai/whisper-large-v3-turbo","object":"model","created":1790949213,"updated":1790949213,"owned_by":"system","pricing":{"input_second":0.00000333},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"deepinfra/Qwen/Qwen3-ASR-0.6B","object":"model","created":1790949213,"updated":1790949213,"owned_by":"system","pricing":{"input_second":0.00000333},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"deepinfra/Qwen/Qwen3-ASR-1.7B","object":"model","created":1790949213,"updated":1790949213,"owned_by":"system","pricing":{"input_second":0.0000075},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"deepinfra/mistralai/Voxtral-Small-24B-2507","object":"model","created":1787251546,"updated":1787251546,"owned_by":"system","pricing":{"input_second":0.00005},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"deepinfra/mistralai/Voxtral-Mini-3B-2507","object":"model","created":1787251546,"updated":1787251546,"owned_by":"system","pricing":{"input_second":0.0000166667},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"deepinfra/nvidia/Nemotron-3.5-ASR-Streaming-Multilingual-0.6b","object":"model","created":1787251546,"updated":1787251546,"owned_by":"system","pricing":{"input_second":0.0000033333},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"openai/gpt-4o-mini-transcribe","object":"model","created":1764339659,"updated":1786007616,"owned_by":"system","pricing":{"input_token":0.00000125,"output_token":0.000005},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"openai/gpt-4o-transcribe","object":"model","created":1764339659,"updated":1786007616,"owned_by":"system","pricing":{"input_token":0.0000025,"output_token":0.00001},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"openai/whisper-1","object":"model","created":1764339659,"updated":1786007616,"owned_by":"system","pricing":{"input_second":0.0001},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"openai/gpt-4o-mini-transcribe-2025-03-20","object":"model","created":1765980692,"updated":1786007616,"owned_by":"system","pricing":{"input_token":0.00000125,"output_token":0.000005},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"openai/gpt-4o-mini-transcribe-2025-12-15","object":"model","created":1765980692,"updated":1786007616,"owned_by":"system","pricing":{"input_token":0.00000125,"output_token":0.000005},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"global"},{"api":"transcription","id":"mistral/voxtral-mini-latest","object":"model","created":1777924745,"updated":1786007616,"owned_by":"system","pricing":{"input_second":0.0001},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"N/A","geolocation":"eu"},{"api":"speech","id":"openai/gpt-4o-mini-tts","object":"model","created":1765295676,"updated":1786007615,"owned_by":"system","pricing":{"input_token":0,"output_token":0.000012},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"openai/tts-1","object":"model","created":1765295676,"updated":1786007615,"owned_by":"system","pricing":{"input_character":0.000015},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"openai/tts-1-hd","object":"model","created":1765295676,"updated":1786007615,"owned_by":"system","pricing":{"input_character":0.00003},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"openai/gpt-4o-mini-tts-2025-03-20","object":"model","created":1765980727,"updated":1786007615,"owned_by":"system","pricing":{"input_token":0,"output_token":0.000012},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"openai/gpt-4o-mini-tts-2025-12-15","object":"model","created":1765980727,"updated":1786007615,"owned_by":"system","pricing":{"input_token":0,"output_token":0.000012},"data_retention":true,"data_retention_days":30,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"deepinfra/hexgrad/Kokoro-82M","object":"model","created":1784810119,"updated":1786007615,"owned_by":"system","pricing":{"input_character":6.2e-7},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"deepinfra/sesame/csm-1b","object":"model","created":1784810119,"updated":1786007615,"owned_by":"system","pricing":{"input_character":0.000007},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"deepinfra/canopylabs/orpheus-3b-0.1-ft","object":"model","created":1784810119,"updated":1786007615,"owned_by":"system","pricing":{"input_character":0.000007},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"deepinfra/Qwen/Qwen3-TTS-VoiceDesign","object":"model","created":1784810119,"updated":1786007615,"owned_by":"system","pricing":{"input_character":0.00002},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"deepinfra/ResembleAI/chatterbox-turbo","object":"model","created":1784810119,"updated":1786007615,"owned_by":"system","pricing":{"input_character":0.000001},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"deepinfra/ResembleAI/chatterbox-multilingual","object":"model","created":1784810119,"updated":1786007615,"owned_by":"system","pricing":{"input_character":0.000001},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"},{"api":"speech","id":"deepinfra/XiaomiMiMo/MiMo-V2.5-tts","object":"model","created":1784810119,"updated":1786007615,"owned_by":"system","pricing":{"input_character":0},"data_retention":false,"data_retention_days":0,"data_used_for_training":false,"privacy_comments":"","geolocation":"global"}]}
