{"object":"list","data":[{"id":"zai/glm-5.3","object":"model","created":1787086655,"owned_by":"zai","model_name":"glm-5.3","context_length":1048576,"description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.6e-7},"list_pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.6e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"cloudflare/@cf/qwen/qwen3.8-27b","object":"model","created":1787029753,"owned_by":"cloudflare","model_name":"@cf/qwen/qwen3.8-27b","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4.5000000000000003e-7,"output_cost_per_token":3.2000000000000003e-6,"cache_read_input_token_cost":5.0000000000000004e-8},"list_pricing":{"input_cost_per_token":4.5000000000000003e-7,"output_cost_per_token":3.2000000000000003e-6,"cache_read_input_token_cost":5.0000000000000004e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.8-27B","object":"model","created":1787029753,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.8-27B","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":2.9999999999999997e-6,"cache_read_input_token_cost":4.000000000000001e-8},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":2.9999999999999997e-6,"cache_read_input_token_cost":4.000000000000001e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"tensorx/deepseek/deepseek-r1-0528","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"deepseek/deepseek-r1-0528","context_length":164000,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":2.6e-6,"cache_read_input_token_cost":1.65e-7},"list_pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":2.6e-6,"cache_read_input_token_cost":1.65e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/deepseek/deepseek-v3.2","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"deepseek/deepseek-v3.2","context_length":163840,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":7.5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/deepseek/deepseek-v4-flash-0731","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"deepseek/deepseek-v4-flash-0731","context_length":1048576,"description":"Served by TensorX at fp4 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":3e-7,"cache_read_input_token_cost":6.25e-8},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":3e-7,"cache_read_input_token_cost":6.25e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/deepseek/deepseek-v4-pro","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"deepseek/deepseek-v4-pro","context_length":1048576,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":4.375e-7},"list_pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":4.375e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/minimax/minimax-m2.5","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"minimax/minimax-m2.5","context_length":196608,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":7.5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/minimax/minimax-m3","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"minimax/minimax-m3","context_length":1048576,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":1e-7},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/moonshotai/kimi-k2.5","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"moonshotai/kimi-k2.5","context_length":262144,"description":"Served by TensorX at int4 quantization.","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.8e-6,"cache_read_input_token_cost":1.25e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.8e-6,"cache_read_input_token_cost":1.25e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/moonshotai/kimi-k2.6","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"moonshotai/kimi-k2.6","context_length":262144,"description":"Served by TensorX at int4 quantization.","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":4e-6,"cache_read_input_token_cost":2.5e-7},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":4e-6,"cache_read_input_token_cost":2.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/moonshotai/kimi-k2.7-code","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"moonshotai/kimi-k2.7-code","context_length":262144,"description":"Served by TensorX at int4 quantization.","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":3.125e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":3.125e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/moonshotai/kimi-k3","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"moonshotai/kimi-k3","context_length":1048576,"description":"Served by TensorX at mxfp4 quantization.","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":7.5e-7},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":7.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/qwen/qwen3-235b-a22b-2507","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"qwen/qwen3-235b-a22b-2507","context_length":131000,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":7.2e-8,"output_cost_per_token":4.64e-7,"cache_read_input_token_cost":1.8e-8},"list_pricing":{"input_cost_per_token":7.2e-8,"output_cost_per_token":4.64e-7,"cache_read_input_token_cost":1.8e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/qwen/qwen3.5-122b-a10b","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"qwen/qwen3.5-122b-a10b","context_length":262144,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":1.25e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":1.25e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/qwen/qwen3.5-9b","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"qwen/qwen3.5-9b","context_length":262144,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":2e-7,"cache_read_input_token_cost":3.75e-8},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":2e-7,"cache_read_input_token_cost":3.75e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/qwen/qwen3.8-2.4t-a95b","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"qwen/qwen3.8-2.4t-a95b","context_length":262144,"description":"Served by TensorX at nvfp4 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":6.25e-7},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":6.25e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/qwen/qwen3.8-27b","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"qwen/qwen3.8-27b","context_length":262144,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2.4e-6,"cache_read_input_token_cost":1e-7},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2.4e-6,"cache_read_input_token_cost":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/z-ai/glm-5","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"z-ai/glm-5","context_length":202752,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3.2e-6,"cache_read_input_token_cost":2.5e-7},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3.2e-6,"cache_read_input_token_cost":2.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/z-ai/glm-5.1","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"z-ai/glm-5.1","context_length":202752,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":3.5e-7},"list_pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":3.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/z-ai/glm-5.2","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"z-ai/glm-5.2","context_length":1048576,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":3.75e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":3.75e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/z-ai/glm-5-turbo","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"z-ai/glm-5-turbo","context_length":202752,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":4e-6,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":4e-6,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"tensorx/z-ai/glm-5v-turbo","object":"model","created":1787003349,"owned_by":"tensorx","model_name":"z-ai/glm-5v-turbo","context_length":202752,"description":"Served by TensorX at fp8 quantization.","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":4e-6,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":4e-6,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-V4-Pro-0813","object":"model","created":1786770560,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-V4-Pro-0813","context_length":1048576,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.2999999999999998e-6,"output_cost_per_token":2.5999999999999997e-6,"cache_read_input_token_cost":1.0000000399999998e-7},"list_pricing":{"input_cost_per_token":1.2999999999999998e-6,"output_cost_per_token":2.5999999999999997e-6,"cache_read_input_token_cost":1.0000000399999998e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/google/gemini-3.7-flash","object":"model","created":1786770560,"owned_by":"deepinfra","model_name":"google/gemini-3.7-flash","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":7.499999999999999e-7,"output_cost_per_token":3.75e-6},"list_pricing":{"input_cost_per_token":7.499999999999999e-7,"output_cost_per_token":3.75e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.8-27b","object":"model","created":1786722910,"owned_by":"qwen","model_name":"qwen3.8-27b","context_length":1000000,"description":"The Qwen3.8 27B native vision-language dense model builds upon the 3.6-27B version, with key improvements in coding and office productivity capabilities across both text and visual modalities. It enables more reliable end-to-end completion of complex tasks, delivering consistently trustworthy results.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.25e-7,"output_cost_per_token":1.95e-6,"cache_read_input_token_cost":6.500000000000001e-8,"cache_creation_input_token_cost":4.0625000000000003e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"cache_read_input_token_cost":1.0000000000000001e-7,"cache_creation_input_token_cost":6.25e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/NVIDIA-Nemotron-3.5-Lightning-30B-A3B","object":"model","created":1786694503,"owned_by":"flexai","model_name":"NVIDIA-Nemotron-3.5-Lightning-30B-A3B","context_length":1048576,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":2e-7},"list_pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/deepseek-v4-flash-0731","object":"model","created":1786684194,"owned_by":"flexai","model_name":"deepseek-v4-flash-0731","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":1.8e-7},"list_pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":1.8e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/DeepSeek-V4-Flash-0731","object":"model","created":1786684194,"owned_by":"flexai","model_name":"DeepSeek-V4-Flash-0731","context_length":786432,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":1.8e-7},"list_pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":1.8e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ionos/meta-llama/Llama-3.3-70B-Instruct","object":"model","created":1786647205,"owned_by":"ionos","model_name":"meta-llama/Llama-3.3-70B-Instruct","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":7.592650781283766e-7,"output_cost_per_token":7.592650781283766e-7},"list_pricing":{"input_cost_per_token":7.592650781283766e-7,"output_cost_per_token":7.592650781283766e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ionos/meta-llama/Meta-Llama-3.1-405B-Instruct-FP8","object":"model","created":1786647205,"owned_by":"ionos","model_name":"meta-llama/Meta-Llama-3.1-405B-Instruct-FP8","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.044175210345629e-6,"output_cost_per_token":2.044175210345629e-6},"list_pricing":{"input_cost_per_token":2.044175210345629e-6,"output_cost_per_token":2.044175210345629e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ionos/meta-llama/Meta-Llama-3.1-8B-Instruct","object":"model","created":1786647205,"owned_by":"ionos","model_name":"meta-llama/Meta-Llama-3.1-8B-Instruct","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":1.7521501802962533e-7},"list_pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":1.7521501802962533e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ionos/mistralai/Mistral-Nemo-Instruct-2407","object":"model","created":1786647205,"owned_by":"ionos","model_name":"mistralai/Mistral-Nemo-Instruct-2407","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":1.7521501802962533e-7},"list_pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":1.7521501802962533e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ionos/mistralai/Mistral-Small-24B-Instruct","object":"model","created":1786647205,"owned_by":"ionos","model_name":"mistralai/Mistral-Small-24B-Instruct","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.1681001201975023e-7,"output_cost_per_token":3.5043003605925066e-7},"list_pricing":{"input_cost_per_token":1.1681001201975023e-7,"output_cost_per_token":3.5043003605925066e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ionos/openai/gpt-oss-120b","object":"model","created":1786647205,"owned_by":"ionos","model_name":"openai/gpt-oss-120b","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":7.592650781283766e-7},"list_pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":7.592650781283766e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ionos/Qwen/Qwen3.5-397B-A17B","object":"model","created":1786647205,"owned_by":"ionos","model_name":"Qwen/Qwen3.5-397B-A17B","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":7.008600721185013e-7,"output_cost_per_token":4.205160432711008e-6},"list_pricing":{"input_cost_per_token":7.008600721185013e-7,"output_cost_per_token":4.205160432711008e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ionos/Qwen/Qwen3.5-9B","object":"model","created":1786647205,"owned_by":"ionos","model_name":"Qwen/Qwen3.5-9B","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.1681001201975023e-7,"output_cost_per_token":1.7521501802962533e-7},"list_pricing":{"input_cost_per_token":1.1681001201975023e-7,"output_cost_per_token":1.7521501802962533e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ionos/Qwen/Qwen3-Coder-Next","object":"model","created":1786647205,"owned_by":"ionos","model_name":"Qwen/Qwen3-Coder-Next","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":9.344800961580018e-7},"list_pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":9.344800961580018e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"google/gemini-3.7-flash","object":"model","created":1786640581,"owned_by":"google","model_name":"gemini-3.7-flash","context_length":1048576,"description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.028,"search_context_size_high":0.028,"search_context_size_medium":0.028},"cache_creation_input_token_cost":8.33333333333334e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.028,"search_context_size_high":0.028,"search_context_size_medium":0.028},"cache_creation_input_token_cost":8.33333333333334e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3.7-flash@eu","object":"model","created":1786640581,"owned_by":"vertex","model_name":"gemini-3.7-flash","context_length":1048576,"description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.056,"search_context_size_high":0.056,"search_context_size_medium":0.056},"cache_creation_input_token_cost":8.33333333333332e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.056,"search_context_size_high":0.056,"search_context_size_medium":0.056},"cache_creation_input_token_cost":8.33333333333332e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3.7-flash","object":"model","created":1786640581,"owned_by":"vertex","model_name":"gemini-3.7-flash","context_length":1048576,"description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.056,"search_context_size_high":0.056,"search_context_size_medium":0.056},"cache_creation_input_token_cost":8.33333333333332e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.056,"search_context_size_high":0.056,"search_context_size_medium":0.056},"cache_creation_input_token_cost":8.33333333333332e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/gemini-3.7-flash@us","object":"model","created":1786640581,"owned_by":"vertex","model_name":"gemini-3.7-flash","context_length":1048576,"description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.056,"search_context_size_high":0.056,"search_context_size_medium":0.056},"cache_creation_input_token_cost":8.33333333333332e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.056,"search_context_size_high":0.056,"search_context_size_medium":0.056},"cache_creation_input_token_cost":8.33333333333332e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"infomaniak/mistralai/Ministral-3-14B-Instruct-2512","object":"model","created":1786631709,"owned_by":"infomaniak","model_name":"mistralai/Ministral-3-14B-Instruct-2512","context_length":100000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.504300360592507e-7,"output_cost_per_token":4.6724004807900097e-7},"list_pricing":{"input_cost_per_token":3.504300360592507e-7,"output_cost_per_token":4.6724004807900097e-7},"discount":null,"regions":[{"code":"ch","name":"Switzerland"},{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/deepseek-v4-flash","object":"model","created":1786613111,"owned_by":"azure","model_name":"deepseek-v4-flash","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.9e-7,"output_cost_per_token":5.1e-7,"cache_read_input_token_cost":2.8e-8},"list_pricing":{"input_cost_per_token":1.9e-7,"output_cost_per_token":5.1e-7,"cache_read_input_token_cost":2.8e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-4.1@eu","object":"model","created":1786613111,"owned_by":"azure","model_name":"gpt-4.1","context_length":1047576,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":8.8e-6,"cache_read_input_token_cost":5.5e-7},"list_pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":8.8e-6,"cache_read_input_token_cost":5.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-4.1","object":"model","created":1786613111,"owned_by":"azure","model_name":"gpt-4.1","context_length":1047576,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-4.1@us","object":"model","created":1786613111,"owned_by":"azure","model_name":"gpt-4.1","context_length":1047576,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":8.8e-6,"cache_read_input_token_cost":5.5e-7},"list_pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":8.8e-6,"cache_read_input_token_cost":5.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-4.1-mini@eu","object":"model","created":1786613111,"owned_by":"azure","model_name":"gpt-4.1-mini","context_length":1047576,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":4.4e-7,"output_cost_per_token":1.76e-6,"cache_read_input_token_cost":1.1e-7},"list_pricing":{"input_cost_per_token":4.4e-7,"output_cost_per_token":1.76e-6,"cache_read_input_token_cost":1.1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-4.1-mini","object":"model","created":1786613111,"owned_by":"azure","model_name":"gpt-4.1-mini","context_length":1047576,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":1.6e-6,"cache_read_input_token_cost":1e-7},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":1.6e-6,"cache_read_input_token_cost":1e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-4.1-mini@us","object":"model","created":1786613111,"owned_by":"azure","model_name":"gpt-4.1-mini","context_length":1047576,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":4.4e-7,"output_cost_per_token":1.76e-6,"cache_read_input_token_cost":1.1e-7},"list_pricing":{"input_cost_per_token":4.4e-7,"output_cost_per_token":1.76e-6,"cache_read_input_token_cost":1.1e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"flexai/gemma-4-26B-A4B-it","object":"model","created":1786610410,"owned_by":"flexai","model_name":"gemma-4-26B-A4B-it","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/gemma-4-31b-it","object":"model","created":1786610410,"owned_by":"flexai","model_name":"gemma-4-31b-it","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":3.4e-7},"list_pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":3.4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/GLM-4.5-Air-FP8","object":"model","created":1786610410,"owned_by":"flexai","model_name":"GLM-4.5-Air-FP8","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.25e-7,"output_cost_per_token":8.5e-7},"list_pricing":{"input_cost_per_token":1.25e-7,"output_cost_per_token":8.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/GLM-5.2","object":"model","created":1786610410,"owned_by":"flexai","model_name":"GLM-5.2","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4.018e-7,"output_cost_per_token":1.2628e-6},"list_pricing":{"input_cost_per_token":4.018e-7,"output_cost_per_token":1.2628e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/gpt-oss-120b","object":"model","created":1786610410,"owned_by":"flexai","model_name":"gpt-oss-120b","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.9e-8,"output_cost_per_token":1e-7},"list_pricing":{"input_cost_per_token":3.9e-8,"output_cost_per_token":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/gpt-oss-20b","object":"model","created":1786610410,"owned_by":"flexai","model_name":"gpt-oss-20b","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-8,"output_cost_per_token":1.3e-7},"list_pricing":{"input_cost_per_token":3e-8,"output_cost_per_token":1.3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/Llama-3.3-70B-Instruct-FP8","object":"model","created":1786610410,"owned_by":"flexai","model_name":"Llama-3.3-70B-Instruct-FP8","context_length":65536,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3.2e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3.2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/Meta-Llama-3.1-8B-Instruct-FP8","object":"model","created":1786610410,"owned_by":"flexai","model_name":"Meta-Llama-3.1-8B-Instruct-FP8","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":3e-8},"list_pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":3e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/MiniMax-M2.7","object":"model","created":1786610410,"owned_by":"flexai","model_name":"MiniMax-M2.7","context_length":204800,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.8e-7,"output_cost_per_token":7.2e-7},"list_pricing":{"input_cost_per_token":1.8e-7,"output_cost_per_token":7.2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/Mistral-Nemo-Instruct-2407-FP8","object":"model","created":1786610410,"owned_by":"flexai","model_name":"Mistral-Nemo-Instruct-2407-FP8","context_length":32768,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.8e-8,"output_cost_per_token":3e-8},"list_pricing":{"input_cost_per_token":1.8e-8,"output_cost_per_token":3e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/Muse-Glimmer-30B","object":"model","created":1786610410,"owned_by":"flexai","model_name":"Muse-Glimmer-30B","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/Nemotron-3-Super-120B-A12B","object":"model","created":1786610410,"owned_by":"flexai","model_name":"Nemotron-3-Super-120B-A12B","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":8.5e-8,"output_cost_per_token":4e-7},"list_pricing":{"input_cost_per_token":8.5e-8,"output_cost_per_token":4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/Qwen3.5-9B","object":"model","created":1786610410,"owned_by":"flexai","model_name":"Qwen3.5-9B","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":1.5e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":1.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/Qwen3.6-27B-FP8","object":"model","created":1786610410,"owned_by":"flexai","model_name":"Qwen3.6-27B-FP8","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.925e-7,"output_cost_per_token":1.755e-6},"list_pricing":{"input_cost_per_token":2.925e-7,"output_cost_per_token":1.755e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"flexai/Qwen3-8B-FP8","object":"model","created":1786610410,"owned_by":"flexai","model_name":"Qwen3-8B-FP8","context_length":40960,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":1e-7},"list_pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.8-2.4T-A95B","object":"model","created":1786551702,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.8-2.4T-A95B","context_length":262144,"description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":5.999999999999999e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":5.999999999999999e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.8-2.4t-a95b","object":"model","created":1786551702,"owned_by":"qwen","model_name":"qwen3.8-2.4t-a95b","context_length":1000000,"description":"Qwen3.8-2.4T-A95B is the open-source release of Qwen's latest flagship, launched August 2026. Its sparse MoE architecture holds 2.4T total parameters with ~95B activated per step, paired with hybrid attention and a 1M token context window. Key benchmarks: GPQA Diamond 92.6, PaperBench 93.0, OSWorld 86.1, BabyVision 82.0. Ranked 4th on CodeArena.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.3e-6,"output_cost_per_token":3.9e-6,"cache_read_input_token_cost":1.625e-7,"cache_creation_input_token_cost":1.6250000000000001e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2.5e-7,"cache_creation_input_token_cost":2.5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/Qwen/Qwen3.8-2.4T-A95B","object":"model","created":1786551702,"owned_by":"together_ai","model_name":"Qwen/Qwen3.8-2.4T-A95B","context_length":1010000,"description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":6.25e-6,"cache_read_input_token_cost":5e-7},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":6.25e-6,"cache_read_input_token_cost":5e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/deepseek-ai/deepseek-v4-pro-0813","object":"model","created":1786549364,"owned_by":"cloudflare","model_name":"@cf/deepseek-ai/deepseek-v4-pro-0813","context_length":1048576,"description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.32e-6,"output_cost_per_token":3.96e-6,"cache_read_input_token_cost":4.4e-8},"list_pricing":{"input_cost_per_token":1.32e-6,"output_cost_per_token":3.96e-6,"cache_read_input_token_cost":4.4e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813","object":"model","created":1786549364,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/deepseek-v4-pro-0813","context_length":1048576,"description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.32e-6,"output_cost_per_token":3.96e-6,"cache_read_input_token_cost":4.4e-8},"list_pricing":{"input_cost_per_token":1.32e-6,"output_cost_per_token":3.96e-6,"cache_read_input_token_cost":4.4e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/deepseek-v4-pro-0813","object":"model","created":1786549364,"owned_by":"qwen","model_name":"deepseek-v4-pro-0813","context_length":1000000,"description":"A flagship Mixture-of-Experts (MoE) large language model with 1.6 trillion total parameters and 49 billion activated parameters. It natively supports context windows of up to 1 million tokens. Trained on extensive high-quality data, the model delivers strong performance in mathematical and logical reasoning, complex reasoning, professional code generation, and in-depth long-document analysis, and is suitable for demanding scenarios such as advanced scientific research, complex enterprise workflows, and sophisticated agentic applications.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4.7190000000000004e-7,"output_cost_per_token":1.4157e-6},"list_pricing":{"input_cost_per_token":7.26e-7,"output_cost_per_token":2.178e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/deepseek-ai/DeepSeek-V4-Pro-0813","object":"model","created":1786549364,"owned_by":"together_ai","model_name":"deepseek-ai/DeepSeek-V4-Pro-0813","context_length":1048576,"description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.32e-6,"output_cost_per_token":3.96e-6,"cache_read_input_token_cost":1.2999999999999997e-7},"list_pricing":{"input_cost_per_token":1.32e-6,"output_cost_per_token":3.96e-6,"cache_read_input_token_cost":1.2999999999999997e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"xai/grok-4.6","object":"model","created":1786548957,"owned_by":"xai","model_name":"grok-4.6","context_length":500000,"description":"Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000012,"cache_read_input_token_cost_above_200k_tokens":1e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000012,"cache_read_input_token_cost_above_200k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning","object":"model","created":1786461139,"owned_by":"deepinfra","model_name":"nvidia/NVIDIA-Nemotron-3.5-Lightning","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":2.0000000000000002e-7,"cache_read_input_token_cost":4e-8},"list_pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":2.0000000000000002e-7,"cache_read_input_token_cost":4e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/nvidia/Nemotron-3_5-Lightning","object":"model","created":1786461139,"owned_by":"nebius","model_name":"nvidia/Nemotron-3_5-Lightning","context_length":1048576,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7,"cache_read_input_token_cost":6e-8},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7,"cache_read_input_token_cost":6e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/deepseek-v4-flash-0731","object":"model","created":1786376637,"owned_by":"scaleway","model_name":"deepseek-v4-flash-0731","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4.6724004807900097e-7,"output_cost_per_token":9.344800961580019e-7},"list_pricing":{"input_cost_per_token":4.6724004807900097e-7,"output_cost_per_token":9.344800961580019e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/meta-models/Muse-Glimmer-30B","object":"model","created":1786302394,"owned_by":"deepinfra","model_name":"meta-models/Muse-Glimmer-30B","context_length":131072,"description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":3.9999999000000004e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":3.9999999000000004e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/muse-glimmer-30b","object":"model","created":1786302394,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/muse-glimmer-30b","context_length":131072,"description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.5e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":4e-8},"list_pricing":{"input_cost_per_token":3.5e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":4e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"together_ai/meta-models/Muse-Glimmer-30B","object":"model","created":1786302394,"owned_by":"together_ai","model_name":"meta-models/Muse-Glimmer-30B","context_length":131072,"description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.5e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":4e-8},"list_pricing":{"input_cost_per_token":3.5e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":4e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"nebius/deepseek-ai/DeepSeek-V4-Flash","object":"model","created":1786165811,"owned_by":"nebius","model_name":"deepseek-ai/DeepSeek-V4-Flash","context_length":8000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4e-7,"output_cost_per_token":2.8e-7,"cache_read_input_token_cost":1.4e-7},"list_pricing":{"input_cost_per_token":1.4e-7,"output_cost_per_token":2.8e-7,"cache_read_input_token_cost":1.4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"google/gemini-robotics-er-1.6-preview","object":"model","created":1785929637,"owned_by":"google","model_name":"gemini-robotics-er-1.6-preview","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"input_cost_per_audio_token":2e-6,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"output_cost_per_reasoning_token":5e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"input_cost_per_audio_token":2e-6,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"output_cost_per_reasoning_token":5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"google/gemini-robotics-er-2-preview","object":"model","created":1785929637,"owned_by":"google","model_name":"gemini-robotics-er-2-preview","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001,"input_cost_per_audio_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"output_cost_per_reasoning_token":0.00001},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001,"input_cost_per_audio_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"output_cost_per_reasoning_token":0.00001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.8-Max","object":"model","created":1785769541,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.8-Max","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.65e-6,"output_cost_per_token":4.9510000000000005e-6,"cache_read_input_token_cost":2.0599999200000002e-7},"list_pricing":{"input_cost_per_token":1.65e-6,"output_cost_per_token":4.9510000000000005e-6,"cache_read_input_token_cost":2.0599999200000002e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.8-max","object":"model","created":1785731612,"owned_by":"qwen","model_name":"qwen3.8-max","context_length":1000000,"description":"2.4-trillion-parameter MoE flagship delivering a comprehensive leap in coding and professional work. Autonomously codes and delivers complete projects spanning 10+ days. Handles hundreds of specialized tasks across legal, financial, design, and other professional domains, producing production-grade results end-to-end in a single conversation. Native visual understanding runs through the full cycle of planning, execution, and verification, enabling deep semantic analysis of ultra-long documents and extended video content. In long-horizon tasks, plans autonomously, iterates through closed feedback loops, and continuously evolves.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3e-6,"output_cost_per_token":3.9e-6,"cache_read_input_token_cost":1.625e-7,"cache_creation_input_token_cost":1.6250000000000001e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2.5e-7,"cache_creation_input_token_cost":2.5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"cloudflare/@cf/deepseek-ai/deepseek-v4-flash-0731","object":"model","created":1785478908,"owned_by":"cloudflare","model_name":"@cf/deepseek-ai/deepseek-v4-flash-0731","context_length":1310720,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4.4e-7,"output_cost_per_token":1.32e-6,"cache_read_input_token_cost":1.4e-8},"list_pricing":{"input_cost_per_token":4.4e-7,"output_cost_per_token":1.32e-6,"cache_read_input_token_cost":1.4e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-V4-Flash-0731","object":"model","created":1785478908,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-V4-Flash-0731","context_length":1048576,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":1.8e-7,"cache_read_input_token_cost":1.6e-8},"list_pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":1.8e-7,"cache_read_input_token_cost":1.6e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-0731","object":"model","created":1785478908,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/deepseek-v4-flash-0731","context_length":1048576,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4e-7,"output_cost_per_token":2.8e-7,"cache_read_input_token_cost":2.8e-8},"list_pricing":{"input_cost_per_token":1.4e-7,"output_cost_per_token":2.8e-7,"cache_read_input_token_cost":2.8e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/deepseek-v4-flash-0731","object":"model","created":1785478908,"owned_by":"qwen","model_name":"deepseek-v4-flash-0731","context_length":1000000,"description":"A highly efficient, lightweight MoE model with 284 billion parameters in total and 13 billion activated parameters, natively supporting context windows of up to one million tokens. It offers fast inference speed, low latency, and cost-effective invocation, delivering well-balanced overall performance. Designed for high-concurrency, lightweight workloads, it is ideally suited for common, essential use cases such as everyday dialogue, content creation, basic RAG applications, and batch text processing.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.573e-7,"output_cost_per_token":4.7190000000000004e-7},"list_pricing":{"input_cost_per_token":2.42e-7,"output_cost_per_token":7.26e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/deepseek-ai/DeepSeek-V4-Flash-0731","object":"model","created":1785478908,"owned_by":"together_ai","model_name":"deepseek-ai/DeepSeek-V4-Flash-0731","context_length":1048576,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3999999999999998e-7,"output_cost_per_token":2.7999999999999997e-7,"cache_read_input_token_cost":3.0000000000000004e-8},"list_pricing":{"input_cost_per_token":1.3999999999999998e-7,"output_cost_per_token":2.7999999999999997e-7,"cache_read_input_token_cost":3.0000000000000004e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/thinkingmachines/Inkling-Small","object":"model","created":1785443117,"owned_by":"deepinfra","model_name":"thinkingmachines/Inkling-Small","context_length":524288,"description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","source":null,"capabilities":{"input_modalities":["text","image","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.5000000000000003e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":9.9999999e-8},"list_pricing":{"input_cost_per_token":4.5000000000000003e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":9.9999999e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"together_ai/thinkingmachines/Inkling-Small","object":"model","created":1785443117,"owned_by":"together_ai","model_name":"thinkingmachines/Inkling-Small","context_length":524288,"description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","source":null,"capabilities":{"input_modalities":["text","image","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"qwen/qwen3-vl-32b-thinking","object":"model","created":1785421029,"owned_by":"qwen","model_name":"qwen3-vl-32b-thinking","context_length":131072,"description":"The largest dense model in the Qwen3-VL series, its reasoning version boasts multimodal reasoning capabilities second only to Qwen3-VL-235B-Thinking. It excels in STEM and math problem-solving, general image and video understanding, and achieves state-of-the-art performance in multimodal agent capabilities, making it ideal for complex multimodal reasoning tasks.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.04e-7,"output_cost_per_token":4.16e-7},"list_pricing":{"input_cost_per_token":1.6e-7,"output_cost_per_token":6.4e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-claude-opus-5","object":"model","created":1785417770,"owned_by":"databricks","model_name":"databricks-claude-opus-5","context_length":null,"description":"Claude Opus 5 is Anthropic's latest Opus-class model, delivering frontier intelligence for complex reasoning, coding, and agentic workflows at Opus-tier pricing. It features an adjustable effort control to trade off thoroughness against latency and cost, and supports large-context tasks such as deep document analysis and multi-step tool use. This model is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.999999970000001e-6,"output_cost_per_token":0.00002499999999},"list_pricing":{"input_cost_per_token":4.999999970000001e-6,"output_cost_per_token":0.00002499999999},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-claude-sonnet-5","object":"model","created":1785417770,"owned_by":"databricks","model_name":"databricks-claude-sonnet-5","context_length":null,"description":"Claude Sonnet 5 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with memory, polished document creation, and confident computer use for web QA and workflow automation. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.9999980000000003e-6,"output_cost_per_token":9.99999e-6},"list_pricing":{"input_cost_per_token":1.9999980000000003e-6,"output_cost_per_token":9.99999e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gemini-3-1-flash-lite","object":"model","created":1785417770,"owned_by":"databricks","model_name":"databricks-gemini-3-1-flash-lite","context_length":1048576,"description":"Gemini 3.1 Flash Lite is Google's fastest and most cost-efficient model in the Gemini 3 series, designed for high-volume and latency-sensitive workloads at scale. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","video","audio","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.4996000000000004e-7,"output_cost_per_token":2.7000400000000002e-6,"input_dbu_cost_per_token":4.464e-6,"output_dbu_cost_per_token":0.000026786},"list_pricing":{"input_cost_per_token":4.4996000000000004e-7,"output_cost_per_token":2.7000400000000002e-6,"input_dbu_cost_per_token":4.464e-6,"output_dbu_cost_per_token":0.000026786},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gemini-3-5-flash","object":"model","created":1785417770,"owned_by":"databricks","model_name":"databricks-gemini-3-5-flash","context_length":null,"description":"Gemini 3.5 Flash is Google's next generation fast and efficient model. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","video","audio","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4999950000000001e-6,"output_cost_per_token":8.99997e-6},"list_pricing":{"input_cost_per_token":1.4999950000000001e-6,"output_cost_per_token":8.99997e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-glm-5-2","object":"model","created":1785417770,"owned_by":"databricks","model_name":"databricks-glm-5-2","context_length":null,"description":"GLM-5.2 is a mixture of experts (MoE) language model with 753B total parameters developed by Zhipu AI. The model supports a context length of 1M tokens and is optimized for long-horizon reasoning, coding, and agentic tool use.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4000000000000001e-6,"output_cost_per_token":4.399997e-6},"list_pricing":{"input_cost_per_token":1.4000000000000001e-6,"output_cost_per_token":4.399997e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-2","object":"model","created":1785417770,"owned_by":"databricks","model_name":"databricks-gpt-5-2","context_length":272000,"description":"GPT-5.2 is a general purpose large language model with reasoning capabilities developed by OpenAI. This model builds directly upon GPT-5.1, offering higher accuracy, improved token efficiency on medium-to-complex tasks, and more deliberate scaffolded reasoning. This model excels at structured extraction, multi-step workflows, and multimodal tasks. It supports multimodal inputs and features a 400K total token context window with 128K maximum output tokens. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.5000000000000004e-6,"output_cost_per_token":0.000014000000000000001,"input_dbu_cost_per_token":0.000025,"output_dbu_cost_per_token":0.0002},"list_pricing":{"input_cost_per_token":3.5000000000000004e-6,"output_cost_per_token":0.000014000000000000001,"input_dbu_cost_per_token":0.000025,"output_dbu_cost_per_token":0.0002},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-3-codex","object":"model","created":1785417770,"owned_by":"databricks","model_name":"databricks-gpt-5-3-codex","context_length":272000,"description":"GPT-5.3 Codex is OpenAI's advanced code-specialized large language model. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.7500000000000002e-6,"output_cost_per_token":0.000014000000000000001,"input_dbu_cost_per_token":0.000025,"output_dbu_cost_per_token":0.0002},"list_pricing":{"input_cost_per_token":1.7500000000000002e-6,"output_cost_per_token":0.000014000000000000001,"input_dbu_cost_per_token":0.000025,"output_dbu_cost_per_token":0.0002},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-5-pro","object":"model","created":1785417770,"owned_by":"databricks","model_name":"databricks-gpt-5-5-pro","context_length":null,"description":"GPT-5.5 Pro is OpenAI's strongest frontier model for agentic work in enterprise, complex document reasoning, and long-horizon coding agents. GPT-5.5 Pro also now powers Codex, OpenAI's coding agent. Pro uses more compute to think harder and provide consistently better answers. This model supports multimodal inputs and features a 400K total token context window with 128K maximum output tokens. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00005999994000000001,"output_cost_per_token":0.00017999940000000002},"list_pricing":{"input_cost_per_token":0.00005999994000000001,"output_cost_per_token":0.00017999940000000002},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-qwen35-122b-a10b","object":"model","created":1785417770,"owned_by":"databricks","model_name":"databricks-qwen35-122b-a10b","context_length":null,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.3999970000000005e-7,"output_cost_per_token":4.399997e-6},"list_pricing":{"input_cost_per_token":4.3999970000000005e-7,"output_cost_per_token":4.399997e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-qwen3-next-80b-a3b-instruct","object":"model","created":1785417770,"owned_by":"databricks","model_name":"databricks-qwen3-next-80b-a3b-instruct","context_length":null,"description":"Qwen3-Next-80B-A3B-Instruct is a highly efficient large language model optimized for instruction-following tasks built and trained by Alibaba Cloud. This model is designed to handle ultra-long contexts and excels at multi-step workflows, retrieval-augmented generation, and enterprise applications that require deterministic outputs at high throughput. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5001e-7,"output_cost_per_token":1.2000100000000002e-6},"list_pricing":{"input_cost_per_token":1.5001e-7,"output_cost_per_token":1.2000100000000002e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-flash-2025-07-28","object":"model","created":1785415829,"owned_by":"qwen","model_name":"qwen-flash-2025-07-28","context_length":1000000,"description":"The Qwen3 Flash model (snapshot 2025-07-28) offers a powerful fusion of thinking and non-thinking modes with dynamic in-conversation switching, excelling in complex reasoning while showing significant gains in instruction following and text comprehension. It supports a 1M context length and is billed on a tiered model corresponding to context usage.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7},{"range":[256000,1000000],"input_cost_per_token":1.625e-7,"output_cost_per_token":1.3e-6}],"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7},{"range":[256000,1000000],"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6}],"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-plus-character-ja","object":"model","created":1785415829,"owned_by":"qwen","model_name":"qwen-plus-character-ja","context_length":8192,"description":"The Qwen Role-Playing Model Series is specifically optimized for Japanese anthropomorphic interaction scenarios. It demonstrates advanced capabilities in character consistency maintenance, context-aware dialogue progression, and empathetic engagement, enabling precise personalized character embodiment. This version significantly enhances Japanese linguistic localization (including dialects and honorifics), human-like role-playing authenticity, narrative coherence control, and scenario-based cognitive intelligence.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.25e-7,"output_cost_per_token":9.1e-7,"cache_read_input_token_cost":6.500000000000001e-8},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.4e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-vl-ocr","object":"model","created":1785415829,"owned_by":"qwen","model_name":"qwen-vl-ocr","context_length":38192,"description":"Qwen-VL_OCR is an OCR model trained based on Qwen-VL. It aggregates various image-text recognition, parsing, and processing tasks through a unified model approach, offering powerful image-text recognition capabilities.","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.5500000000000004e-8,"output_cost_per_token":1.04e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":1.6e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/glm-5.2-fast-preview","object":"model","created":1785412138,"owned_by":"qwen","model_name":"glm-5.2-fast-preview","context_length":1048576,"description":"GLM-5.2-Fast-Preview is the high-speed variant of Zhipu AI&#39;s GLM-5.2, with 1M context and capabilities on par with the standard version. Inference-optimized to deliver 1.5–2× the output TPS, it fits latency-sensitive use cases such as real-time chat, multi-turn agents, and streaming code generation.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.82e-6,"output_cost_per_token":5.72e-6,"cache_read_input_token_cost":3.6400000000000003e-7},"list_pricing":{"input_cost_per_token":2.8e-6,"output_cost_per_token":8.8e-6,"cache_read_input_token_cost":5.6e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-235b-a22b-instruct-2507","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3-235b-a22b-instruct-2507","context_length":131072,"description":"Open-source Qwen3 non-thinking model; compared to the previous version (Qwen3-235B-A22B) shows slight improvements in subjective creativity and model safety.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.495e-7,"output_cost_per_token":5.98e-7},"list_pricing":{"input_cost_per_token":2.3000000000000002e-7,"output_cost_per_token":9.200000000000001e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.5-flash","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3.5-flash","context_length":1000000,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the 3 series, these models deliver a leap forward in performance for both pure text and multimodal tasks, offering fast response times while balancing inference speed and overall performance.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.500000000000001e-8,"output_cost_per_token":2.6000000000000005e-7,"cache_creation_input_token_cost":8.125e-8},"list_pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":4.0000000000000003e-7,"cache_creation_input_token_cost":1.25e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.5-flash-2026-02-23","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3.5-flash-2026-02-23","context_length":1000000,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the 3 series, these models deliver a leap forward in performance for both pure text and multimodal tasks, offering fast response times while balancing inference speed and overall performance.This version is a snapshot as of February 23, 2026.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.500000000000001e-8,"output_cost_per_token":2.6000000000000005e-7},"list_pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":4.0000000000000003e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.6-35b-a3b","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3.6-35b-a3b","context_length":262144,"description":"The Qwen3.6 35B-A3B native vision-language model is built on a hybrid architecture that integrates linear attention mechanisms with a sparse mixture-of-experts framework, achieving higher inference efficiency. Compared with the 3.5-35B-A3B, this model demonstrates significantly improved agentic coding capabilities, mathematical and code reasoning abilities, spatial intelligence, as well as object localization and object detection performance.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.4375e-7,"output_cost_per_token":1.4625000000000002e-6},"list_pricing":{"input_cost_per_token":3.75e-7,"output_cost_per_token":2.25e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-coder-480b-a35b-instruct","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3-coder-480b-a35b-instruct","context_length":262144,"description":"Qwen3-based code generation model with strong coding agent power; code capability reaches open-source SOTA.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":9.75e-7,"output_cost_per_token":4.875e-6},{"range":[32000,128000],"input_cost_per_token":1.7550000000000001e-6,"output_cost_per_token":8.775e-6},{"range":[128000,256000],"input_cost_per_token":2.9250000000000004e-6,"output_cost_per_token":0.000014625000000000001},{"range":[256000,1000000],"input_cost_per_token":5.850000000000001e-6,"output_cost_per_token":0.000058500000000000006}],"input_cost_per_token":9.75e-7,"output_cost_per_token":4.875e-6},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6},{"range":[32000,128000],"input_cost_per_token":2.7e-6,"output_cost_per_token":0.0000135},{"range":[128000,256000],"input_cost_per_token":4.5e-6,"output_cost_per_token":0.0000225},{"range":[256000,1000000],"input_cost_per_token":9e-6,"output_cost_per_token":0.00009}],"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-vl-flash","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3-vl-flash","context_length":262144,"description":"The Qwen3 series of small-scale visual understanding models effectively integrates thinking and non-thinking modes, delivering superior performance compared to the open-source Qwen3-VL-30B-A3B while maintaining fast response speeds. It features a comprehensive upgrade in image/video understanding, supporting ultra-long contexts such as extended videos and documents, spatial awareness, and object recognition across various domains. Equipped with 2D/3D visual localization capabilities, it is well-suited for tackling complex real-world tasks.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7,"cache_read_input_token_cost":6.5e-9,"cache_creation_input_token_cost":4.0625e-8},{"range":[32000,128000],"input_cost_per_token":4.8749999999999996e-8,"output_cost_per_token":3.8999999999999997e-7,"cache_read_input_token_cost":9.75e-9,"cache_creation_input_token_cost":6.09375e-8},{"range":[128000,256000],"input_cost_per_token":7.8e-8,"output_cost_per_token":6.24e-7,"cache_read_input_token_cost":1.56e-8,"cache_creation_input_token_cost":9.749999999999999e-8}],"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7,"cache_read_input_token_cost":6.5e-9,"cache_creation_input_token_cost":4.0625e-8},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":1e-8,"cache_creation_input_token_cost":6.25e-8},{"range":[32000,128000],"input_cost_per_token":7.5e-8,"output_cost_per_token":6e-7,"cache_read_input_token_cost":1.5e-8,"cache_creation_input_token_cost":9.375e-8},{"range":[128000,256000],"input_cost_per_token":1.2e-7,"output_cost_per_token":9.6e-7,"cache_read_input_token_cost":2.4e-8,"cache_creation_input_token_cost":1.5e-7}],"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":1e-8,"cache_creation_input_token_cost":6.25e-8},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-vl-flash-2025-10-15","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3-vl-flash-2025-10-15","context_length":262144,"description":"The Qwen3 series of small-scale visual understanding models effectively integrates thinking and non-thinking modes, delivering superior performance compared to the open-source Qwen3-VL-30B-A3B while maintaining fast response speeds. It features a comprehensive upgrade in image/video understanding, supporting ultra-long contexts such as extended videos and documents, spatial awareness, and object recognition across various domains. Equipped with 2D/3D visual localization capabilities, it is well-suited for tackling complex real-world tasks.This version is a snapshot as of October 15, 2025.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7},{"range":[32000,128000],"input_cost_per_token":4.8749999999999996e-8,"output_cost_per_token":3.8999999999999997e-7},{"range":[128000,256000],"input_cost_per_token":7.8e-8,"output_cost_per_token":6.24e-7}],"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7},{"range":[32000,128000],"input_cost_per_token":7.5e-8,"output_cost_per_token":6e-7},{"range":[128000,256000],"input_cost_per_token":1.2e-7,"output_cost_per_token":9.6e-7}],"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-vl-flash-2026-01-22","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3-vl-flash-2026-01-22","context_length":262144,"description":"The Qwen3 series of small-sized visual understanding models effectively integrates thinking and non-thinking modes. Compared with the snapshot taken on October 15, 2025, the overall performance of the model has improved significantly: it delivers enhanced capabilities in general visual recognition and reasoning, and shows marked improvements in recognition accuracy across various business scenarios such as security, in-store inspections, equipment monitoring, and photo-based problem solving. This version is a snapshot as of January 22, 2026.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7},{"range":[32000,128000],"input_cost_per_token":4.8749999999999996e-8,"output_cost_per_token":3.8999999999999997e-7},{"range":[128000,256000],"input_cost_per_token":7.8e-8,"output_cost_per_token":6.24e-7}],"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7},{"range":[32000,128000],"input_cost_per_token":7.5e-8,"output_cost_per_token":6e-7},{"range":[128000,256000],"input_cost_per_token":1.2e-7,"output_cost_per_token":9.6e-7}],"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-vl-plus","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3-vl-plus","context_length":262144,"description":"The Qwen3 series VL models effectively integrates thinking and non-thinking modes, achieving world-leading performance in visual agent capabilities on public benchmark datasets such as OS World. This version features comprehensive upgrades in areas like visual coding, spatial perception, and multimodal reasoning, significantly enhancing visual perception and recognition abilities, and supporting the understanding of ultra-long videos.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":1.0400000000000002e-6,"cache_read_input_token_cost":2.6e-8,"cache_creation_input_token_cost":1.625e-7},{"range":[32000,128000],"input_cost_per_token":1.9499999999999999e-7,"output_cost_per_token":1.5599999999999999e-6,"cache_read_input_token_cost":3.9e-8,"cache_creation_input_token_cost":2.4375e-7},{"range":[128000,256000],"input_cost_per_token":3.8999999999999997e-7,"output_cost_per_token":3.1199999999999998e-6,"cache_read_input_token_cost":7.8e-8,"cache_creation_input_token_cost":4.875e-7}],"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":1.0400000000000002e-6,"cache_read_input_token_cost":2.6e-8,"cache_creation_input_token_cost":1.625e-7},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":1.6000000000000001e-6,"cache_read_input_token_cost":4e-8,"cache_creation_input_token_cost":2.5e-7},{"range":[32000,128000],"input_cost_per_token":3e-7,"output_cost_per_token":2.4e-6,"cache_read_input_token_cost":6e-8,"cache_creation_input_token_cost":3.75e-7},{"range":[128000,256000],"input_cost_per_token":6e-7,"output_cost_per_token":4.8e-6,"cache_read_input_token_cost":1.2e-7,"cache_creation_input_token_cost":7.5e-7}],"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":1.6000000000000001e-6,"cache_read_input_token_cost":4e-8,"cache_creation_input_token_cost":2.5e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-vl-plus-2025-09-23","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3-vl-plus-2025-09-23","context_length":262144,"description":"The Qwen3 series VL models effectively integrates thinking and non-thinking modes, achieving world-leading performance in visual agent capabilities on public benchmark datasets such as OS World. This version features comprehensive upgrades in areas like visual coding, spatial perception, and multimodal reasoning, significantly enhancing visual perception and recognition abilities, and supporting the understanding of ultra-long videos.This version is a snapshot as of September 23, 2025","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":1.0400000000000002e-6},{"range":[32000,128000],"input_cost_per_token":1.9499999999999999e-7,"output_cost_per_token":1.5599999999999999e-6},{"range":[128000,256000],"input_cost_per_token":3.8999999999999997e-7,"output_cost_per_token":3.1199999999999998e-6}],"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":1.0400000000000002e-6},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":1.6000000000000001e-6},{"range":[32000,128000],"input_cost_per_token":3e-7,"output_cost_per_token":2.4e-6},{"range":[128000,256000],"input_cost_per_token":6e-7,"output_cost_per_token":4.8e-6}],"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":1.6000000000000001e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-vl-plus-2025-12-19","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen3-vl-plus-2025-12-19","context_length":262144,"description":"The Qwen3 series of visual understanding models effectively integrates thinking and non-thinking modes. Compared to the snapshot released on September 23, this version delivers superior performance in reasoning and analysis tasks as well as style control, while also offering lower latency and faster response speeds. This version is based on a snapshot taken on December 19, 2025.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":1.0400000000000002e-6},{"range":[32000,128000],"input_cost_per_token":1.9499999999999999e-7,"output_cost_per_token":1.5599999999999999e-6},{"range":[128000,256000],"input_cost_per_token":3.8999999999999997e-7,"output_cost_per_token":3.1199999999999998e-6}],"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":1.0400000000000002e-6},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":1.6000000000000001e-6},{"range":[32000,128000],"input_cost_per_token":3e-7,"output_cost_per_token":2.4e-6},{"range":[128000,256000],"input_cost_per_token":6e-7,"output_cost_per_token":4.8e-6}],"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":1.6000000000000001e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-flash","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-flash","context_length":1000000,"description":"The Qwen3 Flash model offers a powerful fusion of thinking and non-thinking modes with dynamic in-conversation switching, excelling in complex reasoning while showing significant gains in instruction following and text comprehension. It supports a 1M context length and is billed on a tiered model corresponding to context usage.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7,"cache_read_input_token_cost":6.5e-9,"cache_creation_input_token_cost":4.095e-8},{"range":[256000,1000000],"input_cost_per_token":1.625e-7,"output_cost_per_token":1.3e-6,"cache_read_input_token_cost":3.2500000000000006e-8,"cache_creation_input_token_cost":2.0345e-7}],"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7,"cache_read_input_token_cost":6.5e-9,"cache_creation_input_token_cost":4.095e-8},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":1e-8,"cache_creation_input_token_cost":6.3e-8},{"range":[256000,1000000],"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":5.0000000000000004e-8,"cache_creation_input_token_cost":3.13e-7}],"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":1e-8,"cache_creation_input_token_cost":6.3e-8},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-flash-character","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-flash-character","context_length":8192,"description":"The Qwen Role-Playing Model Series is specifically optimized for muti-language anthropomorphic interaction scenarios. It demonstrates advanced capabilities in character consistency maintenance, context-aware dialogue progression, and empathetic engagement, enabling precise personalized character embodiment. This version significantly enhances Japanese linguistic localization (including dialects and honorifics), human-like role-playing authenticity, narrative coherence control, and scenario-based cognitive intelligence.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":2.6000000000000005e-7,"cache_read_input_token_cost":6.5e-9},"list_pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":1e-8},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-max","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-max","context_length":32768,"description":"Qwen-Max supports a parameter scale of hundreds of billions and multiple input languages such as Chinese and English. Qwen-Max is updated in a rolling manner. Compared to previous versions, it shows significant improvements in both Chinese and English code generation, logical reasoning, and multilingual abilities. The response style has been greatly adjusted to align with human preferences, with noticeable enhancements in the level of detail and clarity of responses. Specialized improvements have been made in creative writing, adherence to JSON formatting, and role-playing abilities.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.0400000000000002e-6,"output_cost_per_token":4.160000000000001e-6,"cache_read_input_token_cost":2.08e-7},"list_pricing":{"input_cost_per_token":1.6000000000000001e-6,"output_cost_per_token":6.4000000000000006e-6,"cache_read_input_token_cost":3.2e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-mt-flash","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-mt-flash","context_length":16384,"description":"Qwen-MT-Flash, a large language model from the Qwen series, has been fully upgraded with the Qwen 3 architecture for significantly enhanced performance and translation quality. It provides rapid, cost-effective translation across 92 languages, while supporting advanced features such as terminology intervention, format preservation, and domain-specific adaptation. It is the ideal choice for applications requiring a powerful balance of speed, quality, and cost.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.04e-7,"output_cost_per_token":3.185e-7},"list_pricing":{"input_cost_per_token":1.6e-7,"output_cost_per_token":4.9e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-mt-lite","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-mt-lite","context_length":16384,"description":"Qwen-MT-Lite is a large language model of the Qwen model series that specializes in multi-lingual translation. It provides high-quality and rapid translation services across 32 languages at a cost-effective price. It offers features such as terminology intervention, format preservation, and domain-specific translation to cater to the diverse needs of various applications, ensuring both efficiency and performance.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.8e-8,"output_cost_per_token":2.34e-7},"list_pricing":{"input_cost_per_token":1.2e-7,"output_cost_per_token":3.6e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-mt-plus","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-mt-plus","context_length":4096,"description":"Qwen-MT-Plus, the flagship translation model from our Qwen series, is now fully upgraded with the Qwen3 architecture. It supports 92 languages and delivers exceptionally accurate and natural-sounding translations. Its advanced capabilities in contextual understanding, terminology control, and format preservation make it a superior choice over traditional models, especially for specialized domains.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.599e-6,"output_cost_per_token":4.7905e-6},"list_pricing":{"input_cost_per_token":2.46e-6,"output_cost_per_token":7.37e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-mt-turbo","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-mt-turbo","context_length":4096,"description":"Qwen-MT-Turbo is a large language model within the Qwen model series that specializes in multi-lingual translation. It provides high-quality and rapid translation services across 92 languages at a cost-effective price point. It also offers features such as terminology intervention, format preservation, and domain-specific translation to cater to the diverse needs of various applications, ensuring both efficiency and performance.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.04e-7,"output_cost_per_token":3.185e-7},"list_pricing":{"input_cost_per_token":1.6e-7,"output_cost_per_token":4.9e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-plus-character","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-plus-character","context_length":32768,"description":"The role-playing model of the Qwen series. This is a dynamically updated version, and notifications will be provided in advance for any model updates. It is suitable for anthropomorphic role-playing and has optimized capabilities in following predefined character instructions, advancing conversations, and demonstrating active listening and empathy. Additionally, it supports the deep restoration of personalized characters.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.25e-7,"output_cost_per_token":9.1e-7,"cache_read_input_token_cost":6.500000000000001e-8},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.4e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-turbo","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-turbo","context_length":131072,"description":"The Turbo model of the Qwen3 series. It effectively integrates thinking mode and non-thinking mode, allowing for mode switching during conversations. Its reasoning capabilities rival those of QwQ-32B with a smaller parameter size, while its general capabilities significantly surpass those of Qwen2.5-Turbo, achieving the SOTA level in the same scale within the industry.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.2500000000000006e-8,"output_cost_per_token":1.3000000000000003e-7,"cache_read_input_token_cost":6.5e-9,"output_cost_per_reasoning_token":3.25e-7},"list_pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":2.0000000000000002e-7,"cache_read_input_token_cost":1e-8,"output_cost_per_reasoning_token":5e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-vl-max","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-vl-max","context_length":131072,"description":"Qwen-VL-Max is a large-scale visual language model of the Qwen series. Compared to the Plus version, it further enhances visual reasoning capabilities and instruction-following abilities, offering higher levels of visual perception and cognition. It delivers optimal performance on more complex tasks.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.200000000000001e-7,"output_cost_per_token":2.0800000000000004e-6,"cache_read_input_token_cost":1.04e-7},"list_pricing":{"input_cost_per_token":8.000000000000001e-7,"output_cost_per_token":3.2000000000000003e-6,"cache_read_input_token_cost":1.6e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-vl-ocr-2025-11-20","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-vl-ocr-2025-11-20","context_length":38192,"description":"This model is a snapshot version from November 20, 2025, and is based on the latest Qwen-VL3 architecture with a comprehensive upgrade. It features significant improvements in document parsing and text localization capabilities, as well as substantial reductions in end-to-end latency and illusions.","source":null,"capabilities":{"input_modalities":["image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.5500000000000004e-8,"output_cost_per_token":1.04e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":1.6e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-vl-plus","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwen-vl-plus","context_length":131072,"description":"Qwen-VL-Plus is the enhanced version of the large visual language model. It significantly improves detail recognition and text recognition capabilities, supporting images with resolutions exceeding one million pixels and any aspect ratio specifications. The model delivers exceptional performance across a wide range of visual tasks.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.365e-7,"output_cost_per_token":4.095e-7,"cache_read_input_token_cost":2.7300000000000003e-8},"list_pricing":{"input_cost_per_token":2.1e-7,"output_cost_per_token":6.3e-7,"cache_read_input_token_cost":4.2000000000000006e-8},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwq-plus","object":"model","created":1785412138,"owned_by":"qwen","model_name":"qwq-plus","context_length":131072,"description":"The enhanced version of the Qwen QwQ reasoning model, trained on the Qwen2.5 model, has significantly improved its reasoning capabilities through reinforcement learning. The model's core metrics in mathematics and coding (e.g., AIME 24/25, LiveCodeBench) as well as some general metrics (e.g., IFEval, LiveBench) have reached the level of the full version of DeepSeek-R1.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.200000000000001e-7,"output_cost_per_token":1.5599999999999999e-6},"list_pricing":{"input_cost_per_token":8.000000000000001e-7,"output_cost_per_token":2.4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/kimi-k2p7-code","object":"model","created":1785320305,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/kimi-k2p7-code","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.9e-7},"list_pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.9e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/minimax-m2p7","object":"model","created":1785320305,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/minimax-m2p7","context_length":196608,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":6e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":6e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/minimax-m3","object":"model","created":1785320305,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/minimax-m3","context_length":512000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":6e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":6e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/routers/kimi-k2p7-code-fast","object":"model","created":1785320305,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/routers/kimi-k2p7-code-fast","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.9e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":3.8e-7},"list_pricing":{"input_cost_per_token":1.9e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":3.8e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5.5@eu","object":"model","created":1785276709,"owned_by":"azure","model_name":"gpt-5.5","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000033,"cache_read_input_token_cost":5.500000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"input_cost_per_token_above_272k_tokens":0.000011000000000000001,"output_cost_per_token_above_272k_tokens":0.000049500000000000004,"cache_read_input_token_cost_above_272k_tokens":1.1000000000000003e-6,"cache_creation_input_token_cost_above_272k_tokens":0.000013750000000000002},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000033,"cache_read_input_token_cost":5.500000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"input_cost_per_token_above_272k_tokens":0.000011000000000000001,"output_cost_per_token_above_272k_tokens":0.000049500000000000004,"cache_read_input_token_cost_above_272k_tokens":1.1000000000000003e-6,"cache_creation_input_token_cost_above_272k_tokens":0.000013750000000000002},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-5.5","object":"model","created":1785276709,"owned_by":"azure","model_name":"gpt-5.5","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5.000000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1.0000000000000002e-6,"cache_creation_input_token_cost_above_272k_tokens":0.0000125},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5.000000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1.0000000000000002e-6,"cache_creation_input_token_cost_above_272k_tokens":0.0000125},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.5@us","object":"model","created":1785276709,"owned_by":"azure","model_name":"gpt-5.5","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000033,"cache_read_input_token_cost":5.500000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"input_cost_per_token_above_272k_tokens":0.000011000000000000001,"output_cost_per_token_above_272k_tokens":0.000049500000000000004,"cache_read_input_token_cost_above_272k_tokens":1.1000000000000003e-6,"cache_creation_input_token_cost_above_272k_tokens":0.000013750000000000002},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000033,"cache_read_input_token_cost":5.500000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"input_cost_per_token_above_272k_tokens":0.000011000000000000001,"output_cost_per_token_above_272k_tokens":0.000049500000000000004,"cache_read_input_token_cost_above_272k_tokens":1.1000000000000003e-6,"cache_creation_input_token_cost_above_272k_tokens":0.000013750000000000002},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/google/gemma-4-31B-it-Ultra","object":"model","created":1785251243,"owned_by":"deepinfra","model_name":"google/gemma-4-31B-it-Ultra","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.7e-7,"output_cost_per_token":7.6e-7},"list_pricing":{"input_cost_per_token":2.7e-7,"output_cost_per_token":7.6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/openai/gpt-oss-120b-Ultra","object":"model","created":1785251243,"owned_by":"deepinfra","model_name":"openai/gpt-oss-120b-Ultra","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":9.499999999999999e-7},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":9.499999999999999e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.7-flash","object":"model","created":1785190561,"owned_by":"qwen","model_name":"qwen3.7-flash","context_length":1000000,"description":"The Qwen3.7 native vision-language Flash model series delivers a comprehensive upgrade over 3.6-Flash in multimodal understanding and agent execution. This model particularly excels in enhanced multimodal foundations with stronger universal object recognition, further improved real-world perception and spatial intelligence, significantly upgraded multimodal agent capabilities for Search Agent and CI Agent scenarios with more stable end-to-end task execution, as well as optimized multimodal coding for a smoother vibe coding experience.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.95e-8,"output_cost_per_token":8.450000000000001e-8,"cache_read_input_token_cost":3.9e-9,"cache_creation_input_token_cost":2.47e-8},{"range":[32000,256000],"input_cost_per_token":6.500000000000001e-8,"output_cost_per_token":2.6000000000000005e-7,"cache_read_input_token_cost":1.3e-8,"cache_creation_input_token_cost":8.125e-8},{"range":[256000,1000000],"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":5.200000000000001e-7,"cache_read_input_token_cost":2.6e-8,"cache_creation_input_token_cost":1.625e-7}],"input_cost_per_token":1.95e-8,"output_cost_per_token":8.450000000000001e-8,"cache_read_input_token_cost":3.9e-9,"cache_creation_input_token_cost":2.47e-8},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":3e-8,"output_cost_per_token":1.3e-7,"cache_read_input_token_cost":6e-9,"cache_creation_input_token_cost":3.7999999999999996e-8},{"range":[32000,256000],"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":2e-8,"cache_creation_input_token_cost":1.25e-7},{"range":[256000,1000000],"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.000000000000001e-7,"cache_read_input_token_cost":4e-8,"cache_creation_input_token_cost":2.5e-7}],"input_cost_per_token":3e-8,"output_cost_per_token":1.3e-7,"cache_read_input_token_cost":6e-9,"cache_creation_input_token_cost":3.7999999999999996e-8},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.7-flash-2026-07-15","object":"model","created":1785190561,"owned_by":"qwen","model_name":"qwen3.7-flash-2026-07-15","context_length":1000000,"description":"The Qwen3.7 native vision-language Flash model series delivers a comprehensive upgrade over 3.6-Flash in multimodal understanding and agent execution. This model particularly excels in enhanced multimodal foundations with stronger universal object recognition, further improved real-world perception and spatial intelligence, significantly upgraded multimodal agent capabilities for Search Agent and CI Agent scenarios with more stable end-to-end task execution, as well as optimized multimodal coding for a smoother vibe coding experience. This version is a snapshot from July 15, 2026.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.95e-8,"output_cost_per_token":8.450000000000001e-8,"cache_read_input_token_cost":3.9e-9,"cache_creation_input_token_cost":2.47e-8},{"range":[32000,256000],"input_cost_per_token":6.500000000000001e-8,"output_cost_per_token":2.6000000000000005e-7,"cache_read_input_token_cost":1.3e-8,"cache_creation_input_token_cost":8.125e-8},{"range":[256000,1000000],"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":5.200000000000001e-7,"cache_read_input_token_cost":2.6e-8,"cache_creation_input_token_cost":1.625e-7}],"input_cost_per_token":1.95e-8,"output_cost_per_token":8.450000000000001e-8,"cache_read_input_token_cost":3.9e-9,"cache_creation_input_token_cost":2.47e-8},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":3e-8,"output_cost_per_token":1.3e-7,"cache_read_input_token_cost":6e-9,"cache_creation_input_token_cost":3.7999999999999996e-8},{"range":[32000,256000],"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":2e-8,"cache_creation_input_token_cost":1.25e-7},{"range":[256000,1000000],"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.000000000000001e-7,"cache_read_input_token_cost":4e-8,"cache_creation_input_token_cost":2.5e-7}],"input_cost_per_token":3e-8,"output_cost_per_token":1.3e-7,"cache_read_input_token_cost":6e-9,"cache_creation_input_token_cost":3.7999999999999996e-8},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/anthropic/claude-opus-5","object":"model","created":1785164744,"owned_by":"deepinfra","model_name":"anthropic/claude-opus-5","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-4o@eu","object":"model","created":1784916453,"owned_by":"azure","model_name":"gpt-4o","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.7500000000000004e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.3750000000000002e-6},"list_pricing":{"input_cost_per_token":2.7500000000000004e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.3750000000000002e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-4o","object":"model","created":1784916453,"owned_by":"azure","model_name":"gpt-4o","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-4o@us","object":"model","created":1784916453,"owned_by":"azure","model_name":"gpt-4o","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.7500000000000004e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.3750000000000002e-6},"list_pricing":{"input_cost_per_token":2.7500000000000004e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.3750000000000002e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-4o-mini","object":"model","created":1784916453,"owned_by":"azure","model_name":"gpt-4o-mini","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.8150000000000003e-7,"output_cost_per_token":7.260000000000001e-7,"cache_read_input_token_cost":8.25e-8},"list_pricing":{"input_cost_per_token":1.8150000000000003e-7,"output_cost_per_token":7.260000000000001e-7,"cache_read_input_token_cost":8.25e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-4o-mini@us","object":"model","created":1784916453,"owned_by":"azure","model_name":"gpt-4o-mini","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.8150000000000003e-7,"output_cost_per_token":7.260000000000001e-7,"cache_read_input_token_cost":8.25e-8},"list_pricing":{"input_cost_per_token":1.8150000000000003e-7,"output_cost_per_token":7.260000000000001e-7,"cache_read_input_token_cost":8.25e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5.4-pro","object":"model","created":1784916453,"owned_by":"azure","model_name":"gpt-5.4-pro","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"list_pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5-pro","object":"model","created":1784916453,"owned_by":"azure","model_name":"gpt-5-pro","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.00012},"list_pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.00012},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/o3@eu","object":"model","created":1784916453,"owned_by":"azure","model_name":"o3","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":8.8e-6,"cache_read_input_token_cost":5.5e-7},"list_pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":8.8e-6,"cache_read_input_token_cost":5.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/o3","object":"model","created":1784916453,"owned_by":"azure","model_name":"o3","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/o3@us","object":"model","created":1784916453,"owned_by":"azure","model_name":"o3","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":8.8e-6,"cache_read_input_token_cost":5.5e-7},"list_pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":8.8e-6,"cache_read_input_token_cost":5.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-5@eu","object":"model","created":1784912544,"owned_by":"amazon","model_name":"anthropic.claude-opus-5","context_length":1000000,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-5","object":"model","created":1784912544,"owned_by":"amazon","model_name":"anthropic.claude-opus-5","context_length":1000000,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-5@us","object":"model","created":1784912544,"owned_by":"amazon","model_name":"anthropic.claude-opus-5","context_length":1000000,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"anthropic/claude-opus-5","object":"model","created":1784912544,"owned_by":"anthropic","model_name":"claude-opus-5","context_length":1000000,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.1-codex","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.1-codex","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.1-codex-max","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.1-codex-max","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.1-codex-mini","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.1-codex-mini","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":2.5e-8},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":2.5e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.2","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.2","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7},"list_pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.2@us","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.2","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.925e-6,"output_cost_per_token":0.0000154,"cache_read_input_token_cost":1.925e-7},"list_pricing":{"input_cost_per_token":1.925e-6,"output_cost_per_token":0.0000154,"cache_read_input_token_cost":1.925e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5.2-codex","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.2-codex","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7},"list_pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.3-codex","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.3-codex","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7},"list_pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.4@eu","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.4","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.7500000000000004e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":2.75e-7,"input_cost_per_token_above_272k_tokens":5.500000000000001e-6,"output_cost_per_token_above_272k_tokens":0.000024750000000000002,"cache_read_input_token_cost_above_272k_tokens":5.5e-7},"list_pricing":{"input_cost_per_token":2.7500000000000004e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":2.75e-7,"input_cost_per_token_above_272k_tokens":5.500000000000001e-6,"output_cost_per_token_above_272k_tokens":0.000024750000000000002,"cache_read_input_token_cost_above_272k_tokens":5.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-5.4","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.4","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":2.5e-7,"input_cost_per_token_above_272k_tokens":5e-6,"output_cost_per_token_above_272k_tokens":0.0000225,"cache_read_input_token_cost_above_272k_tokens":5e-7},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":2.5e-7,"input_cost_per_token_above_272k_tokens":5e-6,"output_cost_per_token_above_272k_tokens":0.0000225,"cache_read_input_token_cost_above_272k_tokens":5e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.4@us","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.4","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.7500000000000004e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":2.75e-7,"input_cost_per_token_above_272k_tokens":5.500000000000001e-6,"output_cost_per_token_above_272k_tokens":0.000024750000000000002,"cache_read_input_token_cost_above_272k_tokens":5.5e-7},"list_pricing":{"input_cost_per_token":2.7500000000000004e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":2.75e-7,"input_cost_per_token_above_272k_tokens":5.500000000000001e-6,"output_cost_per_token_above_272k_tokens":0.000024750000000000002,"cache_read_input_token_cost_above_272k_tokens":5.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5.4-mini@eu","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.4-mini","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8.25e-7,"output_cost_per_token":4.950000000000001e-6,"cache_read_input_token_cost":8.25e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":8.25e-7,"output_cost_per_token":4.950000000000001e-6,"cache_read_input_token_cost":8.25e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-5.4-mini","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.4-mini","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.4-mini@us","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.4-mini","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8.25e-7,"output_cost_per_token":4.950000000000001e-6,"cache_read_input_token_cost":8.25e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":8.25e-7,"output_cost_per_token":4.950000000000001e-6,"cache_read_input_token_cost":8.25e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5.4-nano","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.4-nano","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.25e-6,"cache_read_input_token_cost":2e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.25e-6,"cache_read_input_token_cost":2e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.4-nano@us","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5.4-nano","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-7,"output_cost_per_token":1.3750000000000002e-6,"cache_read_input_token_cost":2.2000000000000002e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2.2e-7,"output_cost_per_token":1.3750000000000002e-6,"cache_read_input_token_cost":2.2000000000000002e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5-codex","object":"model","created":1784832896,"owned_by":"azure","model_name":"gpt-5-codex","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/inclusionAI/Ling-3.0-flash","object":"model","created":1784818580,"owned_by":"deepinfra","model_name":"inclusionAI/Ling-3.0-flash","context_length":131072,"description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.000000000000001e-8,"output_cost_per_token":1.8e-7,"cache_read_input_token_cost":1.2000000000000002e-8},"list_pricing":{"input_cost_per_token":6.000000000000001e-8,"output_cost_per_token":1.8e-7,"cache_read_input_token_cost":1.2000000000000002e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5.6-luna@eu","object":"model","created":1784804086,"owned_by":"azure","model_name":"gpt-5.6-luna","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2000000000000004e-7,"output_cost_per_token":1.32e-6,"cache_read_input_token_cost":2.2000000000000005e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000007e-7,"input_cost_per_token_above_272k_tokens":4.400000000000001e-7,"output_cost_per_token_above_272k_tokens":1.98e-6,"cache_read_input_token_cost_above_272k_tokens":4.400000000000001e-8,"cache_creation_input_token_cost_above_272k_tokens":5.500000000000001e-7},"list_pricing":{"input_cost_per_token":2.2000000000000004e-7,"output_cost_per_token":1.32e-6,"cache_read_input_token_cost":2.2000000000000005e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000007e-7,"input_cost_per_token_above_272k_tokens":4.400000000000001e-7,"output_cost_per_token_above_272k_tokens":1.98e-6,"cache_read_input_token_cost_above_272k_tokens":4.400000000000001e-8,"cache_creation_input_token_cost_above_272k_tokens":5.500000000000001e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-5.6-luna","object":"model","created":1784804086,"owned_by":"azure","model_name":"gpt-5.6-luna","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":2.0000000000000004e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5000000000000004e-7,"input_cost_per_token_above_272k_tokens":4.0000000000000003e-7,"output_cost_per_token_above_272k_tokens":1.8000000000000001e-6,"cache_read_input_token_cost_above_272k_tokens":4.000000000000001e-8,"cache_creation_input_token_cost_above_272k_tokens":5.000000000000001e-7},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":2.0000000000000004e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5000000000000004e-7,"input_cost_per_token_above_272k_tokens":4.0000000000000003e-7,"output_cost_per_token_above_272k_tokens":1.8000000000000001e-6,"cache_read_input_token_cost_above_272k_tokens":4.000000000000001e-8,"cache_creation_input_token_cost_above_272k_tokens":5.000000000000001e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.6-luna@us","object":"model","created":1784804086,"owned_by":"azure","model_name":"gpt-5.6-luna","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2000000000000004e-7,"output_cost_per_token":1.32e-6,"cache_read_input_token_cost":2.2000000000000005e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000007e-7,"input_cost_per_token_above_272k_tokens":4.400000000000001e-7,"output_cost_per_token_above_272k_tokens":1.98e-6,"cache_read_input_token_cost_above_272k_tokens":4.400000000000001e-8,"cache_creation_input_token_cost_above_272k_tokens":5.500000000000001e-7},"list_pricing":{"input_cost_per_token":2.2000000000000004e-7,"output_cost_per_token":1.32e-6,"cache_read_input_token_cost":2.2000000000000005e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000007e-7,"input_cost_per_token_above_272k_tokens":4.400000000000001e-7,"output_cost_per_token_above_272k_tokens":1.98e-6,"cache_read_input_token_cost_above_272k_tokens":4.400000000000001e-8,"cache_creation_input_token_cost_above_272k_tokens":5.500000000000001e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5.6-sol@eu","object":"model","created":1784804086,"owned_by":"azure","model_name":"gpt-5.6-sol","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000033,"cache_read_input_token_cost":5.500000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"input_cost_per_token_above_272k_tokens":0.000011000000000000001,"output_cost_per_token_above_272k_tokens":0.000049500000000000004,"cache_read_input_token_cost_above_272k_tokens":1.1000000000000003e-6,"cache_creation_input_token_cost_above_272k_tokens":0.000013750000000000002},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000033,"cache_read_input_token_cost":5.500000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"input_cost_per_token_above_272k_tokens":0.000011000000000000001,"output_cost_per_token_above_272k_tokens":0.000049500000000000004,"cache_read_input_token_cost_above_272k_tokens":1.1000000000000003e-6,"cache_creation_input_token_cost_above_272k_tokens":0.000013750000000000002},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-5.6-sol","object":"model","created":1784804086,"owned_by":"azure","model_name":"gpt-5.6-sol","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5.000000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1.0000000000000002e-6,"cache_creation_input_token_cost_above_272k_tokens":0.0000125},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5.000000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1.0000000000000002e-6,"cache_creation_input_token_cost_above_272k_tokens":0.0000125},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.6-sol@us","object":"model","created":1784804086,"owned_by":"azure","model_name":"gpt-5.6-sol","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000033,"cache_read_input_token_cost":5.500000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"input_cost_per_token_above_272k_tokens":0.000011000000000000001,"output_cost_per_token_above_272k_tokens":0.000049500000000000004,"cache_read_input_token_cost_above_272k_tokens":1.1000000000000003e-6,"cache_creation_input_token_cost_above_272k_tokens":0.000013750000000000002},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000033,"cache_read_input_token_cost":5.500000000000001e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"input_cost_per_token_above_272k_tokens":0.000011000000000000001,"output_cost_per_token_above_272k_tokens":0.000049500000000000004,"cache_read_input_token_cost_above_272k_tokens":1.1000000000000003e-6,"cache_creation_input_token_cost_above_272k_tokens":0.000013750000000000002},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5.6-terra@eu","object":"model","created":1784804086,"owned_by":"azure","model_name":"gpt-5.6-terra","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.0000132,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.75e-6,"input_cost_per_token_above_272k_tokens":4.4e-6,"output_cost_per_token_above_272k_tokens":0.000019800000000000004,"cache_read_input_token_cost_above_272k_tokens":4.4e-7,"cache_creation_input_token_cost_above_272k_tokens":5.5e-6},"list_pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.0000132,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.75e-6,"input_cost_per_token_above_272k_tokens":4.4e-6,"output_cost_per_token_above_272k_tokens":0.000019800000000000004,"cache_read_input_token_cost_above_272k_tokens":4.4e-7,"cache_creation_input_token_cost_above_272k_tokens":5.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-5.6-terra","object":"model","created":1784804086,"owned_by":"azure","model_name":"gpt-5.6-terra","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.4999999999999998e-6,"input_cost_per_token_above_272k_tokens":4e-6,"output_cost_per_token_above_272k_tokens":0.000018,"cache_read_input_token_cost_above_272k_tokens":4e-7,"cache_creation_input_token_cost_above_272k_tokens":4.9999999999999996e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.4999999999999998e-6,"input_cost_per_token_above_272k_tokens":4e-6,"output_cost_per_token_above_272k_tokens":0.000018,"cache_read_input_token_cost_above_272k_tokens":4e-7,"cache_creation_input_token_cost_above_272k_tokens":4.9999999999999996e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.6-terra@us","object":"model","created":1784804086,"owned_by":"azure","model_name":"gpt-5.6-terra","context_length":1050000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.0000132,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.75e-6,"input_cost_per_token_above_272k_tokens":4.4e-6,"output_cost_per_token_above_272k_tokens":0.000019800000000000004,"cache_read_input_token_cost_above_272k_tokens":4.4e-7,"cache_creation_input_token_cost_above_272k_tokens":5.5e-6},"list_pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.0000132,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.75e-6,"input_cost_per_token_above_272k_tokens":4.4e-6,"output_cost_per_token_above_272k_tokens":0.000019800000000000004,"cache_read_input_token_cost_above_272k_tokens":4.4e-7,"cache_creation_input_token_cost_above_272k_tokens":5.5e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5@eu","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3750000000000002e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.375e-7},"list_pricing":{"input_cost_per_token":1.3750000000000002e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.375e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-5","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5@us","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3750000000000002e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.375e-7},"list_pricing":{"input_cost_per_token":1.3750000000000002e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.375e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5.1@eu","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5.1","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3750000000000002e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.375e-7},"list_pricing":{"input_cost_per_token":1.3750000000000002e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.375e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-5.1","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5.1","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5.1@us","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5.1","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3750000000000002e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.375e-7},"list_pricing":{"input_cost_per_token":1.3750000000000002e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":1.375e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5-mini@eu","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5-mini","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.75e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":2.75e-8},"list_pricing":{"input_cost_per_token":2.75e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":2.75e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-5-mini","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5-mini","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":2.5e-8},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":2.5e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5-mini@us","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5-mini","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.75e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":2.75e-8},"list_pricing":{"input_cost_per_token":2.75e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":2.75e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"azure/gpt-5-nano@eu","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5-nano","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.5e-8,"output_cost_per_token":4.4e-7,"cache_read_input_token_cost":5.5000000000000004e-9},"list_pricing":{"input_cost_per_token":5.5e-8,"output_cost_per_token":4.4e-7,"cache_read_input_token_cost":5.5000000000000004e-9},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"azure/gpt-5-nano","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5-nano","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"cache_read_input_token_cost":5e-9},"list_pricing":{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"cache_read_input_token_cost":5e-9},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"azure/gpt-5-nano@us","object":"model","created":1784721580,"owned_by":"azure","model_name":"gpt-5-nano","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.5e-8,"output_cost_per_token":4.4e-7,"cache_read_input_token_cost":5.5000000000000004e-9},"list_pricing":{"input_cost_per_token":5.5e-8,"output_cost_per_token":4.4e-7,"cache_read_input_token_cost":5.5000000000000004e-9},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"google/gemini-3.6-flash","object":"model","created":1784646733,"owned_by":"google","model_name":"gemini-3.6-flash","context_length":1048576,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":3.75e-6,"input_cost_per_audio_token":7.5e-7,"input_cost_per_image_token":7.5e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":4.16666666666667e-8,"output_cost_per_reasoning_token":3.75e-6,"cache_read_input_audio_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":3.75e-6,"input_cost_per_audio_token":7.5e-7,"input_cost_per_image_token":7.5e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":4.16666666666667e-8,"output_cost_per_reasoning_token":3.75e-6,"cache_read_input_audio_token_cost":7.5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3.6-flash@eu","object":"model","created":1784646733,"owned_by":"vertex","model_name":"gemini-3.6-flash","context_length":1048576,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":3.75e-6,"input_cost_per_audio_token":7.5e-7,"input_cost_per_image_token":7.5e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":4.16666666666667e-8,"output_cost_per_reasoning_token":3.75e-6,"cache_read_input_audio_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":3.75e-6,"input_cost_per_audio_token":7.5e-7,"input_cost_per_image_token":7.5e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":4.16666666666667e-8,"output_cost_per_reasoning_token":3.75e-6,"cache_read_input_audio_token_cost":7.5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3.6-flash","object":"model","created":1784646733,"owned_by":"vertex","model_name":"gemini-3.6-flash","context_length":1048576,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":3.75e-6,"input_cost_per_audio_token":7.5e-7,"input_cost_per_image_token":7.5e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":4.16666666666667e-8,"output_cost_per_reasoning_token":3.75e-6,"cache_read_input_audio_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":3.75e-6,"input_cost_per_audio_token":7.5e-7,"input_cost_per_image_token":7.5e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":4.16666666666667e-8,"output_cost_per_reasoning_token":3.75e-6,"cache_read_input_audio_token_cost":7.5e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/gemini-3.6-flash@us","object":"model","created":1784646733,"owned_by":"vertex","model_name":"gemini-3.6-flash","context_length":1048576,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":3.75e-6,"input_cost_per_audio_token":7.5e-7,"input_cost_per_image_token":7.5e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":4.16666666666667e-8,"output_cost_per_reasoning_token":3.75e-6,"cache_read_input_audio_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":3.75e-6,"input_cost_per_audio_token":7.5e-7,"input_cost_per_image_token":7.5e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":4.16666666666667e-8,"output_cost_per_reasoning_token":3.75e-6,"cache_read_input_audio_token_cost":7.5e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"google/gemini-3.5-flash-lite","object":"model","created":1784646726,"owned_by":"google","model_name":"gemini-3.5-flash-lite","context_length":1048576,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":3e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":3e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3.5-flash-lite@eu","object":"model","created":1784646726,"owned_by":"vertex","model_name":"gemini-3.5-flash-lite","context_length":1048576,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":3e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":3e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3.5-flash-lite","object":"model","created":1784646726,"owned_by":"vertex","model_name":"gemini-3.5-flash-lite","context_length":1048576,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":3e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":3e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/gemini-3.5-flash-lite@us","object":"model","created":1784646726,"owned_by":"vertex","model_name":"gemini-3.5-flash-lite","context_length":1048576,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":3e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":3e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/thinkingmachines/Inkling","object":"model","created":1784325956,"owned_by":"deepinfra","model_name":"thinkingmachines/Inkling","context_length":524288,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","source":null,"capabilities":{"input_modalities":["text","image","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.499999999999999e-7,"output_cost_per_token":4.05e-6,"cache_read_input_token_cost":1.599999975e-7},"list_pricing":{"input_cost_per_token":9.499999999999999e-7,"output_cost_per_token":4.05e-6,"cache_read_input_token_cost":1.599999975e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"together_ai/thinkingmachines/Inkling","object":"model","created":1784325956,"owned_by":"together_ai","model_name":"thinkingmachines/Inkling","context_length":524288,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","source":null,"capabilities":{"input_modalities":["text","image","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":4.05e-6,"cache_read_input_token_cost":1.7000000000000001e-7},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":4.05e-6,"cache_read_input_token_cost":1.7000000000000001e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/moonshotai/Kimi-K3","object":"model","created":1784215858,"owned_by":"deepinfra","model_name":"moonshotai/Kimi-K3","context_length":1048576,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.85e-6,"output_cost_per_token":0.00001425,"cache_read_input_token_cost":2.8499999999999997e-7},"list_pricing":{"input_cost_per_token":2.85e-6,"output_cost_per_token":0.00001425,"cache_read_input_token_cost":2.8499999999999997e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/kimi-k3","object":"model","created":1784215858,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/kimi-k3","context_length":1048576,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/kimi-k3","object":"model","created":1784215858,"owned_by":"fireworks_ai","model_name":"kimi-k3","context_length":1048576,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"moonshot/kimi-k3","object":"model","created":1784215858,"owned_by":"moonshot","model_name":"kimi-k3","context_length":1048576,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"nebius/moonshotai/Kimi-K3","object":"model","created":1784215858,"owned_by":"nebius","model_name":"moonshotai/Kimi-K3","context_length":8000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-6},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/moonshotai/Kimi-K3","object":"model","created":1784215858,"owned_by":"together_ai","model_name":"moonshotai/Kimi-K3","context_length":1048576,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/google/gemma-4-E4B-it","object":"model","created":1784092401,"owned_by":"deepinfra","model_name":"google/gemma-4-E4B-it","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":1.0000000000000001e-7},"list_pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":1.0000000000000001e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/MiniMaxAI/MiniMax-M3","object":"model","created":1783732576,"owned_by":"nebius","model_name":"MiniMaxAI/MiniMax-M3","context_length":8000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-6-luna","object":"model","created":1783590864,"owned_by":"databricks","model_name":"databricks-gpt-5-6-luna","context_length":1050000,"description":"GPT-5.6 Luna is the fast, cost-efficient model in OpenAI's GPT-5.6 family, bringing strong reasoning and agentic capability at the lowest cost in the lineup. This model supports multimodal (text and image) inputs and a long context window. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.9999000000000003e-7,"output_cost_per_token":1.79998e-6},"list_pricing":{"input_cost_per_token":1.9999000000000003e-7,"output_cost_per_token":1.79998e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-5.6-luna","object":"model","created":1783590864,"owned_by":"openai","model_name":"gpt-5.6-luna","context_length":1050000,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":2e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-7,"input_cost_per_token_above_272k_tokens":4e-7,"output_cost_per_token_above_272k_tokens":1.8e-6,"cache_read_input_token_cost_above_272k_tokens":4e-8,"cache_creation_input_token_cost_above_272k_tokens":5e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":2e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-7,"input_cost_per_token_above_272k_tokens":4e-7,"output_cost_per_token_above_272k_tokens":1.8e-6,"cache_read_input_token_cost_above_272k_tokens":4e-8,"cache_creation_input_token_cost_above_272k_tokens":5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-6-terra","object":"model","created":1783590857,"owned_by":"databricks","model_name":"databricks-gpt-5-6-terra","context_length":1050000,"description":"GPT-5.6 Terra is the balanced model in OpenAI's GPT-5.6 family for everyday agentic and reasoning workloads, offering performance competitive with GPT-5.5 at a lower cost. This model supports multimodal (text and image) inputs and a long context window. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.000030000000001e-6,"output_cost_per_token":0.000018000009999999998},"list_pricing":{"input_cost_per_token":5.000030000000001e-6,"output_cost_per_token":0.000018000009999999998},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-5.6-terra","object":"model","created":1783590857,"owned_by":"openai","model_name":"gpt-5.6-terra","context_length":1050000,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-6,"input_cost_per_token_above_272k_tokens":4e-6,"output_cost_per_token_above_272k_tokens":0.000018,"cache_read_input_token_cost_above_272k_tokens":4e-7,"cache_creation_input_token_cost_above_272k_tokens":5e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-6,"input_cost_per_token_above_272k_tokens":4e-6,"output_cost_per_token_above_272k_tokens":0.000018,"cache_read_input_token_cost_above_272k_tokens":4e-7,"cache_creation_input_token_cost_above_272k_tokens":5e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-6-sol","object":"model","created":1783590850,"owned_by":"databricks","model_name":"databricks-gpt-5-6-sol","context_length":1050000,"description":"GPT-5.6 Sol is OpenAI's flagship model in the GPT-5.6 family, delivering state-of-the-art performance on agentic coding, long-horizon reasoning, and complex tool use, with support for a new maximum reasoning effort. This model supports multimodal (text and image) inputs and a long context window. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.99999e-6,"output_cost_per_token":0.00006000001000000001},"list_pricing":{"input_cost_per_token":9.99999e-6,"output_cost_per_token":0.00006000001000000001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-5.6-sol","object":"model","created":1783590850,"owned_by":"openai","model_name":"gpt-5.6-sol","context_length":1050000,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.02,"search_context_size_high":0.02,"search_context_size_medium":0.02},"cache_creation_input_token_cost":6.25e-6,"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1e-6,"cache_creation_input_token_cost_above_272k_tokens":0.0000125},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.02,"search_context_size_high":0.02,"search_context_size_medium":0.02},"cache_creation_input_token_cost":6.25e-6,"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1e-6,"cache_creation_input_token_cost_above_272k_tokens":0.0000125},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"together_ai/LiquidAI/LFM2.5-8B-A1B","object":"model","created":1783573977,"owned_by":"together_ai","model_name":"LiquidAI/LFM2.5-8B-A1B","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.0000000000000004e-8,"output_cost_per_token":1.2000000000000002e-7},"list_pricing":{"input_cost_per_token":3.0000000000000004e-8,"output_cost_per_token":1.2000000000000002e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"xai/grok-4.5","object":"model","created":1783523154,"owned_by":"xai","model_name":"grok-4.5","context_length":500000,"description":"Grok 4.5 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000012,"cache_read_input_token_cost_above_200k_tokens":6e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000012,"cache_read_input_token_cost_above_200k_tokens":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/anthropic/claude-fable-5","object":"model","created":1783401184,"owned_by":"deepinfra","model_name":"anthropic/claude-fable-5","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005},"list_pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/anthropic/claude-sonnet-5","object":"model","created":1783401184,"owned_by":"deepinfra","model_name":"anthropic/claude-sonnet-5","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/anthropic.claude-sonnet-5@eu","object":"model","created":1782843083,"owned_by":"amazon","model_name":"anthropic.claude-sonnet-5","context_length":1000000,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000004e-6,"cache_creation_input_token_cost_above_1hr":4.4e-6},"list_pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000004e-6,"cache_creation_input_token_cost_above_1hr":4.4e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-sonnet-5","object":"model","created":1782843083,"owned_by":"amazon","model_name":"anthropic.claude-sonnet-5","context_length":1000000,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-6,"cache_creation_input_token_cost_above_1hr":4e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-6,"cache_creation_input_token_cost_above_1hr":4e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/anthropic.claude-sonnet-5@us","object":"model","created":1782843083,"owned_by":"amazon","model_name":"anthropic.claude-sonnet-5","context_length":1000000,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000004e-6,"cache_creation_input_token_cost_above_1hr":4.4e-6},"list_pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000004e-6,"cache_creation_input_token_cost_above_1hr":4.4e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"anthropic/claude-sonnet-5","object":"model","created":1782843083,"owned_by":"anthropic","model_name":"claude-sonnet-5","context_length":1000000,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-6,"cache_creation_input_token_cost_above_1hr":4e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-6,"cache_creation_input_token_cost_above_1hr":4e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-sonnet-5@eu","object":"model","created":1782843083,"owned_by":"vertex","model_name":"claude-sonnet-5","context_length":1000000,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000004e-6,"cache_creation_input_token_cost_above_1hr":4.4e-6},"list_pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000004e-6,"cache_creation_input_token_cost_above_1hr":4.4e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/claude-sonnet-5","object":"model","created":1782843083,"owned_by":"vertex","model_name":"claude-sonnet-5","context_length":1000000,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-6,"cache_creation_input_token_cost_above_1hr":4e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-6,"cache_creation_input_token_cost_above_1hr":4e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-sonnet-5@us","object":"model","created":1782843083,"owned_by":"vertex","model_name":"claude-sonnet-5","context_length":1000000,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000004e-6,"cache_creation_input_token_cost_above_1hr":4.4e-6},"list_pricing":{"input_cost_per_token":2.2e-6,"output_cost_per_token":0.000011000000000000001,"cache_read_input_token_cost":2.2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.7500000000000004e-6,"cache_creation_input_token_cost_above_1hr":4.4e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"google/gemini-3.1-flash-lite-image","object":"model","created":1782837225,"owned_by":"google","model_name":"gemini-3.1-flash-lite-image","context_length":65536,"description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["image","text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6,"output_cost_per_image_token":0.00003,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014}},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6,"output_cost_per_image_token":0.00003,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014}},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3.1-flash-lite-image","object":"model","created":1782837225,"owned_by":"vertex","model_name":"gemini-3.1-flash-lite-image","context_length":65536,"description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["image","text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6,"output_cost_per_image_token":0.00003,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014}},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6,"output_cost_per_image_token":0.00003,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014}},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/deepreinforce-ai/Ornith-1.0-35B","object":"model","created":1782745560,"owned_by":"deepinfra","model_name":"deepreinforce-ai/Ornith-1.0-35B","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":7.499999999999999e-7,"cache_read_input_token_cost":5.0000000000000004e-8},"list_pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":7.499999999999999e-7,"cache_read_input_token_cost":5.0000000000000004e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/glm-5p2","object":"model","created":1782745560,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/glm-5p2","context_length":1048576,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":1.4e-7},"list_pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":1.4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/kimi-k2p6","object":"model","created":1782745560,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/kimi-k2p6","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.6e-7},"list_pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"mistral/mistral-medium-2508","object":"model","created":1782745560,"owned_by":"mistral","model_name":"mistral-medium-2508","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-medium-2604","object":"model","created":1782745560,"owned_by":"mistral","model_name":"mistral-medium-2604","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/glm-5.2","object":"model","created":1782376549,"owned_by":"scaleway","model_name":"glm-5.2","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.102580216355504e-6,"output_cost_per_token":6.424550661086263e-6},"list_pricing":{"input_cost_per_token":2.102580216355504e-6,"output_cost_per_token":6.424550661086263e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"cloudflare/@cf/google/gemma-2b-it-lora","object":"model","created":1782278004,"owned_by":"cloudflare","model_name":"@cf/google/gemma-2b-it-lora","context_length":8192,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"list_pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/google/gemma-7b-it-lora","object":"model","created":1782278004,"owned_by":"cloudflare","model_name":"@cf/google/gemma-7b-it-lora","context_length":3500,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"list_pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/meta-llama/llama-2-7b-chat-hf-lora","object":"model","created":1782278004,"owned_by":"cloudflare","model_name":"@cf/meta-llama/llama-2-7b-chat-hf-lora","context_length":8192,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"list_pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/mistral/mistral-7b-instruct-v0.2-lora","object":"model","created":1782278004,"owned_by":"cloudflare","model_name":"@cf/mistral/mistral-7b-instruct-v0.2-lora","context_length":15000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"list_pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"lilac/minimaxai/minimax-m3","object":"model","created":1782191630,"owned_by":"lilac","model_name":"minimaxai/minimax-m3","context_length":1048576,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.1e-7,"output_cost_per_token":8.25e-7,"cache_read_input_token_cost":3.75e-8},"list_pricing":{"input_cost_per_token":2.8e-7,"output_cost_per_token":1.1e-6,"cache_read_input_token_cost":5e-8},"discount":0.25,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"lilac/zai-org/glm-5.2","object":"model","created":1782191630,"owned_by":"lilac","model_name":"zai-org/glm-5.2","context_length":524288,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.75e-7,"output_cost_per_token":2.25e-6,"cache_read_input_token_cost":1.275e-7},"list_pricing":{"input_cost_per_token":9e-7,"output_cost_per_token":3e-6,"cache_read_input_token_cost":1.7e-7},"discount":0.25,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"cloudflare/@cf/aisingapore/gemma-sea-lion-v4-27b-it","object":"model","created":1782119040,"owned_by":"cloudflare","model_name":"@cf/aisingapore/gemma-sea-lion-v4-27b-it","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.5099999999999995e-7,"output_cost_per_token":5.550000000000001e-7},"list_pricing":{"input_cost_per_token":3.5099999999999995e-7,"output_cost_per_token":5.550000000000001e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b","object":"model","created":1782119040,"owned_by":"cloudflare","model_name":"@cf/deepseek-ai/deepseek-r1-distill-qwen-32b","context_length":80000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.97e-7,"output_cost_per_token":4.881e-6},"list_pricing":{"input_cost_per_token":4.97e-7,"output_cost_per_token":4.881e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/meta/llama-3.3-70b-instruct-fp8-fast","object":"model","created":1782119040,"owned_by":"cloudflare","model_name":"@cf/meta/llama-3.3-70b-instruct-fp8-fast","context_length":24000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.93e-7,"output_cost_per_token":2.2530000000000003e-6},"list_pricing":{"input_cost_per_token":2.93e-7,"output_cost_per_token":2.2530000000000003e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/meta/llama-4-scout-17b-16e-instruct","object":"model","created":1782119040,"owned_by":"cloudflare","model_name":"@cf/meta/llama-4-scout-17b-16e-instruct","context_length":131000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.7e-7,"output_cost_per_token":8.5e-7},"list_pricing":{"input_cost_per_token":2.7e-7,"output_cost_per_token":8.5e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/meta/llama-guard-3-8b","object":"model","created":1782119040,"owned_by":"cloudflare","model_name":"@cf/meta/llama-guard-3-8b","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.839999999999999e-7,"output_cost_per_token":3e-8},"list_pricing":{"input_cost_per_token":4.839999999999999e-7,"output_cost_per_token":3e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/nvidia/nemotron-3-120b-a12b","object":"model","created":1782119040,"owned_by":"cloudflare","model_name":"@cf/nvidia/nemotron-3-120b-a12b","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/openai/gpt-oss-120b","object":"model","created":1782119040,"owned_by":"cloudflare","model_name":"@cf/openai/gpt-oss-120b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.5e-7,"output_cost_per_token":7.5e-7},"list_pricing":{"input_cost_per_token":3.5e-7,"output_cost_per_token":7.5e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/openai/gpt-oss-20b","object":"model","created":1782119040,"owned_by":"cloudflare","model_name":"@cf/openai/gpt-oss-20b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/qwen/qwen3-30b-a3b-fp8","object":"model","created":1782119040,"owned_by":"cloudflare","model_name":"@cf/qwen/qwen3-30b-a3b-fp8","context_length":32768,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.09e-8,"output_cost_per_token":3.35e-7},"list_pricing":{"input_cost_per_token":5.09e-8,"output_cost_per_token":3.35e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/qwen/qwq-32b","object":"model","created":1782119040,"owned_by":"cloudflare","model_name":"@cf/qwen/qwq-32b","context_length":24000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":1e-6},"list_pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":1e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/anthropic/claude-haiku-4-5","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"anthropic/claude-haiku-4-5","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/anthropic/claude-opus-4-7","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"anthropic/claude-opus-4-7","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/anthropic/claude-opus-4-8","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"anthropic/claude-opus-4-8","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/anthropic/claude-sonnet-4-6","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"anthropic/claude-sonnet-4-6","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.9999999999999997e-6,"output_cost_per_token":0.000015},"list_pricing":{"input_cost_per_token":2.9999999999999997e-6,"output_cost_per_token":0.000015},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/ByteDance/Seed-1.8","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"ByteDance/Seed-1.8","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":5.0000000000000004e-8},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":5.0000000000000004e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/ByteDance/Seed-2.0-code","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"ByteDance/Seed-2.0-code","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.9999999999999997e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.9999999999999997e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/ByteDance/Seed-2.0-mini","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"ByteDance/Seed-2.0-mini","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":2.0000000000000004e-8},"list_pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":2.0000000000000004e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/ByteDance/Seed-2.0-pro","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"ByteDance/Seed-2.0-pro","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.9999999999999997e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.9999999999999997e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/google/gemini-3.1-flash-lite","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"google/gemini-3.1-flash-lite","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1.4999999999999998e-6},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1.4999999999999998e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/google/gemini-3.1-pro","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"google/gemini-3.1-pro","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000011999999999999999},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000011999999999999999},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/google/gemini-3.5-flash","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"google/gemini-3.5-flash","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4999999999999998e-6,"output_cost_per_token":9e-6},"list_pricing":{"input_cost_per_token":1.4999999999999998e-6,"output_cost_per_token":9e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/google/gemma-4-31B-it-turbo","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"google/gemma-4-31B-it-turbo","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":3.4e-7,"cache_read_input_token_cost":5.000000039999999e-8},"list_pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":3.4e-7,"cache_read_input_token_cost":5.000000039999999e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/MiniMaxAI/MiniMax-M2.7-Turbo","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"MiniMaxAI/MiniMax-M2.7-Turbo","context_length":196608,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.8e-7,"output_cost_per_token":1.7e-6,"cache_read_input_token_cost":7.000000140000002e-8},"list_pricing":{"input_cost_per_token":3.8e-7,"output_cost_per_token":1.7e-6,"cache_read_input_token_cost":7.000000140000002e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.000000000000001e-7},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.000000000000001e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/nvidia/Nemotron-Content-Safety-3.5","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"nvidia/Nemotron-Content-Safety-3.5","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":2.0000000000000002e-7},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":2.0000000000000002e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8.5e-8,"output_cost_per_token":4.0000000000000003e-7},"list_pricing":{"input_cost_per_token":8.5e-8,"output_cost_per_token":4.0000000000000003e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/openai/gpt-oss-120b-Turbo","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"openai/gpt-oss-120b-Turbo","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.7-Max","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.7-Max","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":7.5e-6,"cache_read_input_token_cost":5e-7},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":7.5e-6,"cache_read_input_token_cost":5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-Max","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-Max","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":5.999999999999999e-6,"cache_read_input_token_cost":2.4e-7},"list_pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":5.999999999999999e-6,"cache_read_input_token_cost":2.4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-Max-Thinking","object":"model","created":1782119040,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-Max-Thinking","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":5.999999999999999e-6,"cache_read_input_token_cost":2.4e-7},"list_pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":5.999999999999999e-6,"cache_read_input_token_cost":2.4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"groq/qwen/qwen3.6-27b","object":"model","created":1782119040,"owned_by":"groq","model_name":"qwen/qwen3.6-27b","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":3e-6,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":3e-6,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/deepseek-ai/DeepSeek-V4-Pro","object":"model","created":1782119040,"owned_by":"nebius","model_name":"deepseek-ai/DeepSeek-V4-Pro","context_length":1048576,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":1.75e-6},"list_pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":1.75e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"nebius/nvidia/Cosmos3-Super-Reasoner","object":"model","created":1782119040,"owned_by":"nebius","model_name":"nvidia/Cosmos3-Super-Reasoner","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7,"cache_read_input_token_cost":1e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7,"cache_read_input_token_cost":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"nebius/nvidia/Nemotron-3-Nano-Omni","object":"model","created":1782119040,"owned_by":"nebius","model_name":"nvidia/Nemotron-3-Nano-Omni","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7,"cache_read_input_token_cost":6e-8},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7,"cache_read_input_token_cost":6e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"nebius/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B","object":"model","created":1782119040,"owned_by":"nebius","model_name":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7,"cache_read_input_token_cost":6e-8},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7,"cache_read_input_token_cost":6e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"nebius/openbmb/MiniCPM-V-4_5","object":"model","created":1782119040,"owned_by":"nebius","model_name":"openbmb/MiniCPM-V-4_5","context_length":32000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.58e-7,"output_cost_per_token":1.11e-6,"cache_read_input_token_cost":6.58e-7},"list_pricing":{"input_cost_per_token":6.58e-7,"output_cost_per_token":1.11e-6,"cache_read_input_token_cost":6.58e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"nebius/Qwen/Qwen3-235B-A22B-Instruct-2507","object":"model","created":1782119040,"owned_by":"nebius","model_name":"Qwen/Qwen3-235B-A22B-Instruct-2507","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":2e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-haiku-4-5-20251001-v1:0@eu","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-haiku-4-5-20251001-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":5.500000000000001e-6,"cache_read_input_token_cost":1.1e-7,"cache_creation_input_token_cost":1.3750000000000002e-6,"cache_creation_input_token_cost_above_1hr":2.2e-6},"list_pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":5.500000000000001e-6,"cache_read_input_token_cost":1.1e-7,"cache_creation_input_token_cost":1.3750000000000002e-6,"cache_creation_input_token_cost_above_1hr":2.2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-haiku-4-5-20251001-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-haiku-4-5-20251001-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":1e-7,"cache_creation_input_token_cost":1.25e-6,"cache_creation_input_token_cost_above_1hr":2e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":1e-7,"cache_creation_input_token_cost":1.25e-6,"cache_creation_input_token_cost_above_1hr":2e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/anthropic.claude-haiku-4-5-20251001-v1:0@us","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-haiku-4-5-20251001-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":5.500000000000001e-6,"cache_read_input_token_cost":1.1e-7,"cache_creation_input_token_cost":1.3750000000000002e-6,"cache_creation_input_token_cost_above_1hr":2.2e-6},"list_pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":5.500000000000001e-6,"cache_read_input_token_cost":1.1e-7,"cache_creation_input_token_cost":1.3750000000000002e-6,"cache_creation_input_token_cost_above_1hr":2.2e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-1-20250805-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-1-20250805-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.0000165,"output_cost_per_token":0.0000825,"cache_read_input_token_cost":1.65e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.000020625},"list_pricing":{"input_cost_per_token":0.0000165,"output_cost_per_token":0.0000825,"cache_read_input_token_cost":1.65e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.000020625},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-5-20251101-v1:0@eu","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-5-20251101-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-5-20251101-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-5-20251101-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-5-20251101-v1:0@us","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-5-20251101-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-6-v1@eu","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-6-v1","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-6-v1","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-6-v1","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-6-v1@us","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-6-v1","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/anthropic.claude-sonnet-4-5-20250929-v1:0@eu","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-sonnet-4-5-20250929-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":3.3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":4.125e-6,"input_cost_per_token_above_200k_tokens":6.6e-6,"output_cost_per_token_above_200k_tokens":0.000024750000000000002,"cache_creation_input_token_cost_above_1hr":6.6e-6,"cache_read_input_token_cost_above_200k_tokens":6.6e-7,"cache_creation_input_token_cost_above_200k_tokens":8.25e-6,"cache_creation_input_token_cost_above_1hr_above_200k_tokens":0.0000132},"list_pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":3.3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":4.125e-6,"input_cost_per_token_above_200k_tokens":6.6e-6,"output_cost_per_token_above_200k_tokens":0.000024750000000000002,"cache_creation_input_token_cost_above_1hr":6.6e-6,"cache_read_input_token_cost_above_200k_tokens":6.6e-7,"cache_creation_input_token_cost_above_200k_tokens":8.25e-6,"cache_creation_input_token_cost_above_1hr_above_200k_tokens":0.0000132},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-sonnet-4-5-20250929-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-sonnet-4-5-20250929-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"input_cost_per_token_above_200k_tokens":6e-6,"output_cost_per_token_above_200k_tokens":0.0000225,"cache_creation_input_token_cost_above_1hr":6e-6,"cache_read_input_token_cost_above_200k_tokens":6e-7,"cache_creation_input_token_cost_above_200k_tokens":7.5e-6,"cache_creation_input_token_cost_above_1hr_above_200k_tokens":0.000012},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"input_cost_per_token_above_200k_tokens":6e-6,"output_cost_per_token_above_200k_tokens":0.0000225,"cache_creation_input_token_cost_above_1hr":6e-6,"cache_read_input_token_cost_above_200k_tokens":6e-7,"cache_creation_input_token_cost_above_200k_tokens":7.5e-6,"cache_creation_input_token_cost_above_1hr_above_200k_tokens":0.000012},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/anthropic.claude-sonnet-4-5-20250929-v1:0@us","object":"model","created":1781775085,"owned_by":"amazon","model_name":"anthropic.claude-sonnet-4-5-20250929-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":3.3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":4.125e-6,"input_cost_per_token_above_200k_tokens":6.6e-6,"output_cost_per_token_above_200k_tokens":0.000024750000000000002,"cache_creation_input_token_cost_above_1hr":6.6e-6,"cache_read_input_token_cost_above_200k_tokens":6.6e-7,"cache_creation_input_token_cost_above_200k_tokens":8.25e-6,"cache_creation_input_token_cost_above_1hr_above_200k_tokens":0.0000132},"list_pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":3.3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":4.125e-6,"input_cost_per_token_above_200k_tokens":6.6e-6,"output_cost_per_token_above_200k_tokens":0.000024750000000000002,"cache_creation_input_token_cost_above_1hr":6.6e-6,"cache_read_input_token_cost_above_200k_tokens":6.6e-7,"cache_creation_input_token_cost_above_200k_tokens":8.25e-6,"cache_creation_input_token_cost_above_1hr_above_200k_tokens":0.0000132},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/deepseek.r1-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"deepseek.r1-v1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.35e-6,"output_cost_per_token":5.4e-6},"list_pricing":{"input_cost_per_token":1.35e-6,"output_cost_per_token":5.4e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/meta.llama3-1-70b-instruct-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"meta.llama3-1-70b-instruct-v1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.9e-7,"output_cost_per_token":9.9e-7},"list_pricing":{"input_cost_per_token":9.9e-7,"output_cost_per_token":9.9e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/meta.llama3-1-8b-instruct-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"meta.llama3-1-8b-instruct-v1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-7,"output_cost_per_token":2.2e-7},"list_pricing":{"input_cost_per_token":2.2e-7,"output_cost_per_token":2.2e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/meta.llama3-3-70b-instruct-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"meta.llama3-3-70b-instruct-v1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.2e-7,"output_cost_per_token":7.2e-7},"list_pricing":{"input_cost_per_token":7.2e-7,"output_cost_per_token":7.2e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/mistral.pixtral-large-2502-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"mistral.pixtral-large-2502-v1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.pixtral-large-2502-v1:0@us","object":"model","created":1781775085,"owned_by":"amazon","model_name":"mistral.pixtral-large-2502-v1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/writer.palmyra-x4-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"writer.palmyra-x4-v1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/writer.palmyra-x5-v1:0","object":"model","created":1781775085,"owned_by":"amazon","model_name":"writer.palmyra-x5-v1:0","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":6e-6},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":6e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"vertex/gemini-3.1-flash-image","object":"model","created":1781754065,"owned_by":"vertex","model_name":"gemini-3.1-flash-image","context_length":131072,"description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["image","text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"output_cost_per_image_token":0.00006,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014}},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"output_cost_per_image_token":0.00006,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014}},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3-pro-image","object":"model","created":1781754054,"owned_by":"vertex","model_name":"gemini-3-pro-image","context_length":65536,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["image","text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"output_cost_per_image_token":0.00012,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"output_cost_per_image_token":0.00012,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"cloudflare/@cf/zai-org/glm-5.2","object":"model","created":1781631930,"owned_by":"cloudflare","model_name":"@cf/zai-org/glm-5.2","context_length":262144,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.6e-7},"list_pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.6e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/zai-org/GLM-5.2","object":"model","created":1781631930,"owned_by":"deepinfra","model_name":"zai-org/GLM-5.2","context_length":1048576,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.499999999999999e-7,"output_cost_per_token":2.4e-6,"cache_read_input_token_cost":1.400000025e-7},"list_pricing":{"input_cost_per_token":7.499999999999999e-7,"output_cost_per_token":2.4e-6,"cache_read_input_token_cost":1.400000025e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/glm-5.2","object":"model","created":1781631930,"owned_by":"qwen","model_name":"glm-5.2","context_length":1048576,"description":"GLM-5.2 is the latest flagship model from Zhipu AI, designed for long-horizon tasks with support for an ultra-long 1M context window. It features powerful logical reasoning, long-text comprehension, and code generation capabilities, balancing performance with inference efficiency. It excels across multi-task benchmarks and is well-suited for intelligent interaction, enterprise applications, and development assistance scenarios.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.1e-7,"output_cost_per_token":2.86e-6,"cache_read_input_token_cost":1.8200000000000002e-7},"list_pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.8e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/zai-org/GLM-5.2","object":"model","created":1781631930,"owned_by":"together_ai","model_name":"zai-org/GLM-5.2","context_length":512000,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.5999999999999995e-7},"list_pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.5999999999999995e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"zai/glm-5.2","object":"model","created":1781631930,"owned_by":"zai","model_name":"glm-5.2","context_length":1048576,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.6e-7},"list_pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.6e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"cloudflare/@cf/moonshotai/kimi-k2.7-code","object":"model","created":1781266361,"owned_by":"cloudflare","model_name":"@cf/moonshotai/kimi-k2.7-code","context_length":262144,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.499999999999999e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.9e-7},"list_pricing":{"input_cost_per_token":9.499999999999999e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.9e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/moonshotai/Kimi-K2.7-Code","object":"model","created":1781266361,"owned_by":"deepinfra","model_name":"moonshotai/Kimi-K2.7-Code","context_length":262144,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.8e-7,"output_cost_per_token":3.4e-6,"cache_read_input_token_cost":1.3599999999999997e-7},"list_pricing":{"input_cost_per_token":6.8e-7,"output_cost_per_token":3.4e-6,"cache_read_input_token_cost":1.3599999999999997e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"moonshot/kimi-k2.7-code","object":"model","created":1781266361,"owned_by":"moonshot","model_name":"kimi-k2.7-code","context_length":262144,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.9e-7},"list_pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.9e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"nebius/moonshotai/Kimi-K2.7-Code","object":"model","created":1781266361,"owned_by":"nebius","model_name":"moonshotai/Kimi-K2.7-Code","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":9.5e-7},"list_pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":9.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/kimi-k2.7-code","object":"model","created":1781266361,"owned_by":"qwen","model_name":"kimi-k2.7-code","context_length":262144,"description":"kimi-k2.7-code is Kimi's most intelligent coding model to date. It follows instructions more reliably over long contexts and completes programming tasks with higher success rates. It supports text, image, and video inputs, along with thinking mode, conversation, and agent tasks.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.174999999999999e-7,"output_cost_per_token":2.6e-6,"cache_read_input_token_cost":1.2350000000000001e-7},"list_pricing":{"input_cost_per_token":9.499999999999999e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.9e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/moonshotai/Kimi-K2.7-Code","object":"model","created":1781266361,"owned_by":"together_ai","model_name":"moonshotai/Kimi-K2.7-Code","context_length":262144,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.499999999999999e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.9e-7},"list_pricing":{"input_cost_per_token":9.499999999999999e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.9e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-4-nano","object":"model","created":1781107944,"owned_by":"databricks","model_name":"databricks-gpt-5-4-nano","context_length":272000,"description":"GPT-5.4 Nano is a lightweight, cost-optimized large language model developed by OpenAI. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.9999000000000003e-7,"output_cost_per_token":1.24999e-6,"input_dbu_cost_per_token":2.857e-6,"output_dbu_cost_per_token":0.000017857},"list_pricing":{"input_cost_per_token":1.9999000000000003e-7,"output_cost_per_token":1.24999e-6,"input_dbu_cost_per_token":2.857e-6,"output_dbu_cost_per_token":0.000017857},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-4-mini","object":"model","created":1781107854,"owned_by":"databricks","model_name":"databricks-gpt-5-4-mini","context_length":272000,"description":"GPT-5.4 Mini is a cost-optimized large language model developed by OpenAI. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4999600000000002e-6,"output_cost_per_token":9.000040000000001e-6,"input_dbu_cost_per_token":0.000010714,"output_dbu_cost_per_token":0.000064286},"list_pricing":{"input_cost_per_token":1.4999600000000002e-6,"output_cost_per_token":9.000040000000001e-6,"input_dbu_cost_per_token":0.000010714,"output_dbu_cost_per_token":0.000064286},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-4","object":"model","created":1781107711,"owned_by":"databricks","model_name":"databricks-gpt-5-4","context_length":272000,"description":"GPT-5.4 is a general purpose large language model with reasoning capabilities developed by OpenAI. It delivers improved performance on complex tasks with enhanced accuracy and more deliberate scaffolded reasoning. This model supports multimodal inputs and features a 400K total token context window with 128K maximum output tokens. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.49998e-6,"output_cost_per_token":0.00002250003,"input_dbu_cost_per_token":0.000035714,"output_dbu_cost_per_token":0.000214286},"list_pricing":{"input_cost_per_token":2.49998e-6,"output_cost_per_token":0.00002250003,"input_dbu_cost_per_token":0.000035714,"output_dbu_cost_per_token":0.000214286},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-5","object":"model","created":1781107530,"owned_by":"databricks","model_name":"databricks-gpt-5-5","context_length":1050000,"description":"GPT-5.5 is OpenAI's strongest frontier model for agentic work in enterprise, complex document reasoning, and long-horizon coding agents. GPT-5.5 also now powers Codex, OpenAI's coding agent. This model supports multimodal inputs and features a 400K total token context window with 128K maximum output tokens. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000012500040000000002,"output_cost_per_token":0.00004499999},"list_pricing":{"input_cost_per_token":0.000012500040000000002,"output_cost_per_token":0.00004499999},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"anthropic/claude-fable-5","object":"model","created":1781007515,"owned_by":"anthropic","model_name":"claude-fable-5","context_length":1000000,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005,"cache_read_input_token_cost":1e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.0000125,"cache_creation_input_token_cost_above_1hr":0.00002},"list_pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005,"cache_read_input_token_cost":1e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.0000125,"cache_creation_input_token_cost_above_1hr":0.00002},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-fable-5@eu","object":"model","created":1781007515,"owned_by":"vertex","model_name":"claude-fable-5","context_length":1000000,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000011000000000000001,"output_cost_per_token":0.00005500000000000001,"cache_read_input_token_cost":1.1e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.000013750000000000002,"cache_creation_input_token_cost_above_1hr":0.000022000000000000003},"list_pricing":{"input_cost_per_token":0.000011000000000000001,"output_cost_per_token":0.00005500000000000001,"cache_read_input_token_cost":1.1e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.000013750000000000002,"cache_creation_input_token_cost_above_1hr":0.000022000000000000003},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/claude-fable-5","object":"model","created":1781007515,"owned_by":"vertex","model_name":"claude-fable-5","context_length":1000000,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005,"cache_read_input_token_cost":1e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.0000125,"cache_creation_input_token_cost_above_1hr":0.00002},"list_pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005,"cache_read_input_token_cost":1e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.0000125,"cache_creation_input_token_cost_above_1hr":0.00002},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"databricks/databricks-claude-sonnet-4-6","object":"model","created":1780933650,"owned_by":"databricks","model_name":"databricks-claude-sonnet-4-6","context_length":1000000,"description":"Claude Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with memory, polished document creation, and confident computer use for web QA and workflow automation. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.9999900000000002e-6,"output_cost_per_token":0.000015000020000000002,"input_dbu_cost_per_token":0.000042857,"output_dbu_cost_per_token":0.000214286},"list_pricing":{"input_cost_per_token":2.9999900000000002e-6,"output_cost_per_token":0.000015000020000000002,"input_dbu_cost_per_token":0.000042857,"output_dbu_cost_per_token":0.000214286},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-claude-opus-4-6","object":"model","created":1780933486,"owned_by":"databricks","model_name":"databricks-claude-opus-4-6","context_length":1000000,"description":"Claude Opus 4.6 is Anthropic's most capable hybrid reasoning model with adaptive thinking capabilities. This model introduces a new max effort level for the most demanding tasks, with high effort set as the default for optimal performance. Claude Opus 4.6 excels at complex reasoning, deep analysis, code generation, research, and sophisticated multi-step workflows. It features a 1 million token context window, making it ideal for enterprise applications that require both extensive analysis and comprehensive outputs. This endpoint is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.000030000000001e-6,"output_cost_per_token":0.00002500001,"input_dbu_cost_per_token":0.000071429,"output_dbu_cost_per_token":0.000357143},"list_pricing":{"input_cost_per_token":5.000030000000001e-6,"output_cost_per_token":0.00002500001,"input_dbu_cost_per_token":0.000071429,"output_dbu_cost_per_token":0.000357143},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-claude-opus-4-7","object":"model","created":1780933448,"owned_by":"databricks","model_name":"databricks-claude-opus-4-7","context_length":1000000,"description":"Claude Opus 4.7 is Anthropic's most capable hybrid reasoning model, advancing the Opus series with improved accuracy, efficiency, and enhanced vision capabilities. This model delivers stronger performance on complex extraction and agentic reasoning tasks while using fewer output tokens than its predecessor. Claude Opus 4.7 features a 1 million token context window and increased image resolution support, making it ideal for enterprise applications that require deep analysis, document understanding, and sophisticated multi-step workflows. This model is hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.000030000000001e-6,"output_cost_per_token":0.00002500001},"list_pricing":{"input_cost_per_token":5.000030000000001e-6,"output_cost_per_token":0.00002500001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-claude-opus-4-8","object":"model","created":1780933263,"owned_by":"databricks","model_name":"databricks-claude-opus-4-8","context_length":1000000,"description":"Claude Opus 4.8 is Anthropic's next-generation Opus model, hosted by Databricks.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.000030000000001e-6,"output_cost_per_token":0.00002500001},"list_pricing":{"input_cost_per_token":5.000030000000001e-6,"output_cost_per_token":0.00002500001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"lilac/google/gemma-4-31b-it","object":"model","created":1780919275,"owned_by":"lilac","model_name":"google/gemma-4-31b-it","context_length":262144,"description":"","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8.25e-8,"output_cost_per_token":2.625e-7,"cache_read_input_token_cost":8.25e-8},"list_pricing":{"input_cost_per_token":1.1e-7,"output_cost_per_token":3.5e-7,"cache_read_input_token_cost":1.1e-7},"discount":0.25,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"lilac/moonshotai/kimi-k2.6","object":"model","created":1780919275,"owned_by":"lilac","model_name":"moonshotai/kimi-k2.6","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.25e-7,"output_cost_per_token":2.625e-6,"cache_read_input_token_cost":1.2000000000000002e-7},"list_pricing":{"input_cost_per_token":7e-7,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":1.6e-7},"discount":0.25,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"mistral/ministral-8b-latest","object":"model","created":1780792123,"owned_by":"mistral","model_name":"ministral-8b-latest","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"cerebras/zai-glm-4.7","object":"model","created":1780673223,"owned_by":"cerebras","model_name":"zai-glm-4.7","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.25e-6,"output_cost_per_token":2.75e-6},"list_pricing":{"input_cost_per_token":2.25e-6,"output_cost_per_token":2.75e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"mistral/codestral-latest","object":"model","created":1780673223,"owned_by":"mistral","model_name":"codestral-latest","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-tiny-latest","object":"model","created":1780673223,"owned_by":"mistral","model_name":"mistral-tiny-latest","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2.5e-7},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-4-0613","object":"model","created":1780673223,"owned_by":"openai","model_name":"gpt-4-0613","context_length":8192,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00006},"list_pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00006},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"ovhcloud/Qwen3.5-397B-A17B","object":"model","created":1780673223,"owned_by":"ovhcloud","model_name":"Qwen3.5-397B-A17B","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":7.1e-7,"output_cost_per_token":4.25e-6},"list_pricing":{"input_cost_per_token":7.1e-7,"output_cost_per_token":4.25e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ovhcloud/Qwen3.5-9B","object":"model","created":1780673223,"owned_by":"ovhcloud","model_name":"Qwen3.5-9B","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.2e-7,"output_cost_per_token":1.8e-7},"list_pricing":{"input_cost_per_token":1.2e-7,"output_cost_per_token":1.8e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ovhcloud/Qwen3.6-27B","object":"model","created":1780673223,"owned_by":"ovhcloud","model_name":"Qwen3.6-27B","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4.7e-7,"output_cost_per_token":3.19e-6},"list_pricing":{"input_cost_per_token":4.7e-7,"output_cost_per_token":3.19e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ovhcloud/Qwen3-Coder-30B-A3B-Instruct","object":"model","created":1780673223,"owned_by":"ovhcloud","model_name":"Qwen3-Coder-30B-A3B-Instruct","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":2.6e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":2.6e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"nebius/nvidia/Llama-3_1-Nemotron-Ultra-253B-v1","object":"model","created":1780673222,"owned_by":"nebius","model_name":"nvidia/Llama-3_1-Nemotron-Ultra-253B-v1","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":1.8e-6,"cache_read_input_token_cost":6e-7},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":1.8e-6,"cache_read_input_token_cost":6e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/arize-ai/qwen-2-1.5b-instruct","object":"model","created":1780673222,"owned_by":"together_ai","model_name":"arize-ai/qwen-2-1.5b-instruct","context_length":32768,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":1.0000000000000001e-7},"list_pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":1.0000000000000001e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"together_ai/Qwen/Qwen2.5-7B-Instruct-Turbo","object":"model","created":1780673222,"owned_by":"together_ai","model_name":"Qwen/Qwen2.5-7B-Instruct-Turbo","context_length":32768,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"xai/grok-4.20-0309-non-reasoning","object":"model","created":1780673222,"owned_by":"xai","model_name":"grok-4.20-0309-non-reasoning","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":5e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":5e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/nemotron-3-ultra-550b-a55b","object":"model","created":1780551208,"owned_by":"deepinfra","model_name":"nemotron-3-ultra-550b-a55b","context_length":262144,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":1e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":1e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/nvidia/Nemotron-3-Ultra-550b-a55b","object":"model","created":1780551208,"owned_by":"nebius","model_name":"nvidia/Nemotron-3-Ultra-550b-a55b","context_length":1048576,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6,"cache_read_input_token_cost":1e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6,"cache_read_input_token_cost":1e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/nvidia/nemotron-3-ultra-550b-a55b","object":"model","created":1780551208,"owned_by":"together_ai","model_name":"nvidia/nemotron-3-ultra-550b-a55b","context_length":512288,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":3.6000000000000003e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":3.6000000000000003e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"qwen/qwen3.7-plus","object":"model","created":1780491783,"owned_by":"qwen","model_name":"qwen3.7-plus","context_length":1000000,"description":"Among the Qwen3.7 series, the cost-effective Plus model builds on its robust text capabilities while delivering a comprehensive upgrade to its vision‑language abilities, all while preserving its full‑stack agent‑level intelligence for coding, tool use, and productivity workflows. Its key distinguishing feature is multi‑modal interactive hybrid agent capabilities, enabling it to perceive real‑world scenes, read screens and interact with GUIs, generate code based on visual references, and perform end‑to‑end navigation within mobile apps.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.0400000000000002e-6,"cache_read_input_token_cost":5.2e-8,"cache_creation_input_token_cost":3.25e-7},{"range":[256000,1000000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.1199999999999998e-6,"cache_read_input_token_cost":1.56e-7,"cache_creation_input_token_cost":9.75e-7}],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.0400000000000002e-6,"cache_read_input_token_cost":5.2e-8,"cache_creation_input_token_cost":3.25e-7},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.6000000000000001e-6,"cache_read_input_token_cost":8e-8,"cache_creation_input_token_cost":5e-7},{"range":[256000,1000000],"input_cost_per_token":1.2e-6,"output_cost_per_token":4.8e-6,"cache_read_input_token_cost":2.4e-7,"cache_creation_input_token_cost":1.5e-6}],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.6000000000000001e-6,"cache_read_input_token_cost":8e-8,"cache_creation_input_token_cost":5e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.7-plus-2026-05-26","object":"model","created":1780491783,"owned_by":"qwen","model_name":"qwen3.7-plus-2026-05-26","context_length":1000000,"description":"Among the Qwen3.7 series, the cost-effective Plus model builds on its robust text capabilities while delivering a comprehensive upgrade to its vision‑language abilities, all while preserving its full‑stack agent‑level intelligence for coding, tool use, and productivity workflows. Its key distinguishing feature is multi‑modal interactive hybrid agent capabilities, enabling it to perceive real‑world scenes, read screens and interact with GUIs, generate code based on visual references, and perform end‑to‑end navigation within mobile apps.This version is a snapshot as of May 26, 2026.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.0400000000000002e-6,"cache_read_input_token_cost":5.2e-8,"cache_creation_input_token_cost":3.25e-7},{"range":[256000,1000000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.1199999999999998e-6,"cache_read_input_token_cost":1.56e-7,"cache_creation_input_token_cost":9.75e-7}],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.0400000000000002e-6,"cache_read_input_token_cost":5.2e-8,"cache_creation_input_token_cost":3.25e-7},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.6000000000000001e-6,"cache_read_input_token_cost":8e-8,"cache_creation_input_token_cost":5e-7},{"range":[256000,1000000],"input_cost_per_token":1.2e-6,"output_cost_per_token":4.8e-6,"cache_read_input_token_cost":2.4e-7,"cache_creation_input_token_cost":1.5e-6}],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.6000000000000001e-6,"cache_read_input_token_cost":8e-8,"cache_creation_input_token_cost":5e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/allenai/olmOCR-7B-0725-FP8","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"allenai/olmOCR-7B-0725-FP8","context_length":16384,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.7e-7,"output_cost_per_token":1.5e-6},"list_pricing":{"input_cost_per_token":2.7e-7,"output_cost_per_token":1.5e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-R1","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-R1","context_length":163840,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":7e-7,"output_cost_per_token":2.4e-6},"list_pricing":{"input_cost_per_token":7e-7,"output_cost_per_token":2.4e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-R1-Distill-Qwen-32B","context_length":131072,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.7e-7,"output_cost_per_token":2.7e-7},"list_pricing":{"input_cost_per_token":2.7e-7,"output_cost_per_token":2.7e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-R1-Turbo","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-R1-Turbo","context_length":40960,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Llama-3.2-3B-Instruct","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"meta-llama/Llama-3.2-3B-Instruct","context_length":131072,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":2e-8},"list_pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":2e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Llama-Guard-3-8B","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"meta-llama/Llama-Guard-3-8B","context_length":131072,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5.5e-8,"output_cost_per_token":5.5e-8},"list_pricing":{"input_cost_per_token":5.5e-8,"output_cost_per_token":5.5e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/microsoft/WizardLM-2-8x22B","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"microsoft/WizardLM-2-8x22B","context_length":65536,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4.8e-7,"output_cost_per_token":4.8e-7},"list_pricing":{"input_cost_per_token":4.8e-7,"output_cost_per_token":4.8e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/moonshotai/Kimi-K2-Instruct","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"moonshotai/Kimi-K2-Instruct","context_length":131072,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2e-6},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen2.5-7B-Instruct","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"Qwen/Qwen2.5-7B-Instruct","context_length":32768,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":1e-7},"list_pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":1e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-235B-A22B","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-235B-A22B","context_length":40960,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.8e-7,"output_cost_per_token":5.4e-7},"list_pricing":{"input_cost_per_token":1.8e-7,"output_cost_per_token":5.4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/QwQ-32B","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"Qwen/QwQ-32B","context_length":131072,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":4e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/zai-org/GLM-4.5","object":"model","created":1780377273,"owned_by":"deepinfra","model_name":"zai-org/GLM-4.5","context_length":131072,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":1.6e-6},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":1.6e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/MiniMaxAI/MiniMax-M3","object":"model","created":1780245374,"owned_by":"deepinfra","model_name":"MiniMaxAI/MiniMax-M3","context_length":524288,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.8e-7,"output_cost_per_token":1.1e-6,"cache_read_input_token_cost":5.6000000000000005e-8},"list_pricing":{"input_cost_per_token":2.8e-7,"output_cost_per_token":1.1e-6,"cache_read_input_token_cost":5.6000000000000005e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"minimax/MiniMax-M3","object":"model","created":1780245374,"owned_by":"minimax","model_name":"MiniMax-M3","context_length":524288,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":6e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":6e-8},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"together_ai/MiniMaxAI/MiniMax-M3","object":"model","created":1780245374,"owned_by":"together_ai","model_name":"MiniMaxAI/MiniMax-M3","context_length":524288,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":6.000000000000001e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":6.000000000000001e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/stepfun-ai/Step-3.7-Flash","object":"model","created":1779985069,"owned_by":"deepinfra","model_name":"stepfun-ai/Step-3.7-Flash","context_length":262144,"description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":1.15e-6,"cache_read_input_token_cost":4.000000000000001e-8},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":1.15e-6,"cache_read_input_token_cost":4.000000000000001e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-8@eu","object":"model","created":1779905091,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-8","context_length":1000000,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-8","object":"model","created":1779905091,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-8","context_length":1000000,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-8@us","object":"model","created":1779905091,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-8","context_length":1000000,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"anthropic/claude-opus-4-8","object":"model","created":1779905091,"owned_by":"anthropic","model_name":"claude-opus-4-8","context_length":1000000,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-opus-4-8@eu","object":"model","created":1779905091,"owned_by":"vertex","model_name":"claude-opus-4-8","context_length":1000000,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/claude-opus-4-8","object":"model","created":1779905091,"owned_by":"vertex","model_name":"claude-opus-4-8","context_length":1000000,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-opus-4-8@us","object":"model","created":1779905091,"owned_by":"vertex","model_name":"claude-opus-4-8","context_length":1000000,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.7-max","object":"model","created":1779376861,"owned_by":"qwen","model_name":"qwen3.7-max","context_length":1000000,"description":"The Max model, the largest and most capable in the Qwen3.7 series, currently offers a pure‑text‑only interface for public experimentation. Qwen3.7 is a next‑generation flagship model designed for the agent‑centric era, with its core strengths lying in the breadth and depth of its agent‑level capabilities: it excels at programming, office and productivity tasks, and long‑term autonomous execution.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.6250000000000001e-6,"output_cost_per_token":4.875e-6,"cache_read_input_token_cost":3.25e-7,"cache_creation_input_token_cost":2.0312500000000002e-6},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":7.5e-6,"cache_read_input_token_cost":5e-7,"cache_creation_input_token_cost":3.125e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.7-max-2026-05-17","object":"model","created":1779376861,"owned_by":"qwen","model_name":"qwen3.7-max-2026-05-17","context_length":1000000,"description":"The early version of the Max model in the Qwen3.7 series supports only the reasoning mode and makes its plain-text capabilities available for experimentation. Qwen3.7 is a next‑generation flagship model designed for the agent‑centric era, with major improvements in coding and agent‑level capabilities. This release is a snapshot as of May 17, 2026.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.6250000000000001e-6,"output_cost_per_token":4.875e-6},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":7.5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.7-max-2026-05-20","object":"model","created":1779376861,"owned_by":"qwen","model_name":"qwen3.7-max-2026-05-20","context_length":1000000,"description":"The Max model, the largest and most capable in the Qwen3.7 series, currently offers a pure‑text‑only interface for public experimentation. Qwen3.7 is a next‑generation flagship model designed for the agent‑centric era, with its core strengths lying in the breadth and depth of its agent‑level capabilities: it excels at programming, office and productivity tasks, and long‑term autonomous execution.This version is a snapshot as of May 20, 2026.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.6250000000000001e-6,"output_cost_per_token":4.875e-6,"cache_creation_input_token_cost":2.0312500000000002e-6},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":7.5e-6,"cache_creation_input_token_cost":3.125e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.7-max-2026-06-08","object":"model","created":1779376861,"owned_by":"qwen","model_name":"qwen3.7-max-2026-06-08","context_length":1000000,"description":"The Max model, the largest and most capable in the Qwen3.7 series, has added visual‑modal understanding compared to the May 20 snapshot, enabling it to perceive real‑world scenes and supporting multimodal interactive hybrid agent capabilities. This version is based on a snapshot taken on June 8, 2026.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.6250000000000001e-6,"output_cost_per_token":4.875e-6,"cache_read_input_token_cost":3.25e-7,"cache_creation_input_token_cost":2.0312500000000002e-6},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":7.5e-6,"cache_read_input_token_cost":5e-7,"cache_creation_input_token_cost":3.125e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.7-max-preview","object":"model","created":1779376861,"owned_by":"qwen","model_name":"qwen3.7-max-preview","context_length":1000000,"description":"The Max model, the largest and most capable variant in the Qwen3.7 series, is available as a preview and supports only thinking mode, offering a pure text‑only interface for experimentation. It is primarily optimized for general‑purpose conversational use cases, such as knowledge‑based question answering, instruction following, and creative writing.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.6250000000000001e-6,"output_cost_per_token":4.875e-6},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":7.5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"xai/grok-build-0.1","object":"model","created":1779298123,"owned_by":"xai","model_name":"grok-build-0.1","context_length":256000,"description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":2e-6,"output_cost_per_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":2e-6,"output_cost_per_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"google/gemini-3.5-flash","object":"model","created":1779193800,"owned_by":"google","model_name":"gemini-3.5-flash","context_length":1048576,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":9e-6,"input_cost_per_audio_token":3e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":9e-6,"cache_read_input_audio_token_cost":3e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":9e-6,"input_cost_per_audio_token":3e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":9e-6,"cache_read_input_audio_token_cost":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3.5-flash@eu","object":"model","created":1779193800,"owned_by":"vertex","model_name":"gemini-3.5-flash","context_length":1048576,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":9e-6,"input_cost_per_audio_token":3e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":9e-6,"cache_read_input_audio_token_cost":3e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":9e-6,"input_cost_per_audio_token":3e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":9e-6,"cache_read_input_audio_token_cost":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3.5-flash","object":"model","created":1779193800,"owned_by":"vertex","model_name":"gemini-3.5-flash","context_length":1048576,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":9e-6,"input_cost_per_audio_token":3e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":9e-6,"cache_read_input_audio_token_cost":3e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":9e-6,"input_cost_per_audio_token":3e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":9e-6,"cache_read_input_audio_token_cost":3e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/gemini-3.5-flash@us","object":"model","created":1779193800,"owned_by":"vertex","model_name":"gemini-3.5-flash","context_length":1048576,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":9e-6,"input_cost_per_audio_token":3e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":9e-6,"cache_read_input_audio_token_cost":3e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":9e-6,"input_cost_per_audio_token":3e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":9e-6,"cache_read_input_audio_token_cost":3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4.3-latest","object":"model","created":1778172174,"owned_by":"xai","model_name":"grok-4.3-latest","context_length":1000000,"description":"","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":5e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":5e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"google/gemini-3.1-flash-lite","object":"model","created":1778168828,"owned_by":"google","model_name":"gemini-3.1-flash-lite","context_length":1048576,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-3.1-flash-lite","capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6,"input_cost_per_audio_token":5e-7,"input_cost_per_image_token":2.5e-7,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":1.5e-6,"cache_read_input_audio_token_cost":5e-8},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6,"input_cost_per_audio_token":5e-7,"input_cost_per_image_token":2.5e-7,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":1.5e-6,"cache_read_input_audio_token_cost":5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"google/gemini-3.1-flash-lite-preview","object":"model","created":1778168828,"owned_by":"google","model_name":"gemini-3.1-flash-lite-preview","context_length":1048576,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","source":"https://ai.google.dev/gemini-api/docs/models","capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6,"input_cost_per_audio_token":5e-7,"input_cost_per_image_token":2.5e-7,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":1.5e-6,"cache_read_input_audio_token_cost":5e-8},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6,"input_cost_per_audio_token":5e-7,"input_cost_per_image_token":2.5e-7,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":1.5e-6,"cache_read_input_audio_token_cost":5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/mistral-medium-3.5-128b","object":"model","created":1778158991,"owned_by":"scaleway","model_name":"mistral-medium-3.5-128b","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.7521501802962534e-6,"output_cost_per_token":8.760750901481266e-6},"list_pricing":{"input_cost_per_token":1.7521501802962534e-6,"output_cost_per_token":8.760750901481266e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/qwen3.6-35b-a3b","object":"model","created":1778158991,"owned_by":"scaleway","model_name":"qwen3.6-35b-a3b","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.920250300493756e-7,"output_cost_per_token":1.7521501802962534e-6},"list_pricing":{"input_cost_per_token":2.920250300493756e-7,"output_cost_per_token":1.7521501802962534e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"xai/grok-4.3","object":"model","created":1777591821,"owned_by":"xai","model_name":"grok-4.3","context_length":1000000,"description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":5e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":5e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"mistral/mistral-medium-3-5","object":"model","created":1777570439,"owned_by":"mistral","model_name":"mistral-medium-3-5","context_length":262144,"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-medium-3.5","object":"model","created":1777570439,"owned_by":"mistral","model_name":"mistral-medium-3.5","context_length":262144,"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/gemma-4-26b-a4b-it","object":"model","created":1777469299,"owned_by":"scaleway","model_name":"gemma-4-26b-a4b-it","context_length":256000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.920250300493756e-7,"output_cost_per_token":5.840500600987512e-7},"list_pricing":{"input_cost_per_token":2.920250300493756e-7,"output_cost_per_token":5.840500600987512e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/codestral-2405","object":"model","created":1777389105,"owned_by":"mistral","model_name":"codestral-2405","context_length":32000,"description":"","source":"https://docs.mistral.ai/capabilities/code_generation/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-large-2402","object":"model","created":1777389105,"owned_by":"mistral","model_name":"mistral-large-2402","context_length":32000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4e-6,"output_cost_per_token":0.000012},"list_pricing":{"input_cost_per_token":4e-6,"output_cost_per_token":0.000012},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-medium-2312","object":"model","created":1777389105,"owned_by":"mistral","model_name":"mistral-medium-2312","context_length":32000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.7e-6,"output_cost_per_token":8.1e-6},"list_pricing":{"input_cost_per_token":2.7e-6,"output_cost_per_token":8.1e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-small","object":"model","created":1777389105,"owned_by":"mistral","model_name":"mistral-small","context_length":32000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-tiny","object":"model","created":1777389105,"owned_by":"mistral","model_name":"mistral-tiny","context_length":32000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2.5e-7},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/open-mistral-7b","object":"model","created":1777389105,"owned_by":"mistral","model_name":"open-mistral-7b","context_length":32000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2.5e-7},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/open-mixtral-8x22b","object":"model","created":1777389105,"owned_by":"mistral","model_name":"open-mixtral-8x22b","context_length":65336,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/open-mixtral-8x7b","object":"model","created":1777389105,"owned_by":"mistral","model_name":"open-mixtral-8x7b","context_length":32000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":7e-7,"output_cost_per_token":7e-7},"list_pricing":{"input_cost_per_token":7e-7,"output_cost_per_token":7e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/pixtral-12b-2409","object":"model","created":1777389105,"owned_by":"mistral","model_name":"pixtral-12b-2409","context_length":128000,"description":"","source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.5-plus","object":"model","created":1777261368,"owned_by":"qwen","model_name":"qwen3.5-plus","context_length":1000000,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of task evaluations, the 3.5 series consistently demonstrates performance on par with state-of-the-art leading models. Compared to the 3 series, these models show a leap forward in both pure-text and multimodal capabilities.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.5599999999999999e-6,"cache_creation_input_token_cost":3.25e-7},{"range":[256000,1000000],"input_cost_per_token":3.25e-7,"output_cost_per_token":1.95e-6,"cache_creation_input_token_cost":4.0625000000000003e-7}],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.5599999999999999e-6,"cache_creation_input_token_cost":3.25e-7},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":2.4e-6,"cache_creation_input_token_cost":5e-7},{"range":[256000,1000000],"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"cache_creation_input_token_cost":6.25e-7}],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":2.4e-6,"cache_creation_input_token_cost":5e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.5-plus-2026-02-15","object":"model","created":1777261368,"owned_by":"qwen","model_name":"qwen3.5-plus-2026-02-15","context_length":1000000,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of task evaluations, the 3.5 series consistently demonstrates performance on par with state-of-the-art leading models. Compared to the 3 series, these models show a leap forward in both pure-text and multimodal capabilities.This version is a snapshot as of February 15, 2026.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.5599999999999999e-6},{"range":[256000,1000000],"input_cost_per_token":3.25e-7,"output_cost_per_token":1.95e-6}],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.5599999999999999e-6},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":2.4e-6},{"range":[256000,1000000],"input_cost_per_token":5e-7,"output_cost_per_token":3e-6}],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":2.4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.5-plus-2026-04-20","object":"model","created":1777261368,"owned_by":"qwen","model_name":"qwen3.5-plus-2026-04-20","context_length":1000000,"description":"The Qwen3.5 native vision-language series Plus model has seen a substantial improvement in agentic coding capabilities compared to the February 15th snapshot. Inference speed has also been significantly enhanced, while its knowledge retention, reasoning ability, and long-context processing remain at a high level, making it well-suited for complex agent-based tasks. It is ideal for applications such as coding agents, production workflows, and high-throughput scenarios. This version is based on a snapshot taken on April 20, 2026.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.5599999999999999e-6,"cache_creation_input_token_cost":3.25e-7},{"range":[256000,1000000],"input_cost_per_token":3.25e-7,"output_cost_per_token":1.95e-6,"cache_creation_input_token_cost":4.0625000000000003e-7}],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.5599999999999999e-6,"cache_creation_input_token_cost":3.25e-7},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":2.4e-6,"cache_creation_input_token_cost":5e-7},{"range":[256000,1000000],"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"cache_creation_input_token_cost":6.25e-7}],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":2.4e-6,"cache_creation_input_token_cost":5e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.6-flash","object":"model","created":1777261362,"owned_by":"qwen","model_name":"qwen3.6-flash","context_length":1000000,"description":"The Qwen3.6 native vision-language Flash model series delivers a significant performance boost over the 3.5-Flash version. This model particularly excels in agentic coding capabilities, substantially outperforming its predecessor on multiple code-agent benchmarks, as well as in mathematical and code reasoning. In terms of vision, it features markedly improved spatial intelligence, with especially notable enhancements in object localization and object detection.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":1.625e-7,"output_cost_per_token":9.75e-7,"cache_creation_input_token_cost":2.0312500000000001e-7},{"range":[256000,1000000],"input_cost_per_token":6.5e-7,"output_cost_per_token":2.6e-6,"cache_creation_input_token_cost":8.125000000000001e-7}],"input_cost_per_token":1.625e-7,"output_cost_per_token":9.75e-7,"cache_creation_input_token_cost":2.0312500000000001e-7},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6,"cache_creation_input_token_cost":3.125e-7},{"range":[256000,1000000],"input_cost_per_token":1e-6,"output_cost_per_token":4e-6,"cache_creation_input_token_cost":1.25e-6}],"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6,"cache_creation_input_token_cost":3.125e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.6-flash-2026-04-16","object":"model","created":1777261362,"owned_by":"qwen","model_name":"qwen3.6-flash-2026-04-16","context_length":1000000,"description":"The Qwen3.6 native vision-language Flash model series delivers a significant performance boost over the 3.5-Flash version. This model particularly excels in agentic coding capabilities, substantially outperforming its predecessor on multiple code-agent benchmarks, as well as in mathematical and code reasoning. In terms of vision, it features markedly improved spatial intelligence, with especially notable enhancements in object localization and object detection.This release is based on a snapshot taken on April 16, 2026.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":1.625e-7,"output_cost_per_token":9.75e-7},{"range":[256000,1000000],"input_cost_per_token":6.5e-7,"output_cost_per_token":2.6e-6}],"input_cost_per_token":1.625e-7,"output_cost_per_token":9.75e-7},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6},{"range":[256000,1000000],"input_cost_per_token":1e-6,"output_cost_per_token":4e-6}],"input_cost_per_token":2.5e-7,"output_cost_per_token":1.5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.6-35B-A3B","object":"model","created":1777260255,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.6-35B-A3B","context_length":262144,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":9.499999999999999e-7},"list_pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":9.499999999999999e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.6-max-preview","object":"model","created":1777260242,"owned_by":"qwen","model_name":"qwen3.6-max-preview","context_length":262144,"description":"The Max model, the largest and most capable variant in the Qwen3.6 series, is now available in a preview version. At present, only its plain-text capabilities are open for experimentation. Compared with the previously released Qwen3-Max and Qwen3.6-Plus, this model features enhanced vibe coding abilities, more efficient coding agent execution, and significantly improved front-end development skills. Additionally, its long-tail knowledge retention has been further upgraded.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,128000],"input_cost_per_token":8.450000000000001e-7,"output_cost_per_token":5.07e-6,"cache_creation_input_token_cost":1.05625e-6},{"range":[128000,256000],"input_cost_per_token":1.3e-6,"output_cost_per_token":7.8e-6,"cache_creation_input_token_cost":1.6250000000000001e-6}],"input_cost_per_token":8.450000000000001e-7,"output_cost_per_token":5.07e-6,"cache_creation_input_token_cost":1.05625e-6},"list_pricing":{"tiered_pricing":[{"range":[0,128000],"input_cost_per_token":1.3e-6,"output_cost_per_token":7.8e-6,"cache_creation_input_token_cost":1.625e-6},{"range":[128000,256000],"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"cache_creation_input_token_cost":2.5e-6}],"input_cost_per_token":1.3e-6,"output_cost_per_token":7.8e-6,"cache_creation_input_token_cost":1.625e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.6-27B","object":"model","created":1777255064,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.6-27B","context_length":262144,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.2e-7,"output_cost_per_token":3.2000000000000003e-6},"list_pricing":{"input_cost_per_token":3.2e-7,"output_cost_per_token":3.2000000000000003e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.6-27b","object":"model","created":1777255064,"owned_by":"qwen","model_name":"qwen3.6-27b","context_length":262144,"description":"The Qwen3.6 27B native vision-language dense model builds upon the 3.5-27B version, with key improvements in agentic coding capabilities and enhanced STEM reasoning and inference skills. In the vision modality, it demonstrates significant advances in spatial intelligence, object localization, and detection, while video understanding, document OCR, and visual agent capabilities continue to improve steadily.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.8999999999999997e-7,"output_cost_per_token":2.34e-6},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":3.6000000000000003e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-5.5-pro","object":"model","created":1777051896,"owned_by":"openai","model_name":"gpt-5.5-pro","context_length":1050000,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"list_pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.5-pro-2026-04-23","object":"model","created":1777051896,"owned_by":"openai","model_name":"gpt-5.5-pro-2026-04-23","context_length":1050000,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"list_pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.5","object":"model","created":1777051893,"owned_by":"openai","model_name":"gpt-5.5","context_length":1050000,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1e-6},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.5-2026-04-23","object":"model","created":1777051893,"owned_by":"openai","model_name":"gpt-5.5-2026-04-23","context_length":1050000,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1e-6},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-V4-Pro","object":"model","created":1777000679,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-V4-Pro","context_length":1048576,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.2999999999999998e-6,"output_cost_per_token":2.5999999999999997e-6,"cache_read_input_token_cost":1.0000000399999998e-7},"list_pricing":{"input_cost_per_token":1.2999999999999998e-6,"output_cost_per_token":2.5999999999999997e-6,"cache_read_input_token_cost":1.0000000399999998e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepseek/deepseek-v4-pro","object":"model","created":1777000679,"owned_by":"deepseek","model_name":"deepseek-v4-pro","context_length":1048576,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":1.98e-6,"cache_read_input_token_cost":2.2e-8,"input_cost_per_token_cache_hit":3.625e-9,"cache_creation_input_token_cost":0.0},"list_pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":1.98e-6,"cache_read_input_token_cost":2.2e-8,"input_cost_per_token_cache_hit":3.625e-9,"cache_creation_input_token_cost":0.0},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/deepseek-v4-pro","object":"model","created":1777000679,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/deepseek-v4-pro","context_length":1048576,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.74e-6,"output_cost_per_token":3.48e-6,"cache_read_input_token_cost":1.45e-7},"list_pricing":{"input_cost_per_token":1.74e-6,"output_cost_per_token":3.48e-6,"cache_read_input_token_cost":1.45e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/deepseek-v4-pro","object":"model","created":1777000679,"owned_by":"fireworks_ai","model_name":"deepseek-v4-pro","context_length":1048576,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.74e-6,"output_cost_per_token":3.48e-6,"cache_read_input_token_cost":1.45e-7},"list_pricing":{"input_cost_per_token":1.74e-6,"output_cost_per_token":3.48e-6,"cache_read_input_token_cost":1.45e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/deepseek-v4-pro","object":"model","created":1777000679,"owned_by":"qwen","model_name":"deepseek-v4-pro","context_length":1000000,"description":"A flagship MoE large model with 1.6 trillion parameters and 49 billion activated parameters, natively supporting context lengths of up to one million tokens. Trained on a vast corpus of high-quality data, it excels in advanced mathematical reasoning, complex logical inference, specialized coding, and deep analysis of long-form text, making it well-suited for demanding applications such as cutting-edge research, sophisticated office workflows, and advanced AI agents.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5599999999999999e-6,"output_cost_per_token":3.1199999999999998e-6,"cache_read_input_token_cost":1.3000000000000003e-7},"list_pricing":{"input_cost_per_token":2.4e-6,"output_cost_per_token":4.8e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/deepseek-ai/DeepSeek-V4-Pro","object":"model","created":1777000679,"owned_by":"together_ai","model_name":"deepseek-ai/DeepSeek-V4-Pro","context_length":512000,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.74e-6,"output_cost_per_token":3.48e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"list_pricing":{"input_cost_per_token":1.74e-6,"output_cost_per_token":3.48e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-V4-Flash","object":"model","created":1777000666,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-V4-Flash","context_length":1048576,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":1.8e-7,"cache_read_input_token_cost":1.8e-8},"list_pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":1.8e-7,"cache_read_input_token_cost":1.8e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepseek/deepseek-v4-flash","object":"model","created":1777000666,"owned_by":"deepseek","model_name":"deepseek-v4-flash","context_length":1048576,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.2e-7,"output_cost_per_token":6.6e-7,"cache_read_input_token_cost":7e-9,"input_cost_per_token_cache_hit":2.8e-9,"cache_creation_input_token_cost":0.0},"list_pricing":{"input_cost_per_token":2.2e-7,"output_cost_per_token":6.6e-7,"cache_read_input_token_cost":7e-9,"input_cost_per_token_cache_hit":2.8e-9,"cache_creation_input_token_cost":0.0},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"qwen/deepseek-v4-flash","object":"model","created":1777000666,"owned_by":"qwen","model_name":"deepseek-v4-flash","context_length":1000000,"description":"A highly efficient, lightweight MoE model with 284 billion parameters in total and 13 billion activated parameters, natively supporting context windows of up to one million tokens. It offers fast inference speed, low latency, and cost-effective invocation, delivering well-balanced overall performance. Designed for high-concurrency, lightweight workloads, it is ideally suited for common, essential use cases such as everyday dialogue, content creation, basic RAG applications, and batch text processing.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":2.6000000000000005e-7,"cache_read_input_token_cost":2.6e-8},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":4e-8},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/XiaomiMiMo/MiMo-V2.5-Pro","object":"model","created":1776874273,"owned_by":"deepinfra","model_name":"XiaomiMiMo/MiMo-V2.5-Pro","context_length":1048576,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":2.9999999999999997e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":2.9999999999999997e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xiaomi/mimo-v2.5-pro","object":"model","created":1776874273,"owned_by":"xiaomi","model_name":"mimo-v2.5-pro","context_length":1048576,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4.35e-7,"output_cost_per_token":8.7e-7,"cache_read_input_token_cost":3.6e-9},"list_pricing":{"input_cost_per_token":4.35e-7,"output_cost_per_token":8.7e-7,"cache_read_input_token_cost":3.6e-9},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"deepinfra/XiaomiMiMo/MiMo-V2.5","object":"model","created":1776874269,"owned_by":"deepinfra","model_name":"XiaomiMiMo/MiMo-V2.5","context_length":262144,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","source":null,"capabilities":{"input_modalities":["text","image","audio","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":8.000000000000001e-8},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":8.000000000000001e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xiaomi/mimo-v2.5","object":"model","created":1776874269,"owned_by":"xiaomi","model_name":"mimo-v2.5","context_length":1048576,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","source":null,"capabilities":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.4e-7,"output_cost_per_token":2.8e-7,"cache_read_input_token_cost":2.8e-9},"list_pricing":{"input_cost_per_token":1.4e-7,"output_cost_per_token":2.8e-7,"cache_read_input_token_cost":2.8e-9},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"cloudflare/@cf/moonshotai/kimi-k2.6","object":"model","created":1776699402,"owned_by":"cloudflare","model_name":"@cf/moonshotai/kimi-k2.6","context_length":262144,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.499999999999999e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.6e-7},"list_pricing":{"input_cost_per_token":9.499999999999999e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.6e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/moonshotai/Kimi-K2.6","object":"model","created":1776699402,"owned_by":"deepinfra","model_name":"moonshotai/Kimi-K2.6","context_length":262144,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.499999999999999e-7,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":7.499999999999999e-7,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":1.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"moonshot/kimi-k2.6","object":"model","created":1776699402,"owned_by":"moonshot","model_name":"kimi-k2.6","context_length":262144,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.6e-7},"list_pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.6e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"nebius/moonshotai/Kimi-K2.6","object":"model","created":1776699402,"owned_by":"nebius","model_name":"moonshotai/Kimi-K2.6","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":9.5e-7},"list_pricing":{"input_cost_per_token":9.5e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":9.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/moonshotai/Kimi-K2.6","object":"model","created":1776699402,"owned_by":"together_ai","model_name":"moonshotai/Kimi-K2.6","context_length":262144,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","source":"https://platform.moonshot.ai/docs/guide/kimi-k2-6-quickstart","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"list_pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":2.0000000000000002e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"xai/grok-4.20-0309-reasoning","object":"model","created":1776575593,"owned_by":"xai","model_name":"grok-4.20-0309-reasoning","context_length":1000000,"description":null,"source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":5e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":5e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-7@eu","object":"model","created":1776351100,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-7","context_length":1000000,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-7","object":"model","created":1776351100,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-7","context_length":1000000,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/anthropic.claude-opus-4-7@us","object":"model","created":1776351100,"owned_by":"amazon","model_name":"anthropic.claude-opus-4-7","context_length":1000000,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"anthropic/claude-opus-4-7","object":"model","created":1776351100,"owned_by":"anthropic","model_name":"claude-opus-4-7","context_length":1000000,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-opus-4-7@eu","object":"model","created":1776351100,"owned_by":"vertex","model_name":"claude-opus-4-7","context_length":1000000,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/claude-opus-4-7","object":"model","created":1776351100,"owned_by":"vertex","model_name":"claude-opus-4-7","context_length":1000000,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-opus-4-7@us","object":"model","created":1776351100,"owned_by":"vertex","model_name":"claude-opus-4-7","context_length":1000000,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"list_pricing":{"input_cost_per_token":5.500000000000001e-6,"output_cost_per_token":0.000027500000000000004,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.875000000000001e-6,"cache_creation_input_token_cost_above_1hr":0.000011000000000000001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/zai-org/GLM-5.1","object":"model","created":1775578025,"owned_by":"deepinfra","model_name":"zai-org/GLM-5.1","context_length":202752,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.0500000000000001e-6,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":2.0500000500000001e-7},"list_pricing":{"input_cost_per_token":1.0500000000000001e-6,"output_cost_per_token":3.5e-6,"cache_read_input_token_cost":2.0500000500000001e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/zai-org/GLM-5.1","object":"model","created":1775578025,"owned_by":"nebius","model_name":"zai-org/GLM-5.1","context_length":202752,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":1.4e-6},"list_pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":1.4e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/glm-5.1","object":"model","created":1775578025,"owned_by":"qwen","model_name":"glm-5.1","context_length":202745,"description":"GLM-5.1 is a model developed by Zhipu AI, specifically designed for long-horizon tasks. It has 744 billion parameters, supports an ultra-long context of 200k tokens, and can generate up to 128k tokens in a single response. GLM-5.1 excels in logical reasoning, long-text understanding, and code generation, while balancing performance with inference efficiency. It delivers outstanding results across multiple multi-task benchmarks and is well-suited for applications such as intelligent human-computer interaction, enterprise solutions, and developer assistance.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.1e-7,"output_cost_per_token":2.86e-6,"cache_read_input_token_cost":1.6900000000000002e-7},"list_pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.6e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"zai/glm-5.1","object":"model","created":1775578025,"owned_by":"zai","model_name":"glm-5.1","context_length":202752,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.6e-7,"cache_creation_input_token_cost":0.0},"list_pricing":{"input_cost_per_token":1.4e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.6e-7,"cache_creation_input_token_cost":0.0},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"cloudflare/@cf/google/gemma-4-26b-a4b-it","object":"model","created":1775227989,"owned_by":"cloudflare","model_name":"@cf/google/gemma-4-26b-a4b-it","context_length":256000,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/google/gemma-4-26B-A4B-it","object":"model","created":1775227989,"owned_by":"deepinfra","model_name":"google/gemma-4-26B-A4B-it","context_length":262144,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3.4e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3.4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"google/gemma-4-26b-a4b-it","object":"model","created":1775227989,"owned_by":"google","model_name":"gemma-4-26b-a4b-it","context_length":262144,"description":"","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"list_pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/google/gemma-4-26b-a4b-it-maas@eu","object":"model","created":1775227989,"owned_by":"vertex","model_name":"google/gemma-4-26b-a4b-it-maas","context_length":262144,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":6.6e-7},"list_pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":6.6e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/google/gemma-4-26b-a4b-it-maas","object":"model","created":1775227989,"owned_by":"vertex","model_name":"google/gemma-4-26b-a4b-it-maas","context_length":262144,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/google/gemma-4-26b-a4b-it-maas@us","object":"model","created":1775227989,"owned_by":"vertex","model_name":"google/gemma-4-26b-a4b-it-maas","context_length":262144,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":6.6e-7},"list_pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":6.6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/minimax.minimax-m2.5","object":"model","created":1775193119,"owned_by":"amazon","model_name":"minimax.minimax-m2.5","context_length":1000000,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":1.44e-6},"list_pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":1.44e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/minimax.minimax-m2.5@us","object":"model","created":1775193119,"owned_by":"amazon","model_name":"minimax.minimax-m2.5","context_length":1000000,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":1.44e-6},"list_pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":1.44e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/nvidia.nemotron-super-3-120b","object":"model","created":1775193119,"owned_by":"amazon","model_name":"nvidia.nemotron-super-3-120b","context_length":256000,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6.5e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/nvidia.nemotron-super-3-120b@us","object":"model","created":1775193119,"owned_by":"amazon","model_name":"nvidia.nemotron-super-3-120b","context_length":256000,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6.5e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/zai.glm-5","object":"model","created":1775193119,"owned_by":"amazon","model_name":"zai.glm-5","context_length":200000,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3.2e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3.2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/google/gemma-4-31B-it","object":"model","created":1775148486,"owned_by":"deepinfra","model_name":"google/gemma-4-31B-it","context_length":262144,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3e-7,"output_cost_per_token":3.8e-7,"cache_read_input_token_cost":5e-8},"list_pricing":{"input_cost_per_token":1.3e-7,"output_cost_per_token":3.8e-7,"cache_read_input_token_cost":5e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"google/gemma-4-31b-it","object":"model","created":1775148486,"owned_by":"google","model_name":"gemma-4-31b-it","context_length":262144,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"list_pricing":{"input_cost_per_token":0.0,"output_cost_per_token":0.0},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/google/gemma-4-31B-it","object":"model","created":1775148486,"owned_by":"together_ai","model_name":"google/gemma-4-31B-it","context_length":262144,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.9e-7,"output_cost_per_token":9.7e-7},"list_pricing":{"input_cost_per_token":3.9e-7,"output_cost_per_token":9.7e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"together_ai/pearl-ai/gemma-4-31b-it","object":"model","created":1775148486,"owned_by":"together_ai","model_name":"pearl-ai/gemma-4-31b-it","context_length":262144,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.7999999999999997e-7,"output_cost_per_token":8.6e-7},"list_pricing":{"input_cost_per_token":2.7999999999999997e-7,"output_cost_per_token":8.6e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"bytedance/seed-2-0-mini-260215","object":"model","created":1775134437,"owned_by":"bytedance","model_name":"seed-2-0-mini-260215","context_length":256000,"description":"","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"supports_tools":true,"supports_vision":true,"supports_pdf_input":false,"supports_audio_input":false,"supports_video_input":true,"supports_context_cache":true,"supports_batch_inference":true,"supports_reasoning_effort":true,"supports_structured_output":true},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"cache_read_input_token_cost":2e-8,"input_cost_per_token_above_128k_tokens":2e-7,"output_cost_per_token_above_128k_tokens":8e-7,"cache_read_input_token_cost_above_128k_tokens":4e-8},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"cache_read_input_token_cost":2e-8,"input_cost_per_token_above_128k_tokens":2e-7,"output_cost_per_token_above_128k_tokens":8e-7,"cache_read_input_token_cost_above_128k_tokens":4e-8},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"bytedance/seed-2-0-lite-260228","object":"model","created":1775133977,"owned_by":"bytedance","model_name":"seed-2-0-lite-260228","context_length":256000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"supports_vision":true,"supports_video_input":true,"supports_responses_api":true,"supports_knowledge_base":true,"supports_batch_inference":true,"supports_online_inference":true,"supports_chat_completions_api":true},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":5e-7,"output_cost_per_token_above_128k_tokens":4e-6,"cache_read_input_token_cost_above_128k_tokens":1e-7},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":5e-7,"output_cost_per_token_above_128k_tokens":4e-6,"cache_read_input_token_cost_above_128k_tokens":1e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"qwen/qwen3.6-plus","object":"model","created":1775133557,"owned_by":"qwen","model_name":"qwen3.6-plus","context_length":1000000,"description":"The Qwen3.6 native vision-language Plus series models demonstrate exceptional performance on par with the current state-of-the-art models, with a significant improvement in overall results compared to the 3.5 series. The models have been markedly enhanced in code-related capabilities such as agentic coding, front-end programming, and Vibe coding, as well as in multi-modal general object recognition, OCR, and object localization.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":3.25e-7,"output_cost_per_token":1.95e-6,"cache_creation_input_token_cost":4.0625000000000003e-7},{"range":[256000,1000000],"input_cost_per_token":1.3e-6,"output_cost_per_token":3.9e-6,"cache_creation_input_token_cost":1.6250000000000001e-6}],"input_cost_per_token":3.25e-7,"output_cost_per_token":1.95e-6,"cache_creation_input_token_cost":4.0625000000000003e-7},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"cache_creation_input_token_cost":6.25e-7},{"range":[256000,1000000],"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_creation_input_token_cost":2.5e-6}],"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"cache_creation_input_token_cost":6.25e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3.6-plus-2026-04-02","object":"model","created":1775133557,"owned_by":"qwen","model_name":"qwen3.6-plus-2026-04-02","context_length":1000000,"description":"The Qwen3.6 native vision-language Plus series models demonstrate exceptional performance on par with the current state-of-the-art models, with a significant improvement in overall results compared to the 3.5 series. The models have been markedly enhanced in code-related capabilities such as agentic coding, front-end programming, and Vibe coding, as well as in multi-modal general object recognition, OCR, and object localization.This version is a snapshot as of April 2, 2026.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":3.25e-7,"output_cost_per_token":1.95e-6},{"range":[256000,1000000],"input_cost_per_token":1.3e-6,"output_cost_per_token":3.9e-6}],"input_cost_per_token":3.25e-7,"output_cost_per_token":1.95e-6},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":5e-7,"output_cost_per_token":3e-6},{"range":[256000,1000000],"input_cost_per_token":2e-6,"output_cost_per_token":6e-6}],"input_cost_per_token":5e-7,"output_cost_per_token":3e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"zai/glm-5v-turbo","object":"model","created":1775061458,"owned_by":"zai","model_name":"glm-5v-turbo","context_length":202752,"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":4e-6,"cache_read_input_token_cost":2.4e-7},"list_pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":4e-6,"cache_read_input_token_cost":2.4e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"xai/grok-4.20","object":"model","created":1774979019,"owned_by":"xai","model_name":"grok-4.20","context_length":2000000,"description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":5e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":5e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"google/lyria-3-clip-preview","object":"model","created":1774907255,"owned_by":"google","model_name":"lyria-3-clip-preview","context_length":1048576,"description":"30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text","audio"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.0,"output_cost_per_image":0.04,"output_cost_per_token":0.0},"list_pricing":{"input_cost_per_token":0.0,"output_cost_per_image":0.04,"output_cost_per_token":0.0},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"xai/grok-4.20-beta-0309-non-reasoning","object":"model","created":1773897105,"owned_by":"xai","model_name":"grok-4.20-beta-0309-non-reasoning","context_length":2000000,"description":"","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["image","text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4.20-beta-0309-reasoning","object":"model","created":1773897105,"owned_by":"xai","model_name":"grok-4.20-beta-0309-reasoning","context_length":2000000,"description":"","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/MiniMaxAI/MiniMax-M2.7","object":"model","created":1773836697,"owned_by":"deepinfra","model_name":"MiniMaxAI/MiniMax-M2.7","context_length":196608,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1e-6,"cache_read_input_token_cost":5.0000000000000004e-8},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":1e-6,"cache_read_input_token_cost":5.0000000000000004e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"minimax/MiniMax-M2.7","object":"model","created":1773836697,"owned_by":"minimax","model_name":"MiniMax-M2.7","context_length":204800,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":6e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":6e-8},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"openai/gpt-5.4-nano","object":"model","created":1773748187,"owned_by":"openai","model_name":"gpt-5.4-nano","context_length":400000,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.25e-6,"cache_read_input_token_cost":2e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.25e-6,"cache_read_input_token_cost":2e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.4-nano-2026-03-17","object":"model","created":1773748187,"owned_by":"openai","model_name":"gpt-5.4-nano-2026-03-17","context_length":400000,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.25e-6,"cache_read_input_token_cost":2e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.25e-6,"cache_read_input_token_cost":2e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.4-mini","object":"model","created":1773748178,"owned_by":"openai","model_name":"gpt-5.4-mini","context_length":400000,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.4-mini-2026-03-17","object":"model","created":1773748178,"owned_by":"openai","model_name":"gpt-5.4-mini-2026-03-17","context_length":400000,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"mistral/mistral-small-2603","object":"model","created":1773695685,"owned_by":"mistral","model_name":"mistral-small-2603","context_length":262144,"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":1.5e-8},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":1.5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"zai/glm-5-turbo","object":"model","created":1773583573,"owned_by":"zai","model_name":"glm-5-turbo","context_length":202752,"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":4e-6,"cache_read_input_token_cost":2.4e-7},"list_pricing":{"input_cost_per_token":1.2e-6,"output_cost_per_token":4e-6,"cache_read_input_token_cost":2.4e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"scaleway/qwen3.5-397b-a17b","object":"model","created":1773394870,"owned_by":"scaleway","model_name":"qwen3.5-397b-a17b","context_length":250000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.008600721185013e-7,"output_cost_per_token":4.205160432711008e-6},"list_pricing":{"input_cost_per_token":7.008600721185013e-7,"output_cost_per_token":4.205160432711008e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"nebius/nvidia/nemotron-3-super-120b-a12b","object":"model","created":1773245239,"owned_by":"nebius","model_name":"nvidia/nemotron-3-super-120b-a12b","context_length":8000,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":9e-7,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":9e-7,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"bytedance/seed-2-0-lite-260428","object":"model","created":1773157231,"owned_by":"bytedance","model_name":"seed-2-0-lite-260428","context_length":262144,"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","source":null,"capabilities":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"input_cost_per_token_above_128k_tokens":5e-7,"output_cost_per_token_above_128k_tokens":4e-6},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"input_cost_per_token_above_128k_tokens":5e-7,"output_cost_per_token_above_128k_tokens":4e-6},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.5-9B","object":"model","created":1773152396,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.5-9B","context_length":262144,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":1.5e-7},"list_pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":1.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"together_ai/Qwen/Qwen3.5-9B","object":"model","created":1773152396,"owned_by":"together_ai","model_name":"Qwen/Qwen3.5-9B","context_length":262144,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.7000000000000001e-7,"output_cost_per_token":2.5e-7},"list_pricing":{"input_cost_per_token":1.7000000000000001e-7,"output_cost_per_token":2.5e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"openai/gpt-5.4-pro","object":"model","created":1772734366,"owned_by":"openai","model_name":"gpt-5.4-pro","context_length":1050000,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"list_pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.4-pro-2026-03-05","object":"model","created":1772734366,"owned_by":"openai","model_name":"gpt-5.4-pro-2026-03-05","context_length":1050000,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"list_pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.4","object":"model","created":1772734352,"owned_by":"openai","model_name":"gpt-5.4","context_length":1050000,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":2.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":5e-6,"output_cost_per_token_above_272k_tokens":0.0000225,"cache_read_input_token_cost_above_272k_tokens":5e-7},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":2.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":5e-6,"output_cost_per_token_above_272k_tokens":0.0000225,"cache_read_input_token_cost_above_272k_tokens":5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.4-2026-03-05","object":"model","created":1772734352,"owned_by":"openai","model_name":"gpt-5.4-2026-03-05","context_length":1050000,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":2.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":5e-6,"output_cost_per_token_above_272k_tokens":0.0000225,"cache_read_input_token_cost_above_272k_tokens":5e-7},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":2.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":5e-6,"output_cost_per_token_above_272k_tokens":0.0000225,"cache_read_input_token_cost_above_272k_tokens":5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/mistral.devstral-2-123b","object":"model","created":1772697599,"owned_by":"amazon","model_name":"mistral.devstral-2-123b","context_length":256000,"description":"","source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.devstral-2-123b@us","object":"model","created":1772697599,"owned_by":"amazon","model_name":"mistral.devstral-2-123b","context_length":256000,"description":"","source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/zai.glm-4.7-flash","object":"model","created":1772697599,"owned_by":"amazon","model_name":"zai.glm-4.7-flash","context_length":200000,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":4e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/zai.glm-4.7-flash@us","object":"model","created":1772697599,"owned_by":"amazon","model_name":"zai.glm-4.7-flash","context_length":200000,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":4e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/ai21.jamba-1-5-large-v1:0","object":"model","created":1772632546,"owned_by":"amazon","model_name":"ai21.jamba-1-5-large-v1:0","context_length":256000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/ai21.jamba-1-5-mini-v1:0","object":"model","created":1772632546,"owned_by":"amazon","model_name":"ai21.jamba-1-5-mini-v1:0","context_length":256000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":4e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-3-haiku-20240307-v1:0","object":"model","created":1772632546,"owned_by":"amazon","model_name":"anthropic.claude-3-haiku-20240307-v1:0","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.75e-7,"output_cost_per_token":1.3750000000000002e-6,"cache_read_input_token_cost":2.75e-8,"cache_creation_input_token_cost":3.4375000000000004e-7},"list_pricing":{"input_cost_per_token":2.75e-7,"output_cost_per_token":1.3750000000000002e-6,"cache_read_input_token_cost":2.75e-8,"cache_creation_input_token_cost":3.4375000000000004e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/cohere.command-r-plus-v1:0","object":"model","created":1772632546,"owned_by":"amazon","model_name":"cohere.command-r-plus-v1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/cohere.command-r-v1:0","object":"model","created":1772632546,"owned_by":"amazon","model_name":"cohere.command-r-v1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/meta.llama3-70b-instruct-v1:0","object":"model","created":1772632546,"owned_by":"amazon","model_name":"meta.llama3-70b-instruct-v1:0","context_length":8192,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.18e-6,"output_cost_per_token":4.2e-6},"list_pricing":{"input_cost_per_token":3.18e-6,"output_cost_per_token":4.2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/meta.llama3-8b-instruct-v1:0","object":"model","created":1772632546,"owned_by":"amazon","model_name":"meta.llama3-8b-instruct-v1:0","context_length":8192,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":7.2e-7},"list_pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":7.2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.mistral-7b-instruct-v0:2","object":"model","created":1772632546,"owned_by":"amazon","model_name":"mistral.mistral-7b-instruct-v0:2","context_length":32000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2.6e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2.6e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.mistral-7b-instruct-v0:2@us","object":"model","created":1772632546,"owned_by":"amazon","model_name":"mistral.mistral-7b-instruct-v0:2","context_length":32000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2.6e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2.6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/mistral.mistral-large-2402-v1:0","object":"model","created":1772632546,"owned_by":"amazon","model_name":"mistral.mistral-large-2402-v1:0","context_length":32000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.0000104,"output_cost_per_token":0.0000312},"list_pricing":{"input_cost_per_token":0.0000104,"output_cost_per_token":0.0000312},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.mistral-large-2402-v1:0@us","object":"model","created":1772632546,"owned_by":"amazon","model_name":"mistral.mistral-large-2402-v1:0","context_length":32000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.0000104,"output_cost_per_token":0.0000312},"list_pricing":{"input_cost_per_token":0.0000104,"output_cost_per_token":0.0000312},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/mistral.mistral-small-2402-v1:0","object":"model","created":1772632546,"owned_by":"amazon","model_name":"mistral.mistral-small-2402-v1:0","context_length":32000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.mixtral-8x7b-instruct-v0:1","object":"model","created":1772632546,"owned_by":"amazon","model_name":"mistral.mixtral-8x7b-instruct-v0:1","context_length":32000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.9e-7,"output_cost_per_token":9.1e-7},"list_pricing":{"input_cost_per_token":5.9e-7,"output_cost_per_token":9.1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.mixtral-8x7b-instruct-v0:1@us","object":"model","created":1772632546,"owned_by":"amazon","model_name":"mistral.mixtral-8x7b-instruct-v0:1","context_length":32000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.9e-7,"output_cost_per_token":9.1e-7},"list_pricing":{"input_cost_per_token":5.9e-7,"output_cost_per_token":9.1e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"bytedance/seed-2-0-mini-260428","object":"model","created":1772131107,"owned_by":"bytedance","model_name":"seed-2-0-mini-260428","context_length":262144,"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","source":null,"capabilities":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"input_cost_per_token_above_128k_tokens":2e-7,"output_cost_per_token_above_128k_tokens":8e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"input_cost_per_token_above_128k_tokens":2e-7,"output_cost_per_token_above_128k_tokens":8e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"google/gemini-3.1-flash-image","object":"model","created":1772119558,"owned_by":"google","model_name":"gemini-3.1-flash-image","context_length":65536,"description":null,"source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["image","text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"output_cost_per_image_token":0.00006,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014}},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"output_cost_per_image_token":0.00006,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014}},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"google/gemini-3.1-flash-image-preview","object":"model","created":1772119558,"owned_by":"google","model_name":"gemini-3.1-flash-image-preview","context_length":65536,"description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models","capabilities":{"input_modalities":["image","text"],"output_modalities":["image","text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"output_cost_per_image_token":0.00006,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014}},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"output_cost_per_image_token":0.00006,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014}},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.5-35B-A3B","object":"model","created":1772053822,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.5-35B-A3B","context_length":262144,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4e-7,"output_cost_per_token":1e-6,"cache_read_input_token_cost":5.000000040000001e-8},"list_pricing":{"input_cost_per_token":1.4e-7,"output_cost_per_token":1e-6,"cache_read_input_token_cost":5.000000040000001e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.5-35b-a3b","object":"model","created":1772053822,"owned_by":"qwen","model_name":"qwen3.5-35b-a3b","context_length":262144,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall performance is comparable to that of the Qwen3.5-27B.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.625e-7,"output_cost_per_token":1.3e-6},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.5-27B","object":"model","created":1772053810,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.5-27B","context_length":262144,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.6e-7,"output_cost_per_token":2.5999999999999997e-6},"list_pricing":{"input_cost_per_token":2.6e-7,"output_cost_per_token":2.5999999999999997e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.5-27b","object":"model","created":1772053810,"owned_by":"qwen","model_name":"qwen3.5-27b","context_length":262144,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of the Qwen3.5-122B-A10B.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.9499999999999999e-7,"output_cost_per_token":1.5599999999999999e-6},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.5-122B-A10B","object":"model","created":1772053789,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.5-122B-A10B","context_length":262144,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.9e-7,"output_cost_per_token":2.4e-6},"list_pricing":{"input_cost_per_token":2.9e-7,"output_cost_per_token":2.4e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.5-122b-a10b","object":"model","created":1772053789,"owned_by":"qwen","model_name":"qwen3.5-122b-a10b","context_length":262144,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of overall performance, this model is second only to Qwen3.5-397B-A17B. Its text capabilities significantly outperform those of Qwen3-235B-2507, and its visual capabilities surpass those of Qwen3-VL-235B.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":2.0800000000000004e-6},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":3.2000000000000003e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"google/gemini-3.1-pro-preview-customtools","object":"model","created":1772045923,"owned_by":"google","model_name":"gemini-3.1-pro-preview-customtools","context_length":1048576,"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models","capabilities":{"input_modalities":["text","audio","image","video","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000018,"input_cost_per_audio_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7,"cache_read_input_audio_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000018,"input_cost_per_audio_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7,"cache_read_input_audio_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-5.3-codex","object":"model","created":1771959164,"owned_by":"openai","model_name":"gpt-5.3-codex","context_length":400000,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"mistral/devstral-latest","object":"model","created":1771588786,"owned_by":"mistral","model_name":"devstral-latest","context_length":262144,"description":null,"source":"https://mistral.ai/news/devstral-2-vibe-cli","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/devstral-medium-latest","object":"model","created":1771588786,"owned_by":"mistral","model_name":"devstral-medium-latest","context_length":262144,"description":null,"source":"https://mistral.ai/news/devstral-2-vibe-cli","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/devstral-small-latest","object":"model","created":1771588786,"owned_by":"mistral","model_name":"devstral-small-latest","context_length":256000,"description":"","source":"https://docs.mistral.ai/models/devstral-small-2-25-12","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"google/gemini-3.1-pro-preview","object":"model","created":1771509627,"owned_by":"google","model_name":"gemini-3.1-pro-preview","context_length":1048576,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models","capabilities":{"input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000018,"input_cost_per_audio_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7,"cache_read_input_audio_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000018,"input_cost_per_audio_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7,"cache_read_input_audio_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3.1-pro-preview","object":"model","created":1771509627,"owned_by":"vertex","model_name":"gemini-3.1-pro-preview","context_length":1048576,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","source":null,"capabilities":{"input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_image":0.00012,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000018,"input_cost_per_audio_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7,"cache_creation_input_token_cost_above_200k_tokens":2.5e-7,"cache_read_input_audio_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_image":0.00012,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000018,"input_cost_per_audio_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7,"cache_creation_input_token_cost_above_200k_tokens":2.5e-7,"cache_read_input_audio_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/anthropic.claude-sonnet-4-6@eu","object":"model","created":1771342990,"owned_by":"amazon","model_name":"anthropic.claude-sonnet-4-6","context_length":1000000,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":3.3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":4.125e-6,"cache_creation_input_token_cost_above_1hr":6.6e-6},"list_pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":3.3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":4.125e-6,"cache_creation_input_token_cost_above_1hr":6.6e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/anthropic.claude-sonnet-4-6","object":"model","created":1771342990,"owned_by":"amazon","model_name":"anthropic.claude-sonnet-4-6","context_length":1000000,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"cache_creation_input_token_cost_above_1hr":6e-6},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"cache_creation_input_token_cost_above_1hr":6e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/anthropic.claude-sonnet-4-6@us","object":"model","created":1771342990,"owned_by":"amazon","model_name":"anthropic.claude-sonnet-4-6","context_length":1000000,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":3.3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":4.125e-6,"cache_creation_input_token_cost_above_1hr":6.6e-6},"list_pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":3.3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":4.125e-6,"cache_creation_input_token_cost_above_1hr":6.6e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"anthropic/claude-sonnet-4-6","object":"model","created":1771342990,"owned_by":"anthropic","model_name":"claude-sonnet-4-6","context_length":1000000,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"cache_creation_input_token_cost_above_1hr":6e-6},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"cache_creation_input_token_cost_above_1hr":6e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-sonnet-4-6","object":"model","created":1771342990,"owned_by":"vertex","model_name":"claude-sonnet-4-6","context_length":1000000,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"cache_creation_input_token_cost_above_1hr":6e-6},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"cache_creation_input_token_cost_above_1hr":6e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3.5-397B-A17B","object":"model","created":1771223018,"owned_by":"deepinfra","model_name":"Qwen/Qwen3.5-397B-A17B","context_length":262144,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.5000000000000003e-7,"output_cost_per_token":2.9999999999999997e-6,"cache_read_input_token_cost":2.200000005e-7},"list_pricing":{"input_cost_per_token":4.5000000000000003e-7,"output_cost_per_token":2.9999999999999997e-6,"cache_read_input_token_cost":2.200000005e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3.5-397b-a17b","object":"model","created":1771223018,"owned_by":"qwen","model_name":"qwen3.5-397b-a17b","context_length":262144,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers state-of-the-art performance comparable to leading-edge models across a wide range of tasks, including language understanding, logical reasoning, code generation, agent-based tasks, image understanding, video understanding, and graphical user interface (GUI) interactions. With its robust code-generation and agent capabilities, the model exhibits strong generalization across diverse agent.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.8999999999999997e-7,"output_cost_per_token":2.34e-6},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":3.6000000000000003e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/deepseek.v3.2","object":"model","created":1770991619,"owned_by":"amazon","model_name":"deepseek.v3.2","context_length":163840,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.4e-7,"output_cost_per_token":2.22e-6},"list_pricing":{"input_cost_per_token":7.4e-7,"output_cost_per_token":2.22e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/minimax.minimax-m2.1","object":"model","created":1770991619,"owned_by":"amazon","model_name":"minimax.minimax-m2.1","context_length":196000,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":1.44e-6},"list_pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":1.44e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/minimax.minimax-m2.1@us","object":"model","created":1770991619,"owned_by":"amazon","model_name":"minimax.minimax-m2.1","context_length":196000,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":1.44e-6},"list_pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":1.44e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/nvidia.nemotron-nano-3-30b","object":"model","created":1770991619,"owned_by":"amazon","model_name":"nvidia.nemotron-nano-3-30b","context_length":262144,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/nvidia.nemotron-nano-3-30b@us","object":"model","created":1770991619,"owned_by":"amazon","model_name":"nvidia.nemotron-nano-3-30b","context_length":262144,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/qwen.qwen3-coder-next","object":"model","created":1770991619,"owned_by":"amazon","model_name":"qwen.qwen3-coder-next","context_length":262144,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":1.44e-6},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":1.44e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/zai.glm-4.7","object":"model","created":1770991619,"owned_by":"amazon","model_name":"zai.glm-4.7","context_length":200000,"description":null,"source":"https://aws.amazon.com/bedrock/pricing/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.2e-6},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-5-search-api","object":"model","created":1770991619,"owned_by":"openai","model_name":"gpt-5-search-api","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5-search-api-2025-10-14","object":"model","created":1770991619,"owned_by":"openai","model_name":"gpt-5-search-api-2025-10-14","context_length":272000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/MiniMaxAI/MiniMax-M2.5","object":"model","created":1770908502,"owned_by":"deepinfra","model_name":"MiniMaxAI/MiniMax-M2.5","context_length":196608,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.15e-6,"cache_read_input_token_cost":3e-8},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.15e-6,"cache_read_input_token_cost":3e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"minimax/MiniMax-M2.5","object":"model","created":1770908502,"owned_by":"minimax","model_name":"MiniMax-M2.5","context_length":204800,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":3e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":3e-8},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"nebius/MiniMaxAI/MiniMax-M2.5","object":"model","created":1770908502,"owned_by":"nebius","model_name":"MiniMaxAI/MiniMax-M2.5","context_length":196608,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":3e-7},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/zai-org/GLM-5","object":"model","created":1770829182,"owned_by":"deepinfra","model_name":"zai-org/GLM-5","context_length":202752,"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.08e-6,"cache_read_input_token_cost":1.2e-7},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.08e-6,"cache_read_input_token_cost":1.2e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"zai/glm-5","object":"model","created":1770829182,"owned_by":"zai","model_name":"glm-5","context_length":202752,"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3.2e-6,"cache_read_input_token_cost":2e-7,"cache_creation_input_token_cost":0.0},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3.2e-6,"cache_read_input_token_cost":2e-7,"cache_creation_input_token_cost":0.0},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"anthropic/claude-opus-4-6","object":"model","created":1770219050,"owned_by":"anthropic","model_name":"claude-opus-4-6","context_length":1000000,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-opus-4-6","object":"model","created":1770219050,"owned_by":"vertex","model_name":"claude-opus-4-6","context_length":1000000,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"qwen/qwen3-coder-next","object":"model","created":1770164101,"owned_by":"qwen","model_name":"qwen3-coder-next","context_length":262144,"description":"The new-generation code generation model in the Qwen3 series delivers performance close to that of Qwen3-Coder-Plus while offering even better capabilities. The model has been optimized with a focus on repository-level understanding, supports multi-turn tool interactions, and enhances its compatibility with agentic coding tools.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.9499999999999999e-7,"output_cost_per_token":9.75e-7},{"range":[32000,128000],"input_cost_per_token":3.25e-7,"output_cost_per_token":1.6250000000000001e-6},{"range":[128000,256000],"input_cost_per_token":5.200000000000001e-7,"output_cost_per_token":2.6e-6}],"input_cost_per_token":1.9499999999999999e-7,"output_cost_per_token":9.75e-7},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":3e-7,"output_cost_per_token":1.5e-6},{"range":[32000,128000],"input_cost_per_token":5e-7,"output_cost_per_token":2.5e-6},{"range":[128000,256000],"input_cost_per_token":8.000000000000001e-7,"output_cost_per_token":4e-6}],"input_cost_per_token":3e-7,"output_cost_per_token":1.5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/stepfun-ai/Step-3.5-Flash","object":"model","created":1769728337,"owned_by":"deepinfra","model_name":"stepfun-ai/Step-3.5-Flash","context_length":262144,"description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":3e-7,"cache_read_input_token_cost":1.9999999799999997e-8},"list_pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":3e-7,"cache_read_input_token_cost":1.9999999799999997e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"databricks/databricks-claude-haiku-4-5","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-claude-haiku-4-5","context_length":200000,"description":"Claude Haiku 4.5 is Anthropic's fastest and most cost efficient hybrid reasoning model, optimized for real-time use and high-volume workloads. It features two modes: quick responses for time-sensitive interactions and extended reasoning for complex problem-solving. Haiku 4.5 excels in environments like coding assistance, automated agent workflows, and enterprise-scale analysis. This endpoint is hosted by Databricks.","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.00002e-6,"output_cost_per_token":5.000030000000001e-6,"input_dbu_cost_per_token":0.000014286,"output_dbu_cost_per_token":0.000071429},"list_pricing":{"input_cost_per_token":1.00002e-6,"output_cost_per_token":5.000030000000001e-6,"input_dbu_cost_per_token":0.000014286,"output_dbu_cost_per_token":0.000071429},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-claude-opus-4-1","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-claude-opus-4-1","context_length":200000,"description":"Claude Opus 4.1 is a state-of-the-art, hybrid reasoning model built and trained by Anthropic. This general purpose large language model is designed for both complex reasoning and real-world applications at enterprise scale. It supports text and image input, with a 200K token context window and 32K output token capabilities. This model excels at tasks like code generation, research and content creation, and multi-step agents workflows without constant human intervention. This endpoint is hosted by Databricks.","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000015000020000000002,"output_cost_per_token":0.0000749994,"input_dbu_cost_per_token":0.000214286,"output_dbu_cost_per_token":0.001071429},"list_pricing":{"input_cost_per_token":0.000015000020000000002,"output_cost_per_token":0.0000749994,"input_dbu_cost_per_token":0.000214286,"output_dbu_cost_per_token":0.001071429},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-claude-opus-4-5","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-claude-opus-4-5","context_length":200000,"description":"Claude Opus 4.5 is a large language model built and trained by Anthropic for production software engineering and sophisticated multi-tool agents. It supports text and image input with a 200K token context window. The model is designed for professional tasks requiring complex reasoning across multiple systems, including code generation, document creation, spreadsheets, presentations, and multi-step agent workflows. It features advanced tool use capabilities including tool search and programmatic tool calling for agents working with large tool libraries. This endpoint is hosted by Databricks.","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]},"supports_output_config":true},"pricing":{"input_cost_per_token":5.000030000000001e-6,"output_cost_per_token":0.00002500001,"input_dbu_cost_per_token":0.000071429,"output_dbu_cost_per_token":0.000357143},"list_pricing":{"input_cost_per_token":5.000030000000001e-6,"output_cost_per_token":0.00002500001,"input_dbu_cost_per_token":0.000071429,"output_dbu_cost_per_token":0.000357143},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-claude-sonnet-4-5","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-claude-sonnet-4-5","context_length":200000,"description":"Claude Sonnet 4.5 is Anthropic's most advanced hybrid reasoning model. It offers two modes: near-instant responses and extended thinking for deeper reasoning based on the complexity of the task. Claude Sonnet 4.5 specializes in application that require a balance of practical throughput and advanced thinking such as customer-facing agents, production coding workflows, and content generation at scale. This endpoint is hosted by Databricks","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.9999800000000004e-6,"output_cost_per_token":0.000015000020000000002,"input_dbu_cost_per_token":0.000042857,"output_dbu_cost_per_token":0.000214286},"list_pricing":{"input_cost_per_token":5.9999800000000004e-6,"output_cost_per_token":0.000015000020000000002,"input_dbu_cost_per_token":0.000042857,"output_dbu_cost_per_token":0.000214286},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gemini-2-5-flash","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-gemini-2-5-flash","context_length":1048576,"description":"Gemini 2.5 Flash is Google's best model in terms of price and performance, and offers well-rounded capabilities. Gemini 2.5 Flash is Google's first Flash model that features thinking capabilities, which lets you see the thinking process that the model goes through when generating its response. This endpoint is hosted by Databricks.","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","capabilities":{"input_modalities":["text","image","video","audio","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.0002e-7,"output_cost_per_token":2.49998e-6,"input_dbu_cost_per_token":4.285999999999999e-6,"output_dbu_cost_per_token":0.000035714},"list_pricing":{"input_cost_per_token":3.0002e-7,"output_cost_per_token":2.49998e-6,"input_dbu_cost_per_token":4.285999999999999e-6,"output_dbu_cost_per_token":0.000035714},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gemini-2-5-pro","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-gemini-2-5-pro","context_length":1048576,"description":"Gemini 2.5 Pro is Google's most advanced reasoning Gemini model, capable of solving complex problems. This endpoint is hosted by Databricks.","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","capabilities":{"input_modalities":["text","image","video","audio","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.24999e-6,"output_cost_per_token":9.99999e-6,"input_dbu_cost_per_token":0.000017857,"output_dbu_cost_per_token":0.000142857},"list_pricing":{"input_cost_per_token":1.24999e-6,"output_cost_per_token":9.99999e-6,"input_dbu_cost_per_token":0.000017857,"output_dbu_cost_per_token":0.000142857},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gemma-3-12b","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-gemma-3-12b","context_length":128000,"description":"Gemma 3 12B is a state-of-the-art multimodal language model built and trained by Google. The model supports a context length of 128K tokens and can analyze images and text. With support for over 140 languages and optimized for dialogue use cases, Gemma 3 12B is aligned with human preferences for helpfulness and safety. This endpoint is hosted by Databricks.","source":"https://www.databricks.com/product/pricing/foundation-model-serving","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5001e-7,"output_cost_per_token":5.0001e-7,"input_dbu_cost_per_token":2.1429999999999996e-6,"output_dbu_cost_per_token":7.143e-6},"list_pricing":{"input_cost_per_token":1.5001e-7,"output_cost_per_token":5.0001e-7,"input_dbu_cost_per_token":2.1429999999999996e-6,"output_dbu_cost_per_token":7.143e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-gpt-5","context_length":272000,"description":"GPT-5 is a state-of-the-art, general purpose large language model and reasoning model built and trained by OpenAI. It supports multimodal inputs and features a 128K token context window. The model is built for coding, chat, reasoning and agent-driven tasks. This endpoint is hosted by Databricks.","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.24999e-6,"output_cost_per_token":0.00001999998,"input_dbu_cost_per_token":0.000017857,"output_dbu_cost_per_token":0.000142857},"list_pricing":{"input_cost_per_token":1.24999e-6,"output_cost_per_token":0.00001999998,"input_dbu_cost_per_token":0.000017857,"output_dbu_cost_per_token":0.000142857},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-1","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-gpt-5-1","context_length":272000,"description":"GPT-5.1 is a general purpose large language model with reasoning capabilities developed by OpenAI. This model features both Instant and Thinking modes for fast conversation or deep reasoning, automatically adjusting for simple or complex tasks. The model excels at content creation, tutoring, technical support, and coding, with less reliance on strict prompt engineering than prior versions. It supports multimodal inputs and features a 128K token context window. This endpoint is hosted by Databricks.","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.49998e-6,"output_cost_per_token":9.99999e-6,"input_dbu_cost_per_token":0.000017857,"output_dbu_cost_per_token":0.000142857},"list_pricing":{"input_cost_per_token":2.49998e-6,"output_cost_per_token":9.99999e-6,"input_dbu_cost_per_token":0.000017857,"output_dbu_cost_per_token":0.000142857},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-mini","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-gpt-5-mini","context_length":272000,"description":"GPT-5 mini is a state-of-the-art, general purpose large language model and reasoning model built and trained by OpenAI. It supports multimodal inputs and features a 128K token context window. The model is cost-optimized for reasoning and chat workloads and excels at well-defined tasks that require reliable reasoning, precise language, and rapid output for text and images. This endpoint is hosted by Databricks.","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.4989e-7,"output_cost_per_token":1.9999700000000004e-6,"input_dbu_cost_per_token":3.571e-6,"output_dbu_cost_per_token":0.000028571},"list_pricing":{"input_cost_per_token":4.4989e-7,"output_cost_per_token":1.9999700000000004e-6,"input_dbu_cost_per_token":3.571e-6,"output_dbu_cost_per_token":0.000028571},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-5-nano","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-gpt-5-nano","context_length":272000,"description":"GPT-5 nano is a state-of-the-art, general purpose large language model and reasoning model built and trained by OpenAI. It supports multimodal inputs and features a 128K token context window. The model excels at high-throughput tasks like simple instruction-following or classification for routine business processes or mobile applications. This endpoint is hosted by Databricks.","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.998e-8,"output_cost_per_token":3.9998000000000007e-7,"input_dbu_cost_per_token":7.14e-7,"output_dbu_cost_per_token":5.714000000000001e-6},"list_pricing":{"input_cost_per_token":4.998e-8,"output_cost_per_token":3.9998000000000007e-7,"input_dbu_cost_per_token":7.14e-7,"output_dbu_cost_per_token":5.714000000000001e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-oss-120b","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-gpt-oss-120b","context_length":131072,"description":"GPT OSS 120B is a state-of-the-art, reasoning model with chain-of-thought and adjustable reasoning effort levels built and trained by OpenAI. It is OpenAI's flagship open-weight model that features a 128K token context window. The model is built for high-quality reasoning tasks.","source":"https://www.databricks.com/product/pricing/foundation-model-serving","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5001e-7,"output_cost_per_token":5.9997e-7,"input_dbu_cost_per_token":2.1429999999999996e-6,"output_dbu_cost_per_token":8.571e-6},"list_pricing":{"input_cost_per_token":1.5001e-7,"output_cost_per_token":5.9997e-7,"input_dbu_cost_per_token":2.1429999999999996e-6,"output_dbu_cost_per_token":8.571e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-gpt-oss-20b","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-gpt-oss-20b","context_length":131072,"description":"GPT OSS 20B is a state-of-the-art, lightweight reasoning model built and trained by OpenAI. This model also has a 128K token context window and excels at real-time copilots and batch inference tasks.","source":"https://www.databricks.com/product/pricing/foundation-model-serving","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3.0002e-7,"input_dbu_cost_per_token":1e-6,"output_dbu_cost_per_token":4.285999999999999e-6},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3.0002e-7,"input_dbu_cost_per_token":1e-6,"output_dbu_cost_per_token":4.285999999999999e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-llama-4-maverick","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-llama-4-maverick","context_length":128000,"description":"Llama 4 Maverick is a state-of-the-art mixture of experts (MoE) language model trained and released by Meta. The model has 17B active parameters, 128 experts, and 400 billion total parameters. The model supports a context length of 128K tokens. The model is optimized for multilingual dialogue use cases, supporting 12 languages, and is aligned with human preferences for helpfulness and safety. It is not intended for use in languages other than English. Llama 4 is licensed under the Meta Llama 4 Community License, Copyright © Meta Platforms, Inc. All Rights Reserved. Customers are responsible for ensuring compliance with applicable model licenses.","source":"https://www.databricks.com/product/pricing/foundation-model-serving","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.0001e-7,"output_cost_per_token":1.50003e-6,"input_dbu_cost_per_token":7.143e-6,"output_dbu_cost_per_token":0.000021429},"list_pricing":{"input_cost_per_token":5.0001e-7,"output_cost_per_token":1.50003e-6,"input_dbu_cost_per_token":7.143e-6,"output_dbu_cost_per_token":0.000021429},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-meta-llama-3-1-8b-instruct","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-meta-llama-3-1-8b-instruct","context_length":200000,"description":"Llama 3.1 is a state-of-the-art 8B parameter dense language model trained and released by Meta. The model supports a context length of 128K tokens. The model is optimized for multilingual dialogue use cases and aligned with human preferences for helpfulness and safety. It is not intended for use in languages other than English. Meta Llama 3.1 is licensed under the Meta Llama 3.1 Community License, Copyright © Meta Platforms, Inc. All Rights Reserved. Customers are responsible for ensuring compliance with applicable model licenses.","source":"https://www.databricks.com/product/pricing/foundation-model-serving","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5001e-7,"output_cost_per_token":4.5003000000000007e-7,"input_dbu_cost_per_token":2.1429999999999996e-6,"output_dbu_cost_per_token":6.429000000000001e-6},"list_pricing":{"input_cost_per_token":1.5001e-7,"output_cost_per_token":4.5003000000000007e-7,"input_dbu_cost_per_token":2.1429999999999996e-6,"output_dbu_cost_per_token":6.429000000000001e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"databricks/databricks-meta-llama-3-3-70b-instruct","object":"model","created":1769507058,"owned_by":"databricks","model_name":"databricks-meta-llama-3-3-70b-instruct","context_length":128000,"description":"Llama 3.3 is a state-of-the-art 70B parameter dense language model trained and released by Meta. The model supports a context length of 128K tokens. The model is optimized for multilingual dialogue use cases and aligned with human preferences for helpfulness and safety. It is not intended for use in languages other than English. Meta Llama 3.3 is licensed under the Meta Llama 3.3 Community License, Copyright © Meta Platforms, Inc. All Rights Reserved. Customers are responsible for ensuring compliance with applicable model licenses.","source":"https://www.databricks.com/product/pricing/foundation-model-serving","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.0001e-7,"output_cost_per_token":1.50003e-6,"input_dbu_cost_per_token":7.143e-6,"output_dbu_cost_per_token":0.000021429},"list_pricing":{"input_cost_per_token":5.0001e-7,"output_cost_per_token":1.50003e-6,"input_dbu_cost_per_token":7.143e-6,"output_dbu_cost_per_token":0.000021429},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/gpt-oss-120b","object":"model","created":1769507058,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/gpt-oss-120b","context_length":131072,"description":null,"source":"https://fireworks.ai/pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":1.5e-8},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":1.5e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"ovhcloud/gpt-oss-120b","object":"model","created":1769507058,"owned_by":"ovhcloud","model_name":"gpt-oss-120b","context_length":131072,"description":null,"source":"https://endpoints.ai.cloud.ovh.net/models/gpt-oss-120b","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":4.7e-7},"list_pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":4.7e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ovhcloud/gpt-oss-20b","object":"model","created":1769507058,"owned_by":"ovhcloud","model_name":"gpt-oss-20b","context_length":131072,"description":null,"source":"https://endpoints.ai.cloud.ovh.net/models/gpt-oss-20b","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-8,"output_cost_per_token":1.8e-7},"list_pricing":{"input_cost_per_token":5e-8,"output_cost_per_token":1.8e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ovhcloud/Meta-Llama-3_3-70B-Instruct","object":"model","created":1769507058,"owned_by":"ovhcloud","model_name":"Meta-Llama-3_3-70B-Instruct","context_length":131072,"description":null,"source":"https://endpoints.ai.cloud.ovh.net/models/meta-llama-3-3-70b-instruct","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.4e-7,"output_cost_per_token":7.4e-7},"list_pricing":{"input_cost_per_token":7.4e-7,"output_cost_per_token":7.4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ovhcloud/Mistral-7B-Instruct-v0.3","object":"model","created":1769507058,"owned_by":"ovhcloud","model_name":"Mistral-7B-Instruct-v0.3","context_length":65536,"description":null,"source":"https://endpoints.ai.cloud.ovh.net/models/mistral-7b-instruct-v0-3","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.1e-7,"output_cost_per_token":1.1e-7},"list_pricing":{"input_cost_per_token":1.1e-7,"output_cost_per_token":1.1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ovhcloud/Mistral-Nemo-Instruct-2407","object":"model","created":1769507058,"owned_by":"ovhcloud","model_name":"Mistral-Nemo-Instruct-2407","context_length":65536,"description":null,"source":"https://endpoints.ai.cloud.ovh.net/models/mistral-nemo-instruct-2407","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.4e-7,"output_cost_per_token":1.4e-7},"list_pricing":{"input_cost_per_token":1.4e-7,"output_cost_per_token":1.4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ovhcloud/Mistral-Small-3.2-24B-Instruct-2506","object":"model","created":1769507058,"owned_by":"ovhcloud","model_name":"Mistral-Small-3.2-24B-Instruct-2506","context_length":131072,"description":null,"source":"https://endpoints.ai.cloud.ovh.net/models/mistral-small-3-2-24b-instruct-2506","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3.1e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3.1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ovhcloud/Qwen2.5-VL-72B-Instruct","object":"model","created":1769507058,"owned_by":"ovhcloud","model_name":"Qwen2.5-VL-72B-Instruct","context_length":32768,"description":null,"source":"https://endpoints.ai.cloud.ovh.net/models/qwen2-5-vl-72b-instruct","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.01e-6,"output_cost_per_token":1.01e-6},"list_pricing":{"input_cost_per_token":1.01e-6,"output_cost_per_token":1.01e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"ovhcloud/Qwen3-32B","object":"model","created":1769507058,"owned_by":"ovhcloud","model_name":"Qwen3-32B","context_length":32768,"description":null,"source":"https://endpoints.ai.cloud.ovh.net/models/qwen3-32b","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":2.5e-7},"list_pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":2.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/moonshotai.kimi-k2.5","object":"model","created":1769487076,"owned_by":"amazon","model_name":"moonshotai.kimi-k2.5","context_length":262144,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","source":"https://platform.moonshot.ai/docs/guide/kimi-k2-5-quickstart","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":3e-6},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":3e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/moonshotai/Kimi-K2.5","object":"model","created":1769487076,"owned_by":"deepinfra","model_name":"moonshotai/Kimi-K2.5","context_length":262144,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.5000000000000003e-7,"output_cost_per_token":2.25e-6,"cache_read_input_token_cost":7.000000200000001e-8},"list_pricing":{"input_cost_per_token":4.5000000000000003e-7,"output_cost_per_token":2.25e-6,"cache_read_input_token_cost":7.000000200000001e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"cloudflare/@cf/zai-org/glm-4.7-flash","object":"model","created":1768833913,"owned_by":"cloudflare","model_name":"@cf/zai-org/glm-4.7-flash","context_length":131072,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.049999999999999e-8,"output_cost_per_token":4.0000000000000003e-7},"list_pricing":{"input_cost_per_token":6.049999999999999e-8,"output_cost_per_token":4.0000000000000003e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/zai-org/GLM-4.7-Flash","object":"model","created":1768833913,"owned_by":"deepinfra","model_name":"zai-org/GLM-4.7-Flash","context_length":202752,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.000000000000001e-8,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":1.0000000199999999e-8},"list_pricing":{"input_cost_per_token":6.000000000000001e-8,"output_cost_per_token":4.0000000000000003e-7,"cache_read_input_token_cost":1.0000000199999999e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/google.gemma-3-12b-it","object":"model","created":1767692616,"owned_by":"amazon","model_name":"google.gemma-3-12b-it","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":2.9e-7},"list_pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":2.9e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/google.gemma-3-12b-it@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"google.gemma-3-12b-it","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":2.9e-7},"list_pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":2.9e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/google.gemma-3-27b-it","object":"model","created":1767692616,"owned_by":"amazon","model_name":"google.gemma-3-27b-it","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.3e-7,"output_cost_per_token":3.8e-7},"list_pricing":{"input_cost_per_token":2.3e-7,"output_cost_per_token":3.8e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/google.gemma-3-27b-it@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"google.gemma-3-27b-it","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.3e-7,"output_cost_per_token":3.8e-7},"list_pricing":{"input_cost_per_token":2.3e-7,"output_cost_per_token":3.8e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/google.gemma-3-4b-it","object":"model","created":1767692616,"owned_by":"amazon","model_name":"google.gemma-3-4b-it","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":8e-8},"list_pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":8e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/google.gemma-3-4b-it@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"google.gemma-3-4b-it","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":8e-8},"list_pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":8e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/minimax.minimax-m2","object":"model","created":1767692616,"owned_by":"amazon","model_name":"minimax.minimax-m2","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/minimax.minimax-m2@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"minimax.minimax-m2","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/mistral.magistral-small-2509","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.magistral-small-2509","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.magistral-small-2509@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.magistral-small-2509","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/mistral.ministral-3-14b-instruct","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.ministral-3-14b-instruct","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.ministral-3-14b-instruct@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.ministral-3-14b-instruct","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/mistral.ministral-3-3b-instruct","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.ministral-3-3b-instruct","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":1e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.ministral-3-3b-instruct@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.ministral-3-3b-instruct","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":1e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":1e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/mistral.ministral-3-8b-instruct","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.ministral-3-8b-instruct","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.ministral-3-8b-instruct@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.ministral-3-8b-instruct","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/mistral.mistral-large-3-675b-instruct","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.mistral-large-3-675b-instruct","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.voxtral-mini-3b-2507","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.voxtral-mini-3b-2507","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","audio"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":4e-8},"list_pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":4e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.voxtral-mini-3b-2507@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.voxtral-mini-3b-2507","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","audio"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":4e-8},"list_pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":4e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/mistral.voxtral-small-24b-2507","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.voxtral-small-24b-2507","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","audio"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/mistral.voxtral-small-24b-2507@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"mistral.voxtral-small-24b-2507","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","audio"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/moonshot.kimi-k2-thinking","object":"model","created":1767692616,"owned_by":"amazon","model_name":"moonshot.kimi-k2-thinking","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.5e-6},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/nvidia.nemotron-nano-12b-v2","object":"model","created":1767692616,"owned_by":"amazon","model_name":"nvidia.nemotron-nano-12b-v2","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/nvidia.nemotron-nano-12b-v2@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"nvidia.nemotron-nano-12b-v2","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/nvidia.nemotron-nano-9b-v2","object":"model","created":1767692616,"owned_by":"amazon","model_name":"nvidia.nemotron-nano-9b-v2","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.3e-7},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/nvidia.nemotron-nano-9b-v2@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"nvidia.nemotron-nano-9b-v2","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.3e-7},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/openai.gpt-oss-120b-1:0","object":"model","created":1767692616,"owned_by":"amazon","model_name":"openai.gpt-oss-120b-1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/openai.gpt-oss-120b-1:0@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"openai.gpt-oss-120b-1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/openai.gpt-oss-20b-1:0","object":"model","created":1767692616,"owned_by":"amazon","model_name":"openai.gpt-oss-20b-1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/openai.gpt-oss-20b-1:0@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"openai.gpt-oss-20b-1:0","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/openai.gpt-oss-safeguard-120b","object":"model","created":1767692616,"owned_by":"amazon","model_name":"openai.gpt-oss-safeguard-120b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/openai.gpt-oss-safeguard-120b@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"openai.gpt-oss-safeguard-120b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/openai.gpt-oss-safeguard-20b","object":"model","created":1767692616,"owned_by":"amazon","model_name":"openai.gpt-oss-safeguard-20b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":2e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/openai.gpt-oss-safeguard-20b@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"openai.gpt-oss-safeguard-20b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":2e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":2e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/qwen.qwen3-32b-v1:0","object":"model","created":1767692616,"owned_by":"amazon","model_name":"qwen.qwen3-32b-v1:0","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/qwen.qwen3-32b-v1:0@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"qwen.qwen3-32b-v1:0","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/qwen.qwen3-coder-30b-a3b-v1:0","object":"model","created":1767692616,"owned_by":"amazon","model_name":"qwen.qwen3-coder-30b-a3b-v1:0","context_length":262144,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/qwen.qwen3-coder-30b-a3b-v1:0@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"qwen.qwen3-coder-30b-a3b-v1:0","context_length":262144,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/qwen.qwen3-next-80b-a3b","object":"model","created":1767692616,"owned_by":"amazon","model_name":"qwen.qwen3-next-80b-a3b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/qwen.qwen3-next-80b-a3b@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"qwen.qwen3-next-80b-a3b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/qwen.qwen3-vl-235b-a22b","object":"model","created":1767692616,"owned_by":"amazon","model_name":"qwen.qwen3-vl-235b-a22b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.3e-7,"output_cost_per_token":2.66e-6},"list_pricing":{"input_cost_per_token":5.3e-7,"output_cost_per_token":2.66e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/qwen.qwen3-vl-235b-a22b@us","object":"model","created":1767692616,"owned_by":"amazon","model_name":"qwen.qwen3-vl-235b-a22b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.3e-7,"output_cost_per_token":2.66e-6},"list_pricing":{"input_cost_per_token":5.3e-7,"output_cost_per_token":2.66e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/anthropic/claude-3-7-sonnet-latest","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"anthropic/claude-3-7-sonnet-latest","context_length":200000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":3.3e-7},"list_pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165,"cache_read_input_token_cost":3.3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/anthropic/claude-4-opus","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"anthropic/claude-4-opus","context_length":200000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":0.0000165,"output_cost_per_token":0.0000825},"list_pricing":{"input_cost_per_token":0.0000165,"output_cost_per_token":0.0000825},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/anthropic/claude-4-sonnet","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"anthropic/claude-4-sonnet","context_length":200000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165},"list_pricing":{"input_cost_per_token":3.3e-6,"output_cost_per_token":0.0000165},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-R1-0528-Turbo","context_length":32768,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-V3","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-V3","context_length":163840,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.2e-7,"output_cost_per_token":8.9e-7},"list_pricing":{"input_cost_per_token":3.2e-7,"output_cost_per_token":8.9e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-V3-0324","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-V3-0324","context_length":163840,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.4000000000000003e-7,"output_cost_per_token":9.000000000000001e-7,"cache_read_input_token_cost":1.35e-7},"list_pricing":{"input_cost_per_token":2.4000000000000003e-7,"output_cost_per_token":9.000000000000001e-7,"cache_read_input_token_cost":1.35e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-V3.1","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-V3.1","context_length":163840,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":9.499999999999999e-7,"cache_read_input_token_cost":1.3e-7},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":9.499999999999999e-7,"cache_read_input_token_cost":1.3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/google/gemini-2.5-flash","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"google/gemini-2.5-flash","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/google/gemini-2.5-pro","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"google/gemini-2.5-pro","context_length":1000000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"meta-llama/Llama-3.3-70B-Instruct-Turbo","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":3.2e-7},"list_pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":3.2e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","context_length":1048576,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.000000000000001e-7},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.000000000000001e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"meta-llama/Llama-4-Scout-17B-16E-Instruct","context_length":327680,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":1.0000000000000001e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"meta-llama/Meta-Llama-3.1-70B-Instruct","context_length":131072,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":4e-7},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":4.0000000000000003e-7},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":4.0000000000000003e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"meta-llama/Meta-Llama-3.1-8B-Instruct","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":5.0000000000000004e-8},"list_pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":5.0000000000000004e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":4e-8},"list_pricing":{"input_cost_per_token":2e-8,"output_cost_per_token":4e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Meta-Llama-3-8B-Instruct","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"meta-llama/Meta-Llama-3-8B-Instruct","context_length":8192,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-8,"output_cost_per_token":6e-8},"list_pricing":{"input_cost_per_token":3e-8,"output_cost_per_token":6e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/mistralai/Mistral-Nemo-Instruct-2407","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"mistralai/Mistral-Nemo-Instruct-2407","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.9e-8,"output_cost_per_token":3.0000000000000004e-8},"list_pricing":{"input_cost_per_token":1.9e-8,"output_cost_per_token":3.0000000000000004e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-8,"output_cost_per_token":2.0000000000000002e-7},"list_pricing":{"input_cost_per_token":7.5e-8,"output_cost_per_token":2.0000000000000002e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"mistralai/Mixtral-8x7B-Instruct-v0.1","context_length":32768,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":4e-7},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/moonshotai/Kimi-K2-Instruct-0905","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"moonshotai/Kimi-K2-Instruct-0905","context_length":262144,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":4e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"nvidia/Llama-3.1-Nemotron-70B-Instruct","context_length":131072,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/nvidia/NVIDIA-Nemotron-Nano-9B-v2","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"nvidia/NVIDIA-Nemotron-Nano-9B-v2","context_length":131072,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":1.6e-7},"list_pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":1.6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen2.5-VL-32B-Instruct","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"Qwen/Qwen2.5-VL-32B-Instruct","context_length":128000,"description":"","source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-235B-A22B-Instruct-2507","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":5.5e-7},"list_pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":5.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-Coder-480B-A35B-Instruct","context_length":262144,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":1.6e-6},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":1.6e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1e-6,"cache_read_input_token_cost":9.999999899999999e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1e-6,"cache_read_input_token_cost":9.999999899999999e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Sao10K/L3.1-70B-Euryale-v2.2","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"Sao10K/L3.1-70B-Euryale-v2.2","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8.5e-7,"output_cost_per_token":8.5e-7},"list_pricing":{"input_cost_per_token":8.5e-7,"output_cost_per_token":8.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Sao10K/L3.3-70B-Euryale-v2.3","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"Sao10K/L3.3-70B-Euryale-v2.3","context_length":131072,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6.5e-7,"output_cost_per_token":7.5e-7},"list_pricing":{"input_cost_per_token":6.5e-7,"output_cost_per_token":7.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo","object":"model","created":1767692616,"owned_by":"deepinfra","model_name":"Sao10K/L3-8B-Lunaris-v1-Turbo","context_length":8192,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":5.0000000000000004e-8},"list_pricing":{"input_cost_per_token":4e-8,"output_cost_per_token":5.0000000000000004e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepseek/deepseek-chat","object":"model","created":1767692616,"owned_by":"deepseek","model_name":"deepseek-chat","context_length":131072,"description":null,"source":"https://api-docs.deepseek.com/quick_start/pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.8e-7,"output_cost_per_token":4.2e-7,"cache_read_input_token_cost":2.8e-8,"input_cost_per_token_cache_hit":2.8e-8,"cache_creation_input_token_cost":0.0},"list_pricing":{"input_cost_per_token":2.8e-7,"output_cost_per_token":4.2e-7,"cache_read_input_token_cost":2.8e-8,"input_cost_per_token_cache_hit":2.8e-8,"cache_creation_input_token_cost":0.0},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"deepseek/deepseek-reasoner","object":"model","created":1767692616,"owned_by":"deepseek","model_name":"deepseek-reasoner","context_length":131072,"description":null,"source":"https://api-docs.deepseek.com/quick_start/pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.8e-7,"output_cost_per_token":4.2e-7,"cache_read_input_token_cost":2.8e-8,"input_cost_per_token_cache_hit":2.8e-8},"list_pricing":{"input_cost_per_token":2.8e-7,"output_cost_per_token":4.2e-7,"cache_read_input_token_cost":2.8e-8,"input_cost_per_token_cache_hit":2.8e-8},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"microsoft/gpt-35-turbo","object":"model","created":1767692616,"owned_by":"microsoft","model_name":"gpt-35-turbo","context_length":4097,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"microsoft/o1-mini","object":"model","created":1767692616,"owned_by":"microsoft","model_name":"o1-mini","context_length":128000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":1.21e-6,"output_cost_per_token":4.84e-6,"cache_read_input_token_cost":6.05e-7},"list_pricing":{"input_cost_per_token":1.21e-6,"output_cost_per_token":4.84e-6,"cache_read_input_token_cost":6.05e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"microsoft/o3-mini","object":"model","created":1767692616,"owned_by":"microsoft","model_name":"o3-mini","context_length":200000,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":5.5e-7},"list_pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":5.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/labs-devstral-small-2512","object":"model","created":1767692616,"owned_by":"mistral","model_name":"labs-devstral-small-2512","context_length":256000,"description":"","source":"https://docs.mistral.ai/models/devstral-small-2-25-12","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/magistral-medium-latest","object":"model","created":1767692616,"owned_by":"mistral","model_name":"magistral-medium-latest","context_length":262144,"description":null,"source":"https://mistral.ai/news/magistral","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":5e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/magistral-small-latest","object":"model","created":1767692616,"owned_by":"mistral","model_name":"magistral-small-latest","context_length":262144,"description":null,"source":"https://mistral.ai/pricing#api-pricing","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-medium","object":"model","created":1767692616,"owned_by":"mistral","model_name":"mistral-medium","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.7e-6,"output_cost_per_token":8.1e-6},"list_pricing":{"input_cost_per_token":2.7e-6,"output_cost_per_token":8.1e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-medium-2505","object":"model","created":1767692616,"owned_by":"mistral","model_name":"mistral-medium-2505","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-medium-latest","object":"model","created":1767692616,"owned_by":"mistral","model_name":"mistral-medium-latest","context_length":262144,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-small-latest","object":"model","created":1767692616,"owned_by":"mistral","model_name":"mistral-small-latest","context_length":262144,"description":null,"source":"https://mistral.ai/pricing","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":1.8e-7},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":1.8e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/open-mistral-nemo","object":"model","created":1767692616,"owned_by":"mistral","model_name":"open-mistral-nemo","context_length":131072,"description":null,"source":"https://mistral.ai/technology/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/open-mistral-nemo-2407","object":"model","created":1767692616,"owned_by":"mistral","model_name":"open-mistral-nemo-2407","context_length":131072,"description":null,"source":"https://mistral.ai/technology/","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":3e-7},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-3.5-turbo-0125","object":"model","created":1767692616,"owned_by":"openai","model_name":"gpt-3.5-turbo-0125","context_length":16385,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-3.5-turbo-1106","object":"model","created":1767692616,"owned_by":"openai","model_name":"gpt-3.5-turbo-1106","context_length":16385,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":2e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":2e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"together_ai/meta-llama/Llama-3.3-70B-Instruct-Turbo","object":"model","created":1767692616,"owned_by":"together_ai","model_name":"meta-llama/Llama-3.3-70B-Instruct-Turbo","context_length":131072,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.0399999999999998e-6,"output_cost_per_token":1.0399999999999998e-6},"list_pricing":{"input_cost_per_token":1.0399999999999998e-6,"output_cost_per_token":1.0399999999999998e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"xai/grok-3-fast-beta","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-3-fast-beta","context_length":131072,"description":"","source":"https://x.ai/api#pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":1.25e-6},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":1.25e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-3-fast-latest","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-3-fast-latest","context_length":131072,"description":"","source":"https://x.ai/api#pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":1.25e-6},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":1.25e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-3-latest","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-3-latest","context_length":131072,"description":"","source":"https://x.ai/api#pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":7.5e-7},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":7.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-3-mini-fast","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-3-mini-fast","context_length":131072,"description":"","source":"https://x.ai/api#pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-3-mini-fast-beta","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-3-mini-fast-beta","context_length":131072,"description":"","source":"https://x.ai/api#pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-3-mini-fast-latest","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-3-mini-fast-latest","context_length":131072,"description":"","source":"https://x.ai/api#pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-3-mini-latest","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-3-mini-latest","context_length":131072,"description":"","source":"https://x.ai/api#pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":7.5e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4-0709","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-4-0709","context_length":256000,"description":"","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"input_cost_per_token_above_128k_tokens":6e-6,"output_cost_per_token_above_128k_tokens":0.00003},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"input_cost_per_token_above_128k_tokens":6e-6,"output_cost_per_token_above_128k_tokens":0.00003},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4-1-fast-non-reasoning","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-4-1-fast-non-reasoning","context_length":2000000,"description":"","source":"https://docs.x.ai/docs/models/grok-4-1-fast-non-reasoning","capabilities":{"input_modalities":["audio","image","text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4-1-fast-non-reasoning-latest","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-4-1-fast-non-reasoning-latest","context_length":2000000,"description":"","source":"https://docs.x.ai/docs/models/grok-4-1-fast-non-reasoning","capabilities":{"input_modalities":["audio","image","text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4-1-fast-reasoning","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-4-1-fast-reasoning","context_length":2000000,"description":"","source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","capabilities":{"input_modalities":["audio","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4-1-fast-reasoning-latest","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-4-1-fast-reasoning-latest","context_length":2000000,"description":"","source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","capabilities":{"input_modalities":["audio","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4-fast-non-reasoning","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-4-fast-non-reasoning","context_length":2000000,"description":"","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4-fast-reasoning","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-4-fast-reasoning","context_length":2000000,"description":"","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4-latest","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-4-latest","context_length":256000,"description":"","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"input_cost_per_token_above_128k_tokens":6e-6,"output_cost_per_token_above_128k_tokens":0.00003},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"input_cost_per_token_above_128k_tokens":6e-6,"output_cost_per_token_above_128k_tokens":0.00003},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-code-fast","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-code-fast","context_length":256000,"description":"","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":2e-8},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":2e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-code-fast-1-0825","object":"model","created":1767692616,"owned_by":"xai","model_name":"grok-code-fast-1-0825","context_length":256000,"description":"","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":2e-8},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":2e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"bytedance/seed-1-6-flash-250715","object":"model","created":1766505011,"owned_by":"bytedance","model_name":"seed-1-6-flash-250715","context_length":262144,"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-8,"output_cost_per_token":3e-7,"input_cost_per_token_above_128k_tokens":1e-7,"output_cost_per_token_above_128k_tokens":8e-7},"list_pricing":{"input_cost_per_token":7.5e-8,"output_cost_per_token":3e-7,"input_cost_per_token_above_128k_tokens":1e-7,"output_cost_per_token_above_128k_tokens":8e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"bytedance/seed-1-6-250915","object":"model","created":1766504997,"owned_by":"bytedance","model_name":"seed-1-6-250915","context_length":262144,"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"input_cost_per_token_above_128k_tokens":5e-7,"output_cost_per_token_above_128k_tokens":4e-6},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"input_cost_per_token_above_128k_tokens":5e-7,"output_cost_per_token_above_128k_tokens":4e-6},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"scaleway/devstral-2-123b-instruct-2512","object":"model","created":1766496679,"owned_by":"scaleway","model_name":"devstral-2-123b-instruct-2512","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4.593999547491045e-7,"output_cost_per_token":2.296999773745522e-6},"list_pricing":{"input_cost_per_token":4.593999547491045e-7,"output_cost_per_token":2.296999773745522e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"minimax/MiniMax-M2.1","object":"model","created":1766454997,"owned_by":"minimax","model_name":"MiniMax-M2.1","context_length":204800,"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":3e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":3e-8},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"deepinfra/zai-org/GLM-4.7","object":"model","created":1766378014,"owned_by":"deepinfra","model_name":"zai-org/GLM-4.7","context_length":202752,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.75e-6,"cache_read_input_token_cost":8.000000000000001e-8},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.75e-6,"cache_read_input_token_cost":8.000000000000001e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"vertex/zai-org/glm-4.7-maas@eu","object":"model","created":1766378014,"owned_by":"vertex","model_name":"zai-org/glm-4.7-maas","context_length":200000,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":2.42e-6},"list_pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":2.42e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/zai-org/glm-4.7-maas","object":"model","created":1766378014,"owned_by":"vertex","model_name":"zai-org/glm-4.7-maas","context_length":200000,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.2e-6},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.2e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/zai-org/glm-4.7-maas@us","object":"model","created":1766378014,"owned_by":"vertex","model_name":"zai-org/glm-4.7-maas","context_length":200000,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":2.42e-6},"list_pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":2.42e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"zai/glm-4.7","object":"model","created":1766378014,"owned_by":"zai","model_name":"glm-4.7","context_length":202752,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":1.1e-7,"cache_creation_input_token_cost":0.0},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":1.1e-7,"cache_creation_input_token_cost":0.0},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"google/gemini-3-flash-preview","object":"model","created":1765987078,"owned_by":"google","model_name":"gemini-3-flash-preview","context_length":1048576,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","capabilities":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":5e-7,"cache_read_input_token_cost":5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":3e-6,"cache_read_input_audio_token_cost":1e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":5e-7,"cache_read_input_token_cost":5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":3e-6,"cache_read_input_audio_token_cost":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-3-flash-preview","object":"model","created":1765987078,"owned_by":"vertex","model_name":"gemini-3-flash-preview","context_length":1048576,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","source":null,"capabilities":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":5e-7,"cache_read_input_token_cost":5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":3e-6,"cache_read_input_audio_token_cost":1e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":3e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":5e-7,"cache_read_input_token_cost":5e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":3e-6,"cache_read_input_audio_token_cost":1e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/nvidia/Nemotron-3-Nano-30B-A3B","object":"model","created":1765731275,"owned_by":"deepinfra","model_name":"nvidia/Nemotron-3-Nano-30B-A3B","context_length":262144,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":2.0000000000000002e-7,"cache_read_input_token_cost":2.5000000000000002e-8},"list_pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":2.0000000000000002e-7,"cache_read_input_token_cost":2.5000000000000002e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.2-pro","object":"model","created":1765389780,"owned_by":"openai","model_name":"gpt-5.2-pro","context_length":400000,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000021,"output_cost_per_token":0.000168,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.000021,"output_cost_per_token":0.000168,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.2-pro-2025-12-11","object":"model","created":1765389780,"owned_by":"openai","model_name":"gpt-5.2-pro-2025-12-11","context_length":400000,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000021,"output_cost_per_token":0.000168,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.000021,"output_cost_per_token":0.000168,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.2","object":"model","created":1765389775,"owned_by":"openai","model_name":"gpt-5.2","context_length":400000,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.2-2025-12-11","object":"model","created":1765389775,"owned_by":"openai","model_name":"gpt-5.2-2025-12-11","context_length":400000,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.75e-6,"output_cost_per_token":0.000014,"cache_read_input_token_cost":1.75e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"mistral/devstral-2512","object":"model","created":1765285419,"owned_by":"mistral","model_name":"devstral-2512","context_length":262144,"description":null,"source":"https://mistral.ai/news/devstral-2-vibe-cli","capabilities":{"input_modalities":["text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"zai/glm-4.6v","object":"model","created":1765207462,"owned_by":"zai","model_name":"glm-4.6v","context_length":131072,"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":9e-7,"cache_read_input_token_cost":5e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":9e-7,"cache_read_input_token_cost":5e-8},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"amazon/amazon.nova-2-lite-v1:0@eu","object":"model","created":1764696672,"owned_by":"amazon","model_name":"amazon.nova-2-lite-v1:0","context_length":1000000,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","source":null,"capabilities":{"input_modalities":["text","image","video","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":7.5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/amazon.nova-2-lite-v1:0","object":"model","created":1764696672,"owned_by":"amazon","model_name":"amazon.nova-2-lite-v1:0","context_length":1000000,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","source":null,"capabilities":{"input_modalities":["text","image","video","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":7.5e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"amazon/amazon.nova-2-lite-v1:0@us","object":"model","created":1764696672,"owned_by":"amazon","model_name":"amazon.nova-2-lite-v1:0","context_length":1000000,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","source":null,"capabilities":{"input_modalities":["text","image","video","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":7.5e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"mistral/ministral-14b-2512","object":"model","created":1764681735,"owned_by":"mistral","model_name":"ministral-14b-2512","context_length":262144,"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2e-7,"cache_read_input_token_cost":2e-8},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2e-7,"cache_read_input_token_cost":2e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/ministral-8b-2512","object":"model","created":1764681654,"owned_by":"mistral","model_name":"ministral-8b-2512","context_length":262144,"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","source":"https://mistral.ai/pricing","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.5e-7,"cache_read_input_token_cost":1.5e-8},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.5e-7,"cache_read_input_token_cost":1.5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/ministral-3b-2512","object":"model","created":1764681560,"owned_by":"mistral","model_name":"ministral-3b-2512","context_length":131072,"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":1e-7,"cache_read_input_token_cost":1e-8},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":1e-7,"cache_read_input_token_cost":1e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"mistral/mistral-large-2512","object":"model","created":1764624472,"owned_by":"mistral","model_name":"mistral-large-2512","context_length":262144,"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","source":"https://docs.mistral.ai/models/mistral-large-3-25-12","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":5e-8},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-V3.2","object":"model","created":1764594642,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-V3.2","context_length":163840,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.6e-7,"output_cost_per_token":3.8e-7,"cache_read_input_token_cost":1.3e-7},"list_pricing":{"input_cost_per_token":2.6e-7,"output_cost_per_token":3.8e-7,"cache_read_input_token_cost":1.3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/deepseek-v3.2","object":"model","created":1764594642,"owned_by":"qwen","model_name":"deepseek-v3.2","context_length":131072,"description":"DeepSeek-V3.2 is the official release of a model that incorporates DeepSeek Sparse Attention—a sparse attention mechanism. It’s also the first model launched by DeepSeek that integrates reasoning into tool usage, supporting both reasoning-enabled and non-reasoning tool calls.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.705e-7,"output_cost_per_token":1.1115e-6,"cache_read_input_token_cost":7.410000000000001e-8,"cache_creation_input_token_cost":4.6345e-7},"list_pricing":{"input_cost_per_token":5.699999999999999e-7,"output_cost_per_token":1.71e-6,"cache_read_input_token_cost":1.14e-7,"cache_creation_input_token_cost":7.13e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/deepseek-ai/deepseek-v3.2-maas","object":"model","created":1764594642,"owned_by":"vertex","model_name":"deepseek-ai/deepseek-v3.2-maas","context_length":163840,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.6e-7,"output_cost_per_token":1.68e-6},"list_pricing":{"input_cost_per_token":5.6e-7,"output_cost_per_token":1.68e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/deepseek-ai/deepseek-v3.2-maas@us","object":"model","created":1764594642,"owned_by":"vertex","model_name":"deepseek-ai/deepseek-v3.2-maas","context_length":163840,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.160000000000001e-7,"output_cost_per_token":1.8480000000000001e-6},"list_pricing":{"input_cost_per_token":6.160000000000001e-7,"output_cost_per_token":1.8480000000000001e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"anthropic/claude-opus-4-5","object":"model","created":1764010580,"owned_by":"anthropic","model_name":"claude-opus-4-5","context_length":200000,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"supports_output_config":true},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"anthropic/claude-opus-4-5-20251101","object":"model","created":1764010580,"owned_by":"anthropic","model_name":"claude-opus-4-5-20251101","context_length":200000,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-opus-4-5","object":"model","created":1764010580,"owned_by":"vertex","model_name":"claude-opus-4-5","context_length":200000,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"google/gemini-3-pro-image","object":"model","created":1763653797,"owned_by":"google","model_name":"gemini-3-pro-image","context_length":65536,"description":null,"source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["image","text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"output_cost_per_image_token":0.00012,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"output_cost_per_image_token":0.00012,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"google/gemini-3-pro-image-preview","object":"model","created":1763653797,"owned_by":"google","model_name":"gemini-3-pro-image-preview","context_length":65536,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","source":"https://ai.google.dev/gemini-api/docs/pricing","capabilities":{"input_modalities":["image","text"],"output_modalities":["image","text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"supports_service_tier":true},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"output_cost_per_image_token":0.00012,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"output_cost_per_image_token":0.00012,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"xai/grok-4-1-fast","object":"model","created":1763587502,"owned_by":"xai","model_name":"grok-4-1-fast","context_length":2000000,"description":"Grok 4.1 Fast is xAI's best agentic tool calling model that shines in real-world use cases like customer support and deep research. 2M context window. Reasoning can be enabled/disabled using...","source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","capabilities":{"input_modalities":["audio","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"input_cost_per_token_above_128k_tokens":4e-7,"output_cost_per_token_above_128k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"together_ai/deepcogito/cogito-v2-1-671b","object":"model","created":1763071233,"owned_by":"together_ai","model_name":"deepcogito/cogito-v2-1-671b","context_length":163840,"description":"Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":1.25e-6},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":1.25e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"openai/gpt-5.1","object":"model","created":1763060305,"owned_by":"openai","model_name":"gpt-5.1","context_length":400000,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5.1-2025-11-13","object":"model","created":1763060305,"owned_by":"openai","model_name":"gpt-5.1-2025-11-13","context_length":400000,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"vertex/moonshotai/kimi-k2-thinking-maas","object":"model","created":1762440622,"owned_by":"vertex","model_name":"moonshotai/kimi-k2-thinking-maas","context_length":262144,"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.5e-6},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.5e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"groq/openai/gpt-oss-safeguard-20b","object":"model","created":1761752836,"owned_by":"groq","model_name":"openai/gpt-oss-safeguard-20b","context_length":131072,"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-8,"output_cost_per_token":3e-7,"cache_read_input_token_cost":3.75e-8,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"list_pricing":{"input_cost_per_token":7.5e-8,"output_cost_per_token":3e-7,"cache_read_input_token_cost":3.75e-8,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"minimax/MiniMax-M2","object":"model","created":1761252093,"owned_by":"minimax","model_name":"MiniMax-M2","context_length":204800,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"vertex/minimaxai/minimax-m2-maas","object":"model","created":1761252093,"owned_by":"vertex","model_name":"minimaxai/minimax-m2-maas","context_length":196608,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":1.2e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"qwen/qwen3-vl-32b-instruct","object":"model","created":1761231332,"owned_by":"qwen","model_name":"qwen3-vl-32b-instruct","context_length":131072,"description":"The largest dense model in the Qwen3-VL series, in its non-inference version, delivers overall performance second only to Qwen3-VL-235B-Instruct. It excels in document recognition and comprehension, demonstrates strong spatial awareness and object identification capabilities, and achieves state-of-the-art performance in 2D visual detection and spatial reasoning. It is well-suited for complex perception tasks across a wide range of general-purpose scenarios.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.04e-7,"output_cost_per_token":4.16e-7},"list_pricing":{"input_cost_per_token":1.6e-7,"output_cost_per_token":6.4e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"cloudflare/@cf/ibm-granite/granite-4.0-h-micro","object":"model","created":1760927695,"owned_by":"cloudflare","model_name":"@cf/ibm-granite/granite-4.0-h-micro","context_length":131000,"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.7e-8,"output_cost_per_token":1.12e-7},"list_pricing":{"input_cost_per_token":1.7e-8,"output_cost_per_token":1.12e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"anthropic/claude-haiku-4-5","object":"model","created":1760547638,"owned_by":"anthropic","model_name":"claude-haiku-4-5","context_length":200000,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":1.25e-6,"cache_creation_input_token_cost_above_1hr":2e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":1.25e-6,"cache_creation_input_token_cost_above_1hr":2e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"anthropic/claude-haiku-4-5-20251001","object":"model","created":1760547638,"owned_by":"anthropic","model_name":"claude-haiku-4-5-20251001","context_length":200000,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":1.25e-6,"cache_creation_input_token_cost_above_1hr":2e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":1.25e-6,"cache_creation_input_token_cost_above_1hr":2e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-haiku-4-5","object":"model","created":1760547638,"owned_by":"vertex","model_name":"claude-haiku-4-5","context_length":200000,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":1.25e-6,"cache_creation_input_token_cost_above_1hr":2e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":1.25e-6,"cache_creation_input_token_cost_above_1hr":2e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"qwen/qwen3-vl-8b-thinking","object":"model","created":1760463746,"owned_by":"qwen","model_name":"qwen3-vl-8b-thinking","context_length":131072,"description":"The Thinking version of the 8B Dense model in the Qwen3-VL series consumes less GPU memory and is capable of performing multimodal understanding and reasoning. It supports extremely long contexts such as lengthy videos and documents, 2D/3D visual localization, and features comprehensively upgraded image/video understanding, spatial awareness, and object recognition abilities.","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.17e-7,"output_cost_per_token":1.3650000000000003e-6},"list_pricing":{"input_cost_per_token":1.8e-7,"output_cost_per_token":2.1000000000000002e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-vl-8b-instruct","object":"model","created":1760463308,"owned_by":"qwen","model_name":"qwen3-vl-8b-instruct","context_length":131072,"description":"The Instruct version of the 8B Dense model in the Qwen3-VL series requires less GPU memory and provides comprehensively upgraded image/video understanding, support for extremely long contexts like lengthy videos and documents, spatial awareness, and object recognition capabilities, making it suitable for tackling complex real-world tasks.","source":null,"capabilities":{"input_modalities":["image","text","video"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.17e-7,"output_cost_per_token":4.55e-7},"list_pricing":{"input_cost_per_token":1.8e-7,"output_cost_per_token":7e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/o3-deep-research","object":"model","created":1760129661,"owned_by":"openai","model_name":"o3-deep-research","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00004,"cache_read_input_token_cost":2.5e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00004,"cache_read_input_token_cost":2.5e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o3-deep-research-2025-06-26","object":"model","created":1760129661,"owned_by":"openai","model_name":"o3-deep-research-2025-06-26","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00004,"cache_read_input_token_cost":2.5e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00004,"cache_read_input_token_cost":2.5e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o4-mini-deep-research","object":"model","created":1760129642,"owned_by":"openai","model_name":"o4-mini-deep-research","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o4-mini-deep-research-2025-06-26","object":"model","created":1760129642,"owned_by":"openai","model_name":"o4-mini-deep-research-2025-06-26","context_length":200000,"description":null,"source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/nvidia/Llama-3.3-Nemotron-Super-49B-v1.5","object":"model","created":1760101395,"owned_by":"deepinfra","model_name":"nvidia/Llama-3.3-Nemotron-Super-49B-v1.5","context_length":131072,"description":"Llama-3.3-Nemotron-Super-49B-v1.5 is a 49B-parameter, English-centric reasoning/chat model derived from Meta’s Llama-3.3-70B-Instruct with a 128K context. It’s post-trained for agentic workflows (RAG, tool calling) via SFT across math, code, science, and...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":4e-7},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"google/gemini-2.5-flash-image","object":"model","created":1759870431,"owned_by":"google","model_name":"gemini-2.5-flash-image","context_length":32768,"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-2.5-flash-image","capabilities":{"input_modalities":["audio","image","text","video"],"output_modalities":["image","text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"supports_url_context":true,"supports_service_tier":true},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"output_cost_per_image_token":0.00003,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":1e-7},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"output_cost_per_image_token":0.00003,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-2.5-flash-image","object":"model","created":1759870431,"owned_by":"vertex","model_name":"gemini-2.5-flash-image","context_length":32768,"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["image","text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"output_cost_per_image_token":0.00003,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":1e-7},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"output_cost_per_image_token":0.00003,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-vl-30b-a3b-thinking","object":"model","created":1759794479,"owned_by":"qwen","model_name":"qwen3-vl-30b-a3b-thinking","context_length":131072,"description":"The Thinking version of the second-largest MoE model in the Qwen3-VL series features fast response speeds and enhanced multimodal understanding and reasoning capabilities, visual agents, and support for extremely long contexts such as lengthy videos and documents. It also boasts comprehensively upgraded image/video understanding, spatial awareness, and object recognition abilities, making it well-suited for complex real-world tasks.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":1.5599999999999999e-6},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":2.4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-VL-30B-A3B-Instruct","object":"model","created":1759794476,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-VL-30B-A3B-Instruct","context_length":262144,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3-vl-30b-a3b-instruct","object":"model","created":1759794476,"owned_by":"qwen","model_name":"qwen3-vl-30b-a3b-instruct","context_length":131072,"description":"The Instruct version of the second-largest MoE model in the Qwen3-VL series offers rapid response speeds and supports extremely long contexts like lengthy videos and documents. It includes comprehensively upgraded image/video understanding, spatial awareness, and object recognition capabilities, as well as 2D/3D visual localization, enabling it to handle intricate real-world challenges.","source":null,"capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":5.200000000000001e-7},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.000000000000001e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-5-pro","object":"model","created":1759776663,"owned_by":"openai","model_name":"gpt-5-pro","context_length":400000,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.00012,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.00012,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5-pro-2025-10-06","object":"model","created":1759776663,"owned_by":"openai","model_name":"gpt-5-pro-2025-10-06","context_length":400000,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.00012,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.00012,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/zai-org/GLM-4.6","object":"model","created":1759235576,"owned_by":"deepinfra","model_name":"zai-org/GLM-4.6","context_length":202752,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":1.0000000000000001e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"zai/glm-4.6","object":"model","created":1759235576,"owned_by":"zai","model_name":"glm-4.6","context_length":202752,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":1.1e-7,"cache_creation_input_token_cost":0.0},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":1.1e-7,"cache_creation_input_token_cost":0.0},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"anthropic/claude-sonnet-4-5-20250929","object":"model","created":1759161676,"owned_by":"anthropic","model_name":"claude-sonnet-4-5-20250929","context_length":1000000,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"input_cost_per_token_above_200k_tokens":6e-6,"output_cost_per_token_above_200k_tokens":0.0000225,"cache_creation_input_token_cost_above_1hr":6e-6,"cache_read_input_token_cost_above_200k_tokens":6e-7,"cache_creation_input_token_cost_above_200k_tokens":7.5e-6,"cache_creation_input_token_cost_above_1hr_above_200k_tokens":0.000012},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"input_cost_per_token_above_200k_tokens":6e-6,"output_cost_per_token_above_200k_tokens":0.0000225,"cache_creation_input_token_cost_above_1hr":6e-6,"cache_read_input_token_cost_above_200k_tokens":6e-7,"cache_creation_input_token_cost_above_200k_tokens":7.5e-6,"cache_creation_input_token_cost_above_1hr_above_200k_tokens":0.000012},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-sonnet-4-5","object":"model","created":1759161676,"owned_by":"vertex","model_name":"claude-sonnet-4-5","context_length":1000000,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"input_cost_per_token_above_200k_tokens":6e-6,"output_cost_per_token_above_200k_tokens":0.0000225,"cache_creation_input_token_cost_above_1hr":6e-6,"cache_read_input_token_cost_above_200k_tokens":6e-7,"cache_creation_input_token_cost_above_200k_tokens":7.5e-6,"cache_creation_input_token_cost_above_1hr_above_200k_tokens":0.000012},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":3e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":3.75e-6,"input_cost_per_token_above_200k_tokens":6e-6,"output_cost_per_token_above_200k_tokens":0.0000225,"cache_creation_input_token_cost_above_1hr":6e-6,"cache_read_input_token_cost_above_200k_tokens":6e-7,"cache_creation_input_token_cost_above_200k_tokens":7.5e-6,"cache_creation_input_token_cost_above_1hr_above_200k_tokens":0.000012},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"qwen/qwen3-vl-235b-a22b-thinking","object":"model","created":1758668690,"owned_by":"qwen","model_name":"qwen3-vl-235b-a22b-thinking","context_length":131072,"description":"Qwen3 series VL models feature significantly enhanced multimodal reasoning capabilities, with a particular focus on optimizing the model for STEM and mathematical reasoning. Visual perception and recognition abilities have been comprehensively improved, and OCR capabilities have undergone a major upgrade.","source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":2.6e-6},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-VL-235B-A22B-Instruct","object":"model","created":1758668687,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-VL-235B-A22B-Instruct","context_length":262144,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.799999999999999e-7,"cache_read_input_token_cost":1.1000000000000002e-7},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.799999999999999e-7,"cache_read_input_token_cost":1.1000000000000002e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3-vl-235b-a22b-instruct","object":"model","created":1758668687,"owned_by":"qwen","model_name":"qwen3-vl-235b-a22b-instruct","context_length":131072,"description":"The Qwen3 series VL models has been comprehensively upgraded in areas such as visual coding and spatial perception. Its visual perception and recognition capabilities have significantly improved, supporting the understanding of ultra-long videos, and its OCR functionality has undergone a major enhancement.","source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","capabilities":{"input_modalities":["text","image","video"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":1.0400000000000002e-6},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.6000000000000001e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-max","object":"model","created":1758662808,"owned_by":"qwen","model_name":"qwen3-max","context_length":262144,"description":"The Qwen 3 series Max model has undergone specialized upgrades in agent programming and tool invocation compared to the preview version. The officially released model this time has achieved state-of-the-art (SOTA) performance in its field and is better suited to meet the demands of agents operating in more complex scenarios.","source":"https://www.alibabacloud.com/help/en/model-studio/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.9e-6,"cache_read_input_token_cost":1.56e-7,"cache_creation_input_token_cost":9.75e-7},{"range":[32000,128000],"input_cost_per_token":1.5599999999999999e-6,"output_cost_per_token":7.8e-6,"cache_read_input_token_cost":3.12e-7,"cache_creation_input_token_cost":1.95e-6},{"range":[128000,256000],"input_cost_per_token":1.95e-6,"output_cost_per_token":9.75e-6,"cache_read_input_token_cost":3.8999999999999997e-7,"cache_creation_input_token_cost":2.4375e-6}],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.9e-6,"cache_read_input_token_cost":1.56e-7,"cache_creation_input_token_cost":9.75e-7},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2.4e-7,"cache_creation_input_token_cost":1.5e-6},{"range":[32000,128000],"input_cost_per_token":2.4e-6,"output_cost_per_token":0.000012,"cache_read_input_token_cost":4.8e-7,"cache_creation_input_token_cost":3e-6},{"range":[128000,256000],"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":6e-7,"cache_creation_input_token_cost":3.75e-6}],"input_cost_per_token":1.2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2.4e-7,"cache_creation_input_token_cost":1.5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-max-2025-09-23","object":"model","created":1758662808,"owned_by":"qwen","model_name":"qwen3-max-2025-09-23","context_length":262144,"description":"The Qwen 3 series Max model has undergone specialized upgrades in agent programming and tool invocation compared to the preview version. The officially released model this time has achieved state-of-the-art (SOTA) performance in its field and is better suited to meet the demands of agents operating in more complex scenarios.This version is a snapshot as of September 23, 2025.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.9e-6},{"range":[32000,128000],"input_cost_per_token":1.5599999999999999e-6,"output_cost_per_token":7.8e-6},{"range":[128000,256000],"input_cost_per_token":1.95e-6,"output_cost_per_token":9.75e-6}],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.9e-6},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.2e-6,"output_cost_per_token":6e-6},{"range":[32000,128000],"input_cost_per_token":2.4e-6,"output_cost_per_token":0.000012},{"range":[128000,256000],"input_cost_per_token":3e-6,"output_cost_per_token":0.000015}],"input_cost_per_token":1.2e-6,"output_cost_per_token":6e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-max-2026-01-23","object":"model","created":1758662808,"owned_by":"qwen","model_name":"qwen3-max-2026-01-23","context_length":262144,"description":"Compared with the snapshot as of September 23, 2025, the Qwen-3 series Max model in this release achieves an effective integration of thinking and non-thinking modes, resulting in a comprehensive and substantial improvement in the model’s overall performance. In thinking mode, the model simultaneously supports web search, web information extraction, and a code interpreter tool, enabling it to tackle more complex and challenging problems with greater accuracy by leveraging external tools while engaging in slow, deliberative reasoning. This version is based on a snapshot taken on January 23, 2026.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.9e-6},{"range":[32000,128000],"input_cost_per_token":1.5599999999999999e-6,"output_cost_per_token":7.8e-6},{"range":[128000,256000],"input_cost_per_token":1.95e-6,"output_cost_per_token":9.75e-6}],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.9e-6},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.2e-6,"output_cost_per_token":6e-6},{"range":[32000,128000],"input_cost_per_token":2.4e-6,"output_cost_per_token":0.000012},{"range":[128000,256000],"input_cost_per_token":3e-6,"output_cost_per_token":0.000015}],"input_cost_per_token":1.2e-6,"output_cost_per_token":6e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-max-preview","object":"model","created":1758662808,"owned_by":"qwen","model_name":"qwen3-max-preview","context_length":262144,"description":"A preview version of the Max model in the Qwen 3 series, achieving an effective integration of thinking and non-thinking modes. In thinking mode, there is a significant enhancement in capabilities such as intelligent agent programming, common-sense reasoning, and reasoning across mathematics, science, and general domains.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.9e-6,"cache_read_input_token_cost":1.56e-7},{"range":[32000,128000],"input_cost_per_token":1.5599999999999999e-6,"output_cost_per_token":7.8e-6,"cache_read_input_token_cost":3.12e-7},{"range":[128000,256000],"input_cost_per_token":1.95e-6,"output_cost_per_token":9.75e-6,"cache_read_input_token_cost":3.8999999999999997e-7}],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.9e-6,"cache_read_input_token_cost":1.56e-7},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2.4e-7},{"range":[32000,128000],"input_cost_per_token":2.4e-6,"output_cost_per_token":0.000012,"cache_read_input_token_cost":4.8e-7},{"range":[128000,256000],"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":6e-7}],"input_cost_per_token":1.2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2.4e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-coder-plus","object":"model","created":1758662707,"owned_by":"qwen","model_name":"qwen3-coder-plus","context_length":1000000,"description":"Qwen3-based code generation model with strong coding agent power, excels at tool calling and environment interaction, capable of autonomous programming with outstanding code capability while maintaining general ability.","source":"https://www.alibabacloud.com/help/en/model-studio/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":6.5e-7,"output_cost_per_token":3.2500000000000002e-6,"cache_read_input_token_cost":1.3000000000000003e-7,"cache_creation_input_token_cost":8.125000000000001e-7},{"range":[32000,128000],"input_cost_per_token":1.17e-6,"output_cost_per_token":5.850000000000001e-6,"cache_read_input_token_cost":2.34e-7,"cache_creation_input_token_cost":1.4625000000000002e-6},{"range":[128000,256000],"input_cost_per_token":1.95e-6,"output_cost_per_token":9.75e-6,"cache_read_input_token_cost":3.8999999999999997e-7,"cache_creation_input_token_cost":2.4375e-6},{"range":[256000,1000000],"input_cost_per_token":3.9e-6,"output_cost_per_token":0.000039,"cache_read_input_token_cost":7.799999999999999e-7,"cache_creation_input_token_cost":4.875e-6}],"input_cost_per_token":6.5e-7,"output_cost_per_token":3.2500000000000002e-6,"cache_read_input_token_cost":1.3000000000000003e-7,"cache_creation_input_token_cost":8.125000000000001e-7},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":2.0000000000000002e-7,"cache_creation_input_token_cost":1.25e-6},{"range":[32000,128000],"input_cost_per_token":1.8000000000000001e-6,"output_cost_per_token":9e-6,"cache_read_input_token_cost":3.6e-7,"cache_creation_input_token_cost":2.25e-6},{"range":[128000,256000],"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":6e-7,"cache_creation_input_token_cost":3.75e-6},{"range":[256000,1000000],"input_cost_per_token":6e-6,"output_cost_per_token":0.00006,"cache_read_input_token_cost":1.2e-6,"cache_creation_input_token_cost":7.5e-6}],"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":2.0000000000000002e-7,"cache_creation_input_token_cost":1.25e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-coder-plus-2025-07-22","object":"model","created":1758662707,"owned_by":"qwen","model_name":"qwen3-coder-plus-2025-07-22","context_length":1000000,"description":"Qwen3-based code generation model with strong coding agent power, excels at tool calling and environment interaction, capable of autonomous programming with outstanding code capability while maintaining general ability. This is a snapshot from 22 July, 2025.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":6.5e-7,"output_cost_per_token":3.2500000000000002e-6},{"range":[32000,128000],"input_cost_per_token":1.17e-6,"output_cost_per_token":5.850000000000001e-6},{"range":[128000,256000],"input_cost_per_token":1.95e-6,"output_cost_per_token":9.75e-6},{"range":[256000,1000000],"input_cost_per_token":3.9e-6,"output_cost_per_token":0.000039}],"input_cost_per_token":6.5e-7,"output_cost_per_token":3.2500000000000002e-6},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1e-6,"output_cost_per_token":5e-6},{"range":[32000,128000],"input_cost_per_token":1.8000000000000001e-6,"output_cost_per_token":9e-6},{"range":[128000,256000],"input_cost_per_token":3e-6,"output_cost_per_token":0.000015},{"range":[256000,1000000],"input_cost_per_token":6e-6,"output_cost_per_token":0.00006}],"input_cost_per_token":1e-6,"output_cost_per_token":5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-coder-plus-2025-09-23","object":"model","created":1758662707,"owned_by":"qwen","model_name":"qwen3-coder-plus-2025-09-23","context_length":1000000,"description":"Qwen3-based code generation model with strong coding agent power, excels at tool calling and environment interaction, capable of autonomous programming with outstanding code capability while maintaining general ability. This is a snapshot from 23 September , 2025.Compared to the previous version (snapshot from July 22), it demonstrates improved robustness in downstream task performance and tool invocation, along with enhanced code security.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":6.5e-7,"output_cost_per_token":3.2500000000000002e-6},{"range":[32000,128000],"input_cost_per_token":1.17e-6,"output_cost_per_token":5.850000000000001e-6},{"range":[128000,256000],"input_cost_per_token":1.95e-6,"output_cost_per_token":9.75e-6},{"range":[256000,1000000],"input_cost_per_token":3.9e-6,"output_cost_per_token":0.000039}],"input_cost_per_token":6.5e-7,"output_cost_per_token":3.2500000000000002e-6},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1e-6,"output_cost_per_token":5e-6},{"range":[32000,128000],"input_cost_per_token":1.8000000000000001e-6,"output_cost_per_token":9e-6},{"range":[128000,256000],"input_cost_per_token":3e-6,"output_cost_per_token":0.000015},{"range":[256000,1000000],"input_cost_per_token":6e-6,"output_cost_per_token":0.00006}],"input_cost_per_token":1e-6,"output_cost_per_token":5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-V3.1-Terminus","object":"model","created":1758548275,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-V3.1-Terminus","context_length":163840,"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.7e-7,"output_cost_per_token":9.5e-7,"cache_read_input_token_cost":1.3e-7},"list_pricing":{"input_cost_per_token":2.7e-7,"output_cost_per_token":9.5e-7,"cache_read_input_token_cost":1.3e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-4-fast","object":"model","created":1758240090,"owned_by":"xai","model_name":"grok-4-fast","context_length":2000000,"description":"Grok 4 Fast is xAI's latest multimodal model with SOTA cost-efficiency and a 2M token context window. It comes in two flavors: non-reasoning and reasoning. Read more about the model...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"search_context_cost_per_query":0.005},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":5e-7,"cache_read_input_token_cost":5e-8,"search_context_cost_per_query":0.005},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3-coder-flash","object":"model","created":1758115536,"owned_by":"qwen","model_name":"qwen3-coder-flash","context_length":1000000,"description":"Based on Qwen3, this code generation model inherits the coding agent capabilities of Qwen3-Coder-Plus and supports multi-turn tool interaction. It features focused optimizations on repository-level understanding and enhanced tool-calling stability.","source":"https://www.alibabacloud.com/help/en/model-studio/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.9499999999999999e-7,"output_cost_per_token":9.75e-7,"cache_read_input_token_cost":3.9e-8,"cache_creation_input_token_cost":2.4375e-7},{"range":[32000,128000],"input_cost_per_token":3.25e-7,"output_cost_per_token":1.6250000000000001e-6,"cache_read_input_token_cost":6.500000000000001e-8,"cache_creation_input_token_cost":4.0625000000000003e-7},{"range":[128000,256000],"input_cost_per_token":5.200000000000001e-7,"output_cost_per_token":2.6e-6,"cache_read_input_token_cost":1.04e-7,"cache_creation_input_token_cost":6.5e-7},{"range":[256000,1000000],"input_cost_per_token":1.0400000000000002e-6,"output_cost_per_token":6.2399999999999995e-6,"cache_read_input_token_cost":2.08e-7,"cache_creation_input_token_cost":1.3e-6}],"input_cost_per_token":1.9499999999999999e-7,"output_cost_per_token":9.75e-7,"cache_read_input_token_cost":3.9e-8,"cache_creation_input_token_cost":2.4375e-7},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":3e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":6e-8,"cache_creation_input_token_cost":3.75e-7},{"range":[32000,128000],"input_cost_per_token":5e-7,"output_cost_per_token":2.5e-6,"cache_read_input_token_cost":1.0000000000000001e-7,"cache_creation_input_token_cost":6.25e-7},{"range":[128000,256000],"input_cost_per_token":8.000000000000001e-7,"output_cost_per_token":4e-6,"cache_read_input_token_cost":1.6e-7,"cache_creation_input_token_cost":1e-6},{"range":[256000,1000000],"input_cost_per_token":1.6000000000000001e-6,"output_cost_per_token":9.6e-6,"cache_read_input_token_cost":3.2e-7,"cache_creation_input_token_cost":2e-6}],"input_cost_per_token":3e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":6e-8,"cache_creation_input_token_cost":3.75e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-coder-flash-2025-07-28","object":"model","created":1758115536,"owned_by":"qwen","model_name":"qwen3-coder-flash-2025-07-28","context_length":1000000,"description":"Based on Qwen3, this code generation model inherits the coding agent capabilities of Qwen3-Coder-Plus and supports multi-turn tool interaction. It features focused optimizations on repository-level understanding and enhanced tool-calling stability. This version is a snapshot dated July 28, 2025.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":1.9499999999999999e-7,"output_cost_per_token":9.75e-7},{"range":[32000,128000],"input_cost_per_token":3.25e-7,"output_cost_per_token":1.6250000000000001e-6},{"range":[128000,256000],"input_cost_per_token":5.200000000000001e-7,"output_cost_per_token":2.6e-6},{"range":[256000,1000000],"input_cost_per_token":1.0400000000000002e-6,"output_cost_per_token":6.2399999999999995e-6}],"input_cost_per_token":1.9499999999999999e-7,"output_cost_per_token":9.75e-7},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":3e-7,"output_cost_per_token":1.5e-6},{"range":[32000,128000],"input_cost_per_token":5e-7,"output_cost_per_token":2.5e-6},{"range":[128000,256000],"input_cost_per_token":8.000000000000001e-7,"output_cost_per_token":4e-6},{"range":[256000,1000000],"input_cost_per_token":1.6000000000000001e-6,"output_cost_per_token":9.6e-6}],"input_cost_per_token":3e-7,"output_cost_per_token":1.5e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"nebius/Qwen/Qwen3-Next-80B-A3B-Thinking","object":"model","created":1757612284,"owned_by":"nebius","model_name":"Qwen/Qwen3-Next-80B-A3B-Thinking","context_length":128000,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":1.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-next-80b-a3b-thinking","object":"model","created":1757612284,"owned_by":"qwen","model_name":"qwen3-next-80b-a3b-thinking","context_length":131072,"description":"A new generation of Qwen3-based open-source thinking mode models. This version offers improved instruction following and streamlined summary responses over the previous iteration (Qwen3-235B-A22B-Thinking-2507).","source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.749999999999999e-8,"output_cost_per_token":7.799999999999999e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/qwen/qwen3-next-80b-a3b-thinking-maas@eu","object":"model","created":1757612284,"owned_by":"vertex","model_name":"qwen/qwen3-next-80b-a3b-thinking-maas","context_length":262144,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":1.32e-6},"list_pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":1.32e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/qwen/qwen3-next-80b-a3b-thinking-maas","object":"model","created":1757612284,"owned_by":"vertex","model_name":"qwen/qwen3-next-80b-a3b-thinking-maas","context_length":262144,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/qwen/qwen3-next-80b-a3b-thinking-maas@us","object":"model","created":1757612284,"owned_by":"vertex","model_name":"qwen/qwen3-next-80b-a3b-thinking-maas","context_length":262144,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":1.32e-6},"list_pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":1.32e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-Next-80B-A3B-Instruct","object":"model","created":1757612213,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-Next-80B-A3B-Instruct","context_length":262144,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":1.1e-6},"list_pricing":{"input_cost_per_token":9e-8,"output_cost_per_token":1.1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3-next-80b-a3b-instruct","object":"model","created":1757612213,"owned_by":"qwen","model_name":"qwen3-next-80b-a3b-instruct","context_length":131072,"description":"A new generation of open-source, non-thinking mode model powered by Qwen3. This version demonstrates superior Chinese text understanding, augmented logical reasoning, and enhanced capabilities in text generation tasks over the previous iteration (Qwen3-235B-A22B-Instruct-2507).","source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":9.749999999999999e-8,"output_cost_per_token":7.799999999999999e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/qwen/qwen3-next-80b-a3b-instruct-maas@eu","object":"model","created":1757612213,"owned_by":"vertex","model_name":"qwen/qwen3-next-80b-a3b-instruct-maas","context_length":262144,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":1.32e-6},"list_pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":1.32e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/qwen/qwen3-next-80b-a3b-instruct-maas","object":"model","created":1757612213,"owned_by":"vertex","model_name":"qwen/qwen3-next-80b-a3b-instruct-maas","context_length":262144,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":1.2e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/qwen/qwen3-next-80b-a3b-instruct-maas@us","object":"model","created":1757612213,"owned_by":"vertex","model_name":"qwen/qwen3-next-80b-a3b-instruct-maas","context_length":262144,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":1.32e-6},"list_pricing":{"input_cost_per_token":1.65e-7,"output_cost_per_token":1.32e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen-plus","object":"model","created":1757347599,"owned_by":"qwen","model_name":"qwen-plus","context_length":1000000,"description":"Qwen-Plus is an enhanced version of the Qwen ultra-large language model that supports multiple input languages such as Chinese and English. Compared to previous versions, it shows significant improvements in both Chinese and English code generation, logical reasoning, and multilingual abilities. The response style has been greatly adjusted to align with human preferences, with noticeable enhancements in the level of detail and clarity of responses. Specialized improvements have been made in creative writing, adherence to JSON formatting, and role-playing abilities.","source":"https://www.alibabacloud.com/help/en/model-studio/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"cache_read_input_token_cost":5.2e-8,"cache_creation_input_token_cost":3.25e-7,"output_cost_per_reasoning_token":2.6e-6},{"range":[256000,1000000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":2.34e-6,"cache_read_input_token_cost":1.56e-7,"cache_creation_input_token_cost":9.75e-7,"output_cost_per_reasoning_token":7.8e-6}],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"cache_read_input_token_cost":5.2e-8,"cache_creation_input_token_cost":3.25e-7,"output_cost_per_reasoning_token":2.6e-6},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":8e-8,"cache_creation_input_token_cost":5e-7,"output_cost_per_reasoning_token":4e-6},{"range":[256000,1000000],"input_cost_per_token":1.2e-6,"output_cost_per_token":3.6000000000000003e-6,"cache_read_input_token_cost":2.4e-7,"cache_creation_input_token_cost":1.5e-6,"output_cost_per_reasoning_token":0.000012}],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"cache_read_input_token_cost":8e-8,"cache_creation_input_token_cost":5e-7,"output_cost_per_reasoning_token":4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-plus-2025-01-25","object":"model","created":1757347599,"owned_by":"qwen","model_name":"qwen-plus-2025-01-25","context_length":131072,"description":"The Qwen series of models, which are well-balanced in capabilities, offer reasoning performance and speed that fall between Qwen-Max and Qwen-Turbo, making them suitable for moderately complex tasks. Compared to previous versions, it shows significant improvements in both Chinese and English code generation, logical reasoning, and multilingual abilities. The response style has been greatly adjusted to align with human preferences, with noticeable enhancements in the level of detail and clarity of responses. Specialized improvements have been made in creative writing, adherence to JSON formatting, and role-playing abilities.","source":"https://www.alibabacloud.com/help/en/model-studio/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-plus-2025-04-28","object":"model","created":1757347599,"owned_by":"qwen","model_name":"qwen-plus-2025-04-28","context_length":131072,"description":"The Plus model of the Qwen3 series, effectively integrates thinking mode and non-thinking mode, allowing for mode switching during conversations. Its reasoning capabilities significantly surpass those of QwQ, and its general capabilities notably exceed those of Qwen2.5-Plus, reaching the SOTA level in the same scale within the industry. This model is the snapshot from April 28, 2025.","source":"https://www.alibabacloud.com/help/en/model-studio/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"output_cost_per_reasoning_token":2.6e-6},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"output_cost_per_reasoning_token":4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-plus-2025-07-14","object":"model","created":1757347599,"owned_by":"qwen","model_name":"qwen-plus-2025-07-14","context_length":131072,"description":"The Plus model of the Qwen3 Series, achieving effective integration of thinking mode and non-thinking mode, allows switching modes during conversations. This is a snapshot from July 14, 2025. Compared to the previous version, there has been a significant improvement in both Chinese and English capabilities under non-thinking mode, with enhanced tool-calling abilities.","source":"https://www.alibabacloud.com/help/en/model-studio/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"output_cost_per_reasoning_token":2.6e-6},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"output_cost_per_reasoning_token":4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-plus-2025-07-28","object":"model","created":1757347599,"owned_by":"qwen","model_name":"qwen-plus-2025-07-28","context_length":1000000,"description":"Qwen3 series Plus model, integrates thinking and non-thinking modes and can switch modes during dialogue. Compared to the prior version, adds dedicated enhancements for Chinese & English capabilities and tool calling. This is a snapshot from 28 July, 2025; first to support 1 M context length, and uses tiered pricing.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"output_cost_per_reasoning_token":2.6e-6},{"range":[256000,1000000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":2.34e-6,"output_cost_per_reasoning_token":7.8e-6}],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"output_cost_per_reasoning_token":2.6e-6},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"output_cost_per_reasoning_token":4e-6},{"range":[256000,1000000],"input_cost_per_token":1.2e-6,"output_cost_per_token":3.6000000000000003e-6,"output_cost_per_reasoning_token":0.000012}],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"output_cost_per_reasoning_token":4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-plus-2025-09-11","object":"model","created":1757347599,"owned_by":"qwen","model_name":"qwen-plus-2025-09-11","context_length":1000000,"description":"As of the September 11, 2025 snapshot, this release features enhanced instruction following and streamlined summary responses in thinking mode. Non-thinking mode provides superior Chinese text understanding and augmented logical reasoning capabilities. The model supports a 1M context length with tiered pricing.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"output_cost_per_reasoning_token":2.6e-6},{"range":[256000,1000000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":2.34e-6,"output_cost_per_reasoning_token":7.8e-6}],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"output_cost_per_reasoning_token":2.6e-6},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"output_cost_per_reasoning_token":4e-6},{"range":[256000,1000000],"input_cost_per_token":1.2e-6,"output_cost_per_token":3.6000000000000003e-6,"output_cost_per_reasoning_token":0.000012}],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"output_cost_per_reasoning_token":4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-plus-2025-12-01","object":"model","created":1757347599,"owned_by":"qwen","model_name":"qwen-plus-2025-12-01","context_length":1000000,"description":"This version is a snapshot as of December 1, 2025, and features improved reasoning capabilities compared to the July 28 snapshot. Agent capabilities and multi-turn tool invocation abilities have been further enhanced, and performance on subjective creative tasks has improved significantly. It supports a context length of up to 1 million tokens, with tiered pricing based on context length.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"output_cost_per_reasoning_token":2.6e-6},{"range":[256000,1000000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":2.34e-6,"output_cost_per_reasoning_token":7.8e-6}],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"output_cost_per_reasoning_token":2.6e-6},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"output_cost_per_reasoning_token":4e-6},{"range":[256000,1000000],"input_cost_per_token":1.2e-6,"output_cost_per_token":3.6000000000000003e-6,"output_cost_per_reasoning_token":0.000012}],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"output_cost_per_reasoning_token":4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen-plus-latest","object":"model","created":1757347599,"owned_by":"qwen","model_name":"qwen-plus-latest","context_length":1000000,"description":"The Qwen series model optimized for balanced performance, offering inference efficiency between Qwen-Max and Qwen-Turbo, is designed to handle moderately complex tasks effectively. This dynamically updated version implements changes without prior notice.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"output_cost_per_reasoning_token":2.6e-6},{"range":[256000,1000000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":2.34e-6,"output_cost_per_reasoning_token":7.8e-6}],"input_cost_per_token":2.6000000000000005e-7,"output_cost_per_token":7.799999999999999e-7,"output_cost_per_reasoning_token":2.6e-6},"list_pricing":{"tiered_pricing":[{"range":[0,256000],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"output_cost_per_reasoning_token":4e-6},{"range":[256000,1000000],"input_cost_per_token":1.2e-6,"output_cost_per_token":3.6000000000000003e-6,"output_cost_per_reasoning_token":0.000012}],"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":1.2e-6,"output_cost_per_reasoning_token":4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-30b-a3b-thinking-2507","object":"model","created":1756399192,"owned_by":"qwen","model_name":"qwen3-30b-a3b-thinking-2507","context_length":81920,"description":"Open-source Qwen3 thinking model; compared to the previous version (Qwen3-30B-A3B) excels in complex thinking tasks, including logic, math, science, code, and other challenging scenarios; instruction following, text understanding, and multilingual translation capabilities significantly improved.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":1.5599999999999999e-6},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":2.4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"xai/grok-code-fast-1","object":"model","created":1756238927,"owned_by":"xai","model_name":"grok-code-fast-1","context_length":256000,"description":"Grok Code Fast 1 is a speedy and economical reasoning model that excels at agentic coding. With reasoning traces visible in the response, developers can steer Grok Code for high-quality...","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":2e-8},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.5e-6,"cache_read_input_token_cost":2e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/NousResearch/Hermes-4-70B","object":"model","created":1756236182,"owned_by":"nebius","model_name":"NousResearch/Hermes-4-70B","context_length":131072,"description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3e-7,"output_cost_per_token":4e-7,"cache_read_input_token_cost":1.3e-7},"list_pricing":{"input_cost_per_token":1.3e-7,"output_cost_per_token":4e-7,"cache_read_input_token_cost":1.3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"nebius/NousResearch/Hermes-4-405B","object":"model","created":1756235463,"owned_by":"nebius","model_name":"NousResearch/Hermes-4-405B","context_length":131072,"description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6,"cache_read_input_token_cost":1e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":3e-6,"cache_read_input_token_cost":1e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/gpt-oss-120b","object":"model","created":1755093042,"owned_by":"scaleway","model_name":"gpt-oss-120b","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":7.008600721185013e-7},"list_pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":7.008600721185013e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/holo2-30b-a3b","object":"model","created":1755006551,"owned_by":"scaleway","model_name":"holo2-30b-a3b","context_length":22000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.4604989808830504e-7,"output_cost_per_token":8.074497622060451e-7},"list_pricing":{"input_cost_per_token":3.4604989808830504e-7,"output_cost_per_token":8.074497622060451e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/mistral-small-3.2-24b-instruct-2506","object":"model","created":1755006551,"owned_by":"scaleway","model_name":"mistral-small-3.2-24b-instruct-2506","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":4.088350420691258e-7},"list_pricing":{"input_cost_per_token":1.7521501802962533e-7,"output_cost_per_token":4.088350420691258e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/qwen3-coder-30b-a3b-instruct","object":"model","created":1755006551,"owned_by":"scaleway","model_name":"qwen3-coder-30b-a3b-instruct","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.3362002403950049e-7,"output_cost_per_token":9.344800961580019e-7},"list_pricing":{"input_cost_per_token":2.3362002403950049e-7,"output_cost_per_token":9.344800961580019e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"zai/glm-4.5v","object":"model","created":1754922288,"owned_by":"zai","model_name":"glm-4.5v","context_length":65536,"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":1.8e-6,"cache_read_input_token_cost":1.1e-7},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":1.8e-6,"cache_read_input_token_cost":1.1e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"openai/gpt-5","object":"model","created":1754587413,"owned_by":"openai","model_name":"gpt-5","context_length":400000,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5-2025-08-07","object":"model","created":1754587413,"owned_by":"openai","model_name":"gpt-5-2025-08-07","context_length":400000,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5-mini","object":"model","created":1754587407,"owned_by":"openai","model_name":"gpt-5-mini","context_length":400000,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5-mini-2025-08-07","object":"model","created":1754587407,"owned_by":"openai","model_name":"gpt-5-mini-2025-08-07","context_length":400000,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5-nano","object":"model","created":1754587402,"owned_by":"openai","model_name":"gpt-5-nano","context_length":400000,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"cache_read_input_token_cost":5e-9,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"cache_read_input_token_cost":5e-9,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-5-nano-2025-08-07","object":"model","created":1754587402,"owned_by":"openai","model_name":"gpt-5-nano-2025-08-07","context_length":400000,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"cache_read_input_token_cost":5e-9,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"cache_read_input_token_cost":5e-9,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"cerebras/gpt-oss-120b","object":"model","created":1754414231,"owned_by":"cerebras","model_name":"gpt-oss-120b","context_length":131072,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","source":"https://www.cerebras.ai/blog/openai-gpt-oss-120b-runs-fastest-on-cerebras","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.5e-7,"output_cost_per_token":7.5e-7,"cache_read_input_token_cost":3.5e-7},"list_pricing":{"input_cost_per_token":3.5e-7,"output_cost_per_token":7.5e-7,"cache_read_input_token_cost":3.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/openai/gpt-oss-120b","object":"model","created":1754414231,"owned_by":"deepinfra","model_name":"openai/gpt-oss-120b","context_length":131072,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.7e-8,"output_cost_per_token":1.7e-7},"list_pricing":{"input_cost_per_token":3.7e-8,"output_cost_per_token":1.7e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/gpt-oss-120b","object":"model","created":1754414231,"owned_by":"fireworks_ai","model_name":"gpt-oss-120b","context_length":131072,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":1.4e-8},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":1.4e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"groq/openai/gpt-oss-120b","object":"model","created":1754414231,"owned_by":"groq","model_name":"openai/gpt-oss-120b","context_length":131072,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/openai/gpt-oss-120b","object":"model","created":1754414231,"owned_by":"nebius","model_name":"openai/gpt-oss-120b","context_length":131072,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":1.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"together_ai/openai/gpt-oss-120b","object":"model","created":1754414231,"owned_by":"together_ai","model_name":"openai/gpt-oss-120b","context_length":131072,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","source":"https://www.together.ai/models/gpt-oss-120b","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/openai/gpt-oss-20b","object":"model","created":1754414229,"owned_by":"deepinfra","model_name":"openai/gpt-oss-20b","context_length":131072,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.0000000000000004e-8,"output_cost_per_token":1.4e-7},"list_pricing":{"input_cost_per_token":3.0000000000000004e-8,"output_cost_per_token":1.4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/accounts/fireworks/models/gpt-oss-20b","object":"model","created":1754414229,"owned_by":"fireworks_ai","model_name":"accounts/fireworks/models/gpt-oss-20b","context_length":131072,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","source":"https://fireworks.ai/pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3e-7,"cache_read_input_token_cost":3.5e-8},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3e-7,"cache_read_input_token_cost":3.5e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"fireworks_ai/gpt-oss-20b","object":"model","created":1754414229,"owned_by":"fireworks_ai","model_name":"gpt-oss-20b","context_length":131072,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3e-7,"cache_read_input_token_cost":3.5e-8},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":3e-7,"cache_read_input_token_cost":3.5e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"groq/openai/gpt-oss-20b","object":"model","created":1754414229,"owned_by":"groq","model_name":"openai/gpt-oss-20b","context_length":131072,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-8,"output_cost_per_token":3e-7,"cache_read_input_token_cost":3.75e-8,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"list_pricing":{"input_cost_per_token":7.5e-8,"output_cost_per_token":3e-7,"cache_read_input_token_cost":3.75e-8,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"together_ai/openai/gpt-oss-20b","object":"model","created":1754414229,"owned_by":"together_ai","model_name":"openai/gpt-oss-20b","context_length":131072,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","source":"https://www.together.ai/models/gpt-oss-20b","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":2.0000000000000002e-7},"list_pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":2.0000000000000002e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"vertex/claude-opus-4-1","object":"model","created":1754411591,"owned_by":"vertex","model_name":"claude-opus-4-1","context_length":200000,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.000075,"cache_read_input_token_cost":1.5e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.00001875,"cache_creation_input_token_cost_above_1hr":0.00003},"list_pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.000075,"cache_read_input_token_cost":1.5e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.00001875,"cache_creation_input_token_cost_above_1hr":0.00003},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"mistral/codestral-2508","object":"model","created":1754079630,"owned_by":"mistral","model_name":"codestral-2508","context_length":256000,"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","source":"https://mistral.ai/news/codestral-25-08","capabilities":{"input_modalities":["text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":9e-7,"cache_read_input_token_cost":3e-8},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":9e-7,"cache_read_input_token_cost":3e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/qwen3-235b-a22b-instruct-2507","object":"model","created":1754049528,"owned_by":"scaleway","model_name":"qwen3-235b-a22b-instruct-2507","context_length":250000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8.760750901481267e-7,"output_cost_per_token":2.62822527044438e-6},"list_pricing":{"input_cost_per_token":8.760750901481267e-7,"output_cost_per_token":2.62822527044438e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct","object":"model","created":1753972379,"owned_by":"qwen","model_name":"qwen3-coder-30b-a3b-instruct","context_length":262144,"description":"Qwen3-based code generation model that inherits the coding agent ability of Qwen3-Coder-480B-A35B-Instruct; code capability reaches SOTA at the same scale.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":2.925e-7,"output_cost_per_token":1.4625000000000002e-6},{"range":[32000,128000],"input_cost_per_token":4.875e-7,"output_cost_per_token":2.4375e-6},{"range":[128000,256000],"input_cost_per_token":7.799999999999999e-7,"output_cost_per_token":3.9e-6},{"range":[256000,1000000],"input_cost_per_token":1.5599999999999999e-6,"output_cost_per_token":9.36e-6}],"input_cost_per_token":2.925e-7,"output_cost_per_token":1.4625000000000002e-6},"list_pricing":{"tiered_pricing":[{"range":[0,32000],"input_cost_per_token":4.5000000000000003e-7,"output_cost_per_token":2.25e-6},{"range":[32000,128000],"input_cost_per_token":7.5e-7,"output_cost_per_token":3.75e-6},{"range":[128000,256000],"input_cost_per_token":1.2e-6,"output_cost_per_token":6e-6},{"range":[256000,1000000],"input_cost_per_token":2.4e-6,"output_cost_per_token":0.000014400000000000001}],"input_cost_per_token":4.5000000000000003e-7,"output_cost_per_token":2.25e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"nebius/Qwen/Qwen3-30B-A3B-Instruct-2507","object":"model","created":1753806965,"owned_by":"nebius","model_name":"Qwen/Qwen3-30B-A3B-Instruct-2507","context_length":262144,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7,"cache_read_input_token_cost":1e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7,"cache_read_input_token_cost":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507","object":"model","created":1753806965,"owned_by":"qwen","model_name":"qwen3-30b-a3b-instruct-2507","context_length":131072,"description":"Open-source Qwen3 non-thinking model; compared to the previous version (Qwen3-30B-A3B) shows major improvements in Chinese, English, and overall multilingual general capabilities. Optimized for subjective open-ended tasks, delivering responses significantly more aligned with user preferences and more helpful.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":5.200000000000001e-7},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.000000000000001e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"zai/glm-4.5","object":"model","created":1753471347,"owned_by":"zai","model_name":"glm-4.5","context_length":131072,"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":1.1e-7},"list_pricing":{"input_cost_per_token":6e-7,"output_cost_per_token":2.2e-6,"cache_read_input_token_cost":1.1e-7},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"zai/glm-4.5-air","object":"model","created":1753471258,"owned_by":"zai","model_name":"glm-4.5-air","context_length":131072,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.1e-6,"cache_read_input_token_cost":3e-8},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":1.1e-6,"cache_read_input_token_cost":3e-8},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507","object":"model","created":1753449557,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-235B-A22B-Thinking-2507","context_length":262144,"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.3e-7,"output_cost_per_token":2.3e-6,"cache_read_input_token_cost":2e-7},"list_pricing":{"input_cost_per_token":2.3e-7,"output_cost_per_token":2.3e-6,"cache_read_input_token_cost":2e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3-235b-a22b-thinking-2507","object":"model","created":1753449557,"owned_by":"qwen","model_name":"qwen3-235b-a22b-thinking-2507","context_length":131072,"description":"Open-source Qwen3 thinking model; compared to the previous version (Qwen3-235B-A22B) shows major improvements in logical ability, general capabilities, knowledge enhancement, and creativity, suitable for high-difficulty, strong-thinking scenarios.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.495e-7,"output_cost_per_token":1.495e-6},"list_pricing":{"input_cost_per_token":2.3000000000000002e-7,"output_cost_per_token":2.3e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"google/gemini-2.5-flash-lite","object":"model","created":1753200276,"owned_by":"google","model_name":"gemini-2.5-flash-lite","context_length":1048576,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","capabilities":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":1e-7,"cache_read_input_token_cost":1e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":4e-7,"cache_read_input_audio_token_cost":3e-8},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":1e-7,"cache_read_input_token_cost":1e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":4e-7,"cache_read_input_audio_token_cost":3e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-2.5-flash-lite","object":"model","created":1753200276,"owned_by":"vertex","model_name":"gemini-2.5-flash-lite","context_length":1048576,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","source":null,"capabilities":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":1e-7,"cache_read_input_token_cost":1e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":4e-7,"cache_read_input_audio_token_cost":3e-8},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"input_cost_per_audio_token":3e-7,"input_cost_per_image_token":1e-7,"cache_read_input_token_cost":1e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":4e-7,"cache_read_input_audio_token_cost":3e-8},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"xai/grok-4","object":"model","created":1752087689,"owned_by":"xai","model_name":"grok-4","context_length":256000,"description":"Grok 4 is xAI's latest reasoning model with a 256k context window. It supports parallel tool calling, structured outputs, and both image and text inputs. Note that reasoning is not...","source":"https://docs.x.ai/docs/models","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"minimax/minimax-m1","object":"model","created":1750200414,"owned_by":"minimax","model_name":"minimax-m1","context_length":1000000,"description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2.2e-6},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2.2e-6},"discount":null,"regions":[{"code":"ap","name":"Asia"}],"alias_of":null},{"id":"google/gemini-2.5-flash","object":"model","created":1750172488,"owned_by":"google","model_name":"gemini-2.5-flash","context_length":1048576,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","capabilities":{"input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":1e-7},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-2.5-flash","object":"model","created":1750172488,"owned_by":"vertex","model_name":"gemini-2.5-flash","context_length":1048576,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","source":null,"capabilities":{"input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":1e-7},"list_pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":2.5e-6,"input_cost_per_audio_token":1e-6,"input_cost_per_image_token":3e-7,"cache_read_input_token_cost":3e-8,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":8.33333333333333e-8,"output_cost_per_reasoning_token":2.5e-6,"cache_read_input_audio_token_cost":1e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"google/gemini-2.5-pro","object":"model","created":1750169544,"owned_by":"google","model_name":"gemini-2.5-pro","context_length":1048576,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","capabilities":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"input_cost_per_audio_token":1.25e-6,"input_cost_per_image_token":1.25e-6,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.00001,"cache_read_input_audio_token_cost":1.25e-7,"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":0.000015,"input_cost_per_audio_token_above_200k_tokens":2.5e-6,"cache_read_input_token_cost_above_200k_tokens":2.5e-7,"cache_read_input_audio_token_cost_above_200k_tokens":2.5e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"input_cost_per_audio_token":1.25e-6,"input_cost_per_image_token":1.25e-6,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.00001,"cache_read_input_audio_token_cost":1.25e-7,"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":0.000015,"input_cost_per_audio_token_above_200k_tokens":2.5e-6,"cache_read_input_token_cost_above_200k_tokens":2.5e-7,"cache_read_input_audio_token_cost_above_200k_tokens":2.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"vertex/gemini-2.5-pro","object":"model","created":1750169544,"owned_by":"vertex","model_name":"gemini-2.5-pro","context_length":1048576,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","source":null,"capabilities":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"input_cost_per_audio_token":1.25e-6,"input_cost_per_image_token":1.25e-6,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.00001,"cache_read_input_audio_token_cost":1.25e-7,"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":0.000015,"input_cost_per_audio_token_above_200k_tokens":2.5e-6,"cache_read_input_token_cost_above_200k_tokens":2.5e-7,"cache_creation_input_token_cost_above_200k_tokens":2.5e-7,"cache_read_input_audio_token_cost_above_200k_tokens":2.5e-7},"list_pricing":{"input_cost_per_token":1.25e-6,"output_cost_per_token":0.00001,"input_cost_per_audio_token":1.25e-6,"input_cost_per_image_token":1.25e-6,"cache_read_input_token_cost":1.25e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.00001,"cache_read_input_audio_token_cost":1.25e-7,"input_cost_per_token_above_200k_tokens":2.5e-6,"output_cost_per_token_above_200k_tokens":0.000015,"input_cost_per_audio_token_above_200k_tokens":2.5e-6,"cache_read_input_token_cost_above_200k_tokens":2.5e-7,"cache_creation_input_token_cost_above_200k_tokens":2.5e-7,"cache_read_input_audio_token_cost_above_200k_tokens":2.5e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"openai/o3-pro","object":"model","created":1749598352,"owned_by":"openai","model_name":"o3-pro","context_length":200000,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","source":null,"capabilities":{"input_modalities":["text","file","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00002,"output_cost_per_token":0.00008,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.00002,"output_cost_per_token":0.00008,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o3-pro-2025-06-10","object":"model","created":1749598352,"owned_by":"openai","model_name":"o3-pro-2025-06-10","context_length":200000,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","source":null,"capabilities":{"input_modalities":["text","file","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00002,"output_cost_per_token":0.00008,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.00002,"output_cost_per_token":0.00008,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-3","object":"model","created":1749582908,"owned_by":"xai","model_name":"grok-3","context_length":131072,"description":"Grok 3 is the latest model from xAI. It's their flagship model that excels at enterprise use cases like data extraction, coding, and text summarization. Possesses deep domain knowledge in...","source":"https://x.ai/api#pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":7.5e-7},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":7.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-R1-0528","object":"model","created":1748455170,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-R1-0528","context_length":163840,"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.1499999999999997e-6,"cache_read_input_token_cost":3.5e-7},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":2.1499999999999997e-6,"cache_read_input_token_cost":3.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"together_ai/google/gemma-3n-E4B-it","object":"model","created":1747776824,"owned_by":"together_ai","model_name":"google/gemma-3n-E4B-it","context_length":32768,"description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.000000000000001e-8,"output_cost_per_token":1.2000000000000002e-7},"list_pricing":{"input_cost_per_token":6.000000000000001e-8,"output_cost_per_token":1.2000000000000002e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"mistral/mistral-medium-3","object":"model","created":1746627341,"owned_by":"mistral","model_name":"mistral-medium-3","context_length":262144,"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":4e-8},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":2e-6,"cache_read_input_token_cost":4e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/meta-llama/Llama-Guard-4-12B","object":"model","created":1745975193,"owned_by":"deepinfra","model_name":"meta-llama/Llama-Guard-4-12B","context_length":163840,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.8e-7,"output_cost_per_token":1.8e-7},"list_pricing":{"input_cost_per_token":1.8e-7,"output_cost_per_token":1.8e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"together_ai/meta-llama/Llama-Guard-4-12B","object":"model","created":1745975193,"owned_by":"together_ai","model_name":"meta-llama/Llama-Guard-4-12B","context_length":1048576,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","source":null,"capabilities":{"input_modalities":["image","text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2e-7},"list_pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":2e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-30B-A3B","object":"model","created":1745878604,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-30B-A3B","context_length":40960,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.2000000000000002e-7,"output_cost_per_token":5e-7},"list_pricing":{"input_cost_per_token":1.2000000000000002e-7,"output_cost_per_token":5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3-30b-a3b","object":"model","created":1745878604,"owned_by":"qwen","model_name":"qwen3-30b-a3b","context_length":131072,"description":"Achieves effective integration of thinking and non-thinking modes, allowing mode switching during conversations. Its reasoning capability rivals QwQ-32B with a smaller parameter size, and its general capability significantly surpasses Qwen2.5-14B, reaching the SOTA level in the same scale industry.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3000000000000003e-7,"output_cost_per_token":5.200000000000001e-7,"output_cost_per_reasoning_token":1.5599999999999999e-6},"list_pricing":{"input_cost_per_token":2.0000000000000002e-7,"output_cost_per_token":8.000000000000001e-7,"output_cost_per_reasoning_token":2.4e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-8b","object":"model","created":1745876632,"owned_by":"qwen","model_name":"qwen3-8b","context_length":131072,"description":"Achieves effective integration of thinking and non-thinking modes, allowing mode switching during conversations. Its reasoning capability reaches the SOTA level in the same scale industry, and its general capability significantly surpasses Qwen2.5-7B.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.17e-7,"output_cost_per_token":4.55e-7,"output_cost_per_reasoning_token":1.3650000000000003e-6},"list_pricing":{"input_cost_per_token":1.8e-7,"output_cost_per_token":7e-7,"output_cost_per_reasoning_token":2.1000000000000002e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-14B","object":"model","created":1745876478,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-14B","context_length":40960,"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.2000000000000002e-7,"output_cost_per_token":2.4000000000000003e-7},"list_pricing":{"input_cost_per_token":1.2000000000000002e-7,"output_cost_per_token":2.4000000000000003e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"qwen/qwen3-14b","object":"model","created":1745876478,"owned_by":"qwen","model_name":"qwen3-14b","context_length":131072,"description":"Achieves effective integration of thinking and non-thinking modes, allowing mode switching during conversations. Its reasoning capability reaches the SOTA level in the same scale industry, and its general capability significantly surpasses Qwen2.5-14B.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.275e-7,"output_cost_per_token":9.1e-7,"output_cost_per_reasoning_token":2.7300000000000005e-6},"list_pricing":{"input_cost_per_token":3.5e-7,"output_cost_per_token":1.4e-6,"output_cost_per_reasoning_token":4.2000000000000004e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen3-32B","object":"model","created":1745875945,"owned_by":"deepinfra","model_name":"Qwen/Qwen3-32B","context_length":40960,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":2.8e-7},"list_pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":2.8e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/Qwen/Qwen3-32B","object":"model","created":1745875945,"owned_by":"nebius","model_name":"Qwen/Qwen3-32B","context_length":40960,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","source":"https://nebius.com/prices-ai-studio","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7,"cache_read_input_token_cost":1e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7,"cache_read_input_token_cost":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-32b","object":"model","created":1745875945,"owned_by":"qwen","model_name":"qwen3-32b","context_length":131072,"description":"Achieves effective integration of thinking and non-thinking modes, allowing mode switching during conversations. Its reasoning capability significantly surpasses QwQ, and its general capability markedly exceeds Qwen2.5-32B-Instruct, reaching the SOTA level in the same scale industry.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.04e-7,"output_cost_per_token":4.16e-7,"output_cost_per_reasoning_token":4.16e-7},"list_pricing":{"input_cost_per_token":1.6e-7,"output_cost_per_token":6.4e-7,"output_cost_per_reasoning_token":6.4e-7},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"qwen/qwen3-235b-a22b","object":"model","created":1745875757,"owned_by":"qwen","model_name":"qwen3-235b-a22b","context_length":131072,"description":"Achieves effective integration of thinking and non-thinking modes, allowing mode switching during conversations. Its reasoning capability significantly surpasses QwQ, and its general capability markedly exceeds Qwen2.5-72B-Instruct, reaching the SOTA level in the same scale industry.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.55e-7,"output_cost_per_token":1.82e-6,"output_cost_per_reasoning_token":5.460000000000001e-6},"list_pricing":{"input_cost_per_token":7e-7,"output_cost_per_token":2.8e-6,"output_cost_per_reasoning_token":8.400000000000001e-6},"discount":0.35,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/o3","object":"model","created":1744823457,"owned_by":"openai","model_name":"o3","context_length":200000,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o3-2025-04-16","object":"model","created":1744823457,"owned_by":"openai","model_name":"o3-2025-04-16","context_length":200000,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o4-mini","object":"model","created":1744820942,"owned_by":"openai","model_name":"o4-mini","context_length":200000,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.75e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.75e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o4-mini-2025-04-16","object":"model","created":1744820942,"owned_by":"openai","model_name":"o4-mini-2025-04-16","context_length":200000,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.75e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":2.75e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4.1","object":"model","created":1744651385,"owned_by":"openai","model_name":"gpt-4.1","context_length":1047576,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4.1-2025-04-14","object":"model","created":1744651385,"owned_by":"openai","model_name":"gpt-4.1-2025-04-14","context_length":1047576,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4.1-mini","object":"model","created":1744651381,"owned_by":"openai","model_name":"gpt-4.1-mini","context_length":1047576,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":1.6e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":1.6e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4.1-mini-2025-04-14","object":"model","created":1744651381,"owned_by":"openai","model_name":"gpt-4.1-mini-2025-04-14","context_length":1047576,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":1.6e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":4e-7,"output_cost_per_token":1.6e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4.1-nano","object":"model","created":1744651369,"owned_by":"openai","model_name":"gpt-4.1-nano","context_length":1047576,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4.1-nano-2025-04-14","object":"model","created":1744651369,"owned_by":"openai","model_name":"gpt-4.1-nano-2025-04-14","context_length":1047576,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","source":null,"capabilities":{"input_modalities":["image","text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7,"cache_read_input_token_cost":2.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"xai/grok-3-beta","object":"model","created":1744240068,"owned_by":"xai","model_name":"grok-3-beta","context_length":131072,"description":"Grok 3 is the latest model from xAI. It's their flagship model that excels at enterprise use cases like data extraction, coding, and text summarization. Possesses deep domain knowledge in...","source":"https://x.ai/api#pricing","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":7.5e-7},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"cache_read_input_token_cost":7.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o1-pro","object":"model","created":1742423211,"owned_by":"openai","model_name":"o1-pro","context_length":200000,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00015,"output_cost_per_token":0.0006,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.00015,"output_cost_per_token":0.0006,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o1-pro-2025-03-19","object":"model","created":1742423211,"owned_by":"openai","model_name":"o1-pro-2025-03-19","context_length":200000,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00015,"output_cost_per_token":0.0006,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.00015,"output_cost_per_token":0.0006,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"cloudflare/@cf/mistralai/mistral-small-3.1-24b-instruct","object":"model","created":1742238937,"owned_by":"cloudflare","model_name":"@cf/mistralai/mistral-small-3.1-24b-instruct","context_length":128000,"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.5099999999999995e-7,"output_cost_per_token":5.550000000000001e-7},"list_pricing":{"input_cost_per_token":3.5099999999999995e-7,"output_cost_per_token":5.550000000000001e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/google/gemma-3-4b-it","object":"model","created":1741905510,"owned_by":"deepinfra","model_name":"google/gemma-3-4b-it","context_length":131072,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":1.0000000000000001e-7},"list_pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":1.0000000000000001e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/google/gemma-3-12b-it","object":"model","created":1741902625,"owned_by":"deepinfra","model_name":"google/gemma-3-12b-it","context_length":131072,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":1.5e-7},"list_pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":1.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"cohere/command-a-03-2025","object":"model","created":1741894342,"owned_by":"cohere","model_name":"command-a-03-2025","context_length":288000,"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4o-mini-search-preview","object":"model","created":1741818122,"owned_by":"openai","model_name":"gpt-4o-mini-search-preview","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.025,"search_context_size_high":0.03,"search_context_size_medium":0.0275}},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.025,"search_context_size_high":0.03,"search_context_size_medium":0.0275}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4o-search-preview","object":"model","created":1741817949,"owned_by":"openai","model_name":"gpt-4o-search-preview","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6,"search_context_cost_per_query":{"search_context_size_low":0.03,"search_context_size_high":0.05,"search_context_size_medium":0.035}},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6,"search_context_cost_per_query":{"search_context_size_low":0.03,"search_context_size_high":0.05,"search_context_size_medium":0.035}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/google/gemma-3-27b-it","object":"model","created":1741756359,"owned_by":"deepinfra","model_name":"google/gemma-3-27b-it","context_length":131072,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":1.6e-7},"list_pricing":{"input_cost_per_token":8e-8,"output_cost_per_token":1.6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/google/gemma-3-27b-it","object":"model","created":1741756359,"owned_by":"nebius","model_name":"google/gemma-3-27b-it","context_length":110000,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","source":"https://nebius.com/prices-ai-studio","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7,"cache_read_input_token_cost":1e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3e-7,"cache_read_input_token_cost":1e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"perplexityai/sonar-reasoning-pro","object":"model","created":1741313308,"owned_by":"perplexityai","model_name":"sonar-reasoning-pro","context_length":128000,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"perplexityai/sonar-pro","object":"model","created":1741312423,"owned_by":"perplexityai","model_name":"sonar-pro","context_length":200000,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":0.000015,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"perplexityai/sonar-deep-research","object":"model","created":1741311246,"owned_by":"perplexityai","model_name":"sonar-deep-research","context_length":128000,"description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"citation_cost_per_token":2e-6,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"output_cost_per_reasoning_token":3e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":8e-6,"citation_cost_per_token":2e-6,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"output_cost_per_reasoning_token":3e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/Qwen/Qwen2.5-VL-72B-Instruct","object":"model","created":1738410311,"owned_by":"nebius","model_name":"Qwen/Qwen2.5-VL-72B-Instruct","context_length":32000,"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","source":"https://nebius.com/prices-ai-studio","capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":7.5e-7,"cache_read_input_token_cost":2.5e-7},"list_pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":7.5e-7,"cache_read_input_token_cost":2.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/o3-mini","object":"model","created":1738351721,"owned_by":"openai","model_name":"o3-mini","context_length":200000,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","source":null,"capabilities":{"input_modalities":["text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o3-mini-2025-01-31","object":"model","created":1738351721,"owned_by":"openai","model_name":"o3-mini-2025-01-31","context_length":200000,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","source":null,"capabilities":{"input_modalities":["text","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":1.1e-6,"output_cost_per_token":4.4e-6,"cache_read_input_token_cost":5.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/mistralai/Mistral-Small-24B-Instruct-2501","object":"model","created":1738255409,"owned_by":"deepinfra","model_name":"mistralai/Mistral-Small-24B-Instruct-2501","context_length":32768,"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":8e-8},"list_pricing":{"input_cost_per_token":5.0000000000000004e-8,"output_cost_per_token":8e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"perplexityai/sonar","object":"model","created":1738013808,"owned_by":"perplexityai","model_name":"sonar","context_length":127072,"description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":1e-6,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":1e-6,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B","object":"model","created":1737663169,"owned_by":"deepinfra","model_name":"deepseek-ai/DeepSeek-R1-Distill-Llama-70B","context_length":131072,"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":7e-7,"output_cost_per_token":8e-7},"list_pricing":{"input_cost_per_token":7e-7,"output_cost_per_token":8e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/microsoft/phi-4","object":"model","created":1736489872,"owned_by":"deepinfra","model_name":"microsoft/phi-4","context_length":16384,"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":1.4e-7},"list_pricing":{"input_cost_per_token":7e-8,"output_cost_per_token":1.4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"scaleway/llama-3.3-70b-instruct","object":"model","created":1736258559,"owned_by":"scaleway","model_name":"llama-3.3-70b-instruct","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.051290108177752e-6,"output_cost_per_token":1.051290108177752e-6},"list_pricing":{"input_cost_per_token":1.051290108177752e-6,"output_cost_per_token":1.051290108177752e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/o1","object":"model","created":1734459999,"owned_by":"openai","model_name":"o1","context_length":200000,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.00006,"cache_read_input_token_cost":7.5e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.00006,"cache_read_input_token_cost":7.5e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/o1-2024-12-17","object":"model","created":1734459999,"owned_by":"openai","model_name":"o1-2024-12-17","context_length":200000,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.00006,"cache_read_input_token_cost":7.5e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":0.000015,"output_cost_per_token":0.00006,"cache_read_input_token_cost":7.5e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"cohere/command-r7b-12-2024","object":"model","created":1734158152,"owned_by":"cohere","model_name":"command-r7b-12-2024","context_length":132000,"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","source":"https://docs.cohere.com/v2/docs/command-r7b","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.75e-8,"output_cost_per_token":1.5e-7},"list_pricing":{"input_cost_per_token":3.75e-8,"output_cost_per_token":1.5e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/meta-llama/Llama-3.3-70B-Instruct","object":"model","created":1733506137,"owned_by":"deepinfra","model_name":"meta-llama/Llama-3.3-70B-Instruct","context_length":131072,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3.2e-7},"list_pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":3.2e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"nebius/meta-llama/Llama-3.3-70B-Instruct","object":"model","created":1733506137,"owned_by":"nebius","model_name":"meta-llama/Llama-3.3-70B-Instruct","context_length":131072,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","source":"https://nebius.com/prices-ai-studio","capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.3e-7,"output_cost_per_token":4e-7,"cache_read_input_token_cost":1.3e-7},"list_pricing":{"input_cost_per_token":1.3e-7,"output_cost_per_token":4e-7,"cache_read_input_token_cost":1.3e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/amazon.nova-lite-v1:0","object":"model","created":1733437363,"owned_by":"amazon","model_name":"amazon.nova-lite-v1:0","context_length":300000,"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/amazon.nova-lite-v1:0@us","object":"model","created":1733437363,"owned_by":"amazon","model_name":"amazon.nova-lite-v1:0","context_length":300000,"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7},"list_pricing":{"input_cost_per_token":6e-8,"output_cost_per_token":2.4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/amazon.nova-micro-v1:0","object":"model","created":1733437237,"owned_by":"amazon","model_name":"amazon.nova-micro-v1:0","context_length":128000,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.5e-8,"output_cost_per_token":1.4e-7},"list_pricing":{"input_cost_per_token":3.5e-8,"output_cost_per_token":1.4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/amazon.nova-micro-v1:0@us","object":"model","created":1733437237,"owned_by":"amazon","model_name":"amazon.nova-micro-v1:0","context_length":128000,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.5e-8,"output_cost_per_token":1.4e-7},"list_pricing":{"input_cost_per_token":3.5e-8,"output_cost_per_token":1.4e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"amazon/amazon.nova-pro-v1:0","object":"model","created":1733436303,"owned_by":"amazon","model_name":"amazon.nova-pro-v1:0","context_length":300000,"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8e-7,"output_cost_per_token":3.2e-6},"list_pricing":{"input_cost_per_token":8e-7,"output_cost_per_token":3.2e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"amazon/amazon.nova-pro-v1:0@us","object":"model","created":1733436303,"owned_by":"amazon","model_name":"amazon.nova-pro-v1:0","context_length":300000,"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":8e-7,"output_cost_per_token":3.2e-6},"list_pricing":{"input_cost_per_token":8e-7,"output_cost_per_token":3.2e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4o","object":"model","created":1732127594,"owned_by":"openai","model_name":"gpt-4o","context_length":128000,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4o-2024-08-06","object":"model","created":1732127594,"owned_by":"openai","model_name":"gpt-4o-2024-08-06","context_length":128000,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4o-2024-11-20","object":"model","created":1732127594,"owned_by":"openai","model_name":"gpt-4o-2024-11-20","context_length":128000,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"mistral/mistral-large-2407","object":"model","created":1731978415,"owned_by":"mistral","model_name":"mistral-large-2407","context_length":131072,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","source":null,"capabilities":{"input_modalities":["text","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct","object":"model","created":1731368400,"owned_by":"cloudflare","model_name":"@cf/qwen/qwen2.5-coder-32b-instruct","context_length":32768,"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":1e-6},"list_pricing":{"input_cost_per_token":6.6e-7,"output_cost_per_token":1e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"scaleway/gemma-3-27b-it","object":"model","created":1730385501,"owned_by":"scaleway","model_name":"gemma-3-27b-it","context_length":40000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2.8712497171819026e-7,"output_cost_per_token":5.742499434363805e-7},"list_pricing":{"input_cost_per_token":2.8712497171819026e-7,"output_cost_per_token":5.742499434363805e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"scaleway/pixtral-12b-2409","object":"model","created":1730385501,"owned_by":"scaleway","model_name":"pixtral-12b-2409","context_length":128000,"description":null,"source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.3362002403950049e-7,"output_cost_per_token":2.3362002403950049e-7},"list_pricing":{"input_cost_per_token":2.3362002403950049e-7,"output_cost_per_token":2.3362002403950049e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"cloudflare/@cf/meta/llama-3.2-1b-instruct","object":"model","created":1727222400,"owned_by":"cloudflare","model_name":"@cf/meta/llama-3.2-1b-instruct","context_length":60000,"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.7e-8,"output_cost_per_token":2.01e-7},"list_pricing":{"input_cost_per_token":2.7e-8,"output_cost_per_token":2.01e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"cloudflare/@cf/meta/llama-3.2-3b-instruct","object":"model","created":1727222400,"owned_by":"cloudflare","model_name":"@cf/meta/llama-3.2-3b-instruct","context_length":80000,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5.09e-8,"output_cost_per_token":3.35e-7},"list_pricing":{"input_cost_per_token":5.09e-8,"output_cost_per_token":3.35e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct","object":"model","created":1727222400,"owned_by":"deepinfra","model_name":"meta-llama/Llama-3.2-11B-Vision-Instruct","context_length":131072,"description":"Llama 3.2 11B Vision is a multimodal model with 11 billion parameters, designed to handle tasks combining visual and textual data. It excels in tasks such as image captioning and...","source":null,"capabilities":{"input_modalities":["text","image"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3.45e-7,"output_cost_per_token":3.45e-7},"list_pricing":{"input_cost_per_token":3.45e-7,"output_cost_per_token":3.45e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Qwen/Qwen2.5-72B-Instruct","object":"model","created":1726704000,"owned_by":"deepinfra","model_name":"Qwen/Qwen2.5-72B-Instruct","context_length":32768,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":4.0000000000000003e-7},"list_pricing":{"input_cost_per_token":3.6e-7,"output_cost_per_token":4.0000000000000003e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"cohere/command-r-08-2024","object":"model","created":1724976000,"owned_by":"cohere","model_name":"command-r-08-2024","context_length":128000,"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"cohere/command-r-plus-08-2024","object":"model","created":1724976000,"owned_by":"cohere","model_name":"command-r-plus-08-2024","context_length":128000,"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/NousResearch/Hermes-3-Llama-3.1-70B","object":"model","created":1723939200,"owned_by":"deepinfra","model_name":"NousResearch/Hermes-3-Llama-3.1-70B","context_length":131072,"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7e-7,"output_cost_per_token":7e-7},"list_pricing":{"input_cost_per_token":7e-7,"output_cost_per_token":7e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/NousResearch/Hermes-3-Llama-3.1-405B","object":"model","created":1723766400,"owned_by":"deepinfra","model_name":"NousResearch/Hermes-3-Llama-3.1-405B","context_length":131072,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":1e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"cloudflare/@cf/meta/llama-3.1-8b-instruct-fp8","object":"model","created":1721692800,"owned_by":"cloudflare","model_name":"@cf/meta/llama-3.1-8b-instruct-fp8","context_length":32000,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5199999999999998e-7,"output_cost_per_token":2.8699999999999996e-7},"list_pricing":{"input_cost_per_token":1.5199999999999998e-7,"output_cost_per_token":2.8699999999999996e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":null},{"id":"microsoft/gpt-4o-mini","object":"model","created":1721260800,"owned_by":"microsoft","model_name":"gpt-4o-mini","context_length":128000,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":false,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":false,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":7.5e-8},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-4o-mini","object":"model","created":1721260800,"owned_by":"openai","model_name":"gpt-4o-mini","context_length":128000,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":7.5e-8},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":7.5e-8},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4o-mini-2024-07-18","object":"model","created":1721260800,"owned_by":"openai","model_name":"gpt-4o-mini-2024-07-18","context_length":128000,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.025,"search_context_size_high":0.03,"search_context_size_medium":0.0275}},"list_pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.025,"search_context_size_high":0.03,"search_context_size_medium":0.0275}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"microsoft/gpt-4o","object":"model","created":1715558400,"owned_by":"microsoft","model_name":"gpt-4o","context_length":128000,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6},"list_pricing":{"input_cost_per_token":2.5e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":1.25e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-4-turbo","object":"model","created":1712620800,"owned_by":"openai","model_name":"gpt-4-turbo","context_length":128000,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00003},"list_pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00003},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4-turbo-2024-04-09","object":"model","created":1712620800,"owned_by":"openai","model_name":"gpt-4-turbo-2024-04-09","context_length":128000,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00003},"list_pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00003},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"mistral/mistral-large-latest","object":"model","created":1708905600,"owned_by":"mistral","model_name":"mistral-large-latest","context_length":262144,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","source":"https://docs.mistral.ai/models/mistral-large-3-25-12","capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":2e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"microsoft/gpt-35-turbo-16k","object":"model","created":1693180800,"owned_by":"microsoft","model_name":"gpt-35-turbo-16k","context_length":16385,"description":"","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":4e-6},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":4e-6},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-3.5-turbo-16k","object":"model","created":1693180800,"owned_by":"openai","model_name":"gpt-3.5-turbo-16k","context_length":16385,"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":4e-6},"list_pricing":{"input_cost_per_token":3e-6,"output_cost_per_token":4e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"deepinfra/Gryphe/MythoMax-L2-13b","object":"model","created":1688256000,"owned_by":"deepinfra","model_name":"Gryphe/MythoMax-L2-13b","context_length":4096,"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":4.0000000000000003e-7},"list_pricing":{"input_cost_per_token":4.0000000000000003e-7,"output_cost_per_token":4.0000000000000003e-7},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"microsoft/gpt-4","object":"model","created":1685232000,"owned_by":"microsoft","model_name":"gpt-4","context_length":8191,"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":false,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00006},"list_pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00006},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":null},{"id":"openai/gpt-3.5-turbo","object":"model","created":1685232000,"owned_by":"openai","model_name":"gpt-3.5-turbo","context_length":16385,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"list_pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":1.5e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"openai/gpt-4","object":"model","created":1685232000,"owned_by":"openai","model_name":"gpt-4","context_length":8191,"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","source":null,"capabilities":{"input_modalities":["text"],"output_modalities":["text"],"supports_reasoning":false,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":false,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00006},"list_pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00006},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":null},{"id":"google/gemini-flash-latest","object":"model","created":1786640581,"owned_by":"google","model_name":"gemini-3.7-flash","context_length":1048576,"description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.028,"search_context_size_high":0.028,"search_context_size_medium":0.028},"cache_creation_input_token_cost":8.33333333333334e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.028,"search_context_size_high":0.028,"search_context_size_medium":0.028},"cache_creation_input_token_cost":8.33333333333334e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":"google/gemini-3.7-flash"},{"id":"vertex/gemini-flash-latest","object":"model","created":1786640581,"owned_by":"vertex","model_name":"gemini-3.7-flash","context_length":1048576,"description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","source":null,"capabilities":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true},"pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.056,"search_context_size_high":0.056,"search_context_size_medium":0.056},"cache_creation_input_token_cost":8.33333333333332e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"list_pricing":{"input_cost_per_token":1.5e-6,"output_cost_per_token":7.5e-6,"input_cost_per_audio_token":1.5e-6,"input_cost_per_image_token":1.5e-6,"cache_read_input_token_cost":1.5e-7,"search_context_cost_per_query":{"search_context_size_low":0.056,"search_context_size_high":0.056,"search_context_size_medium":0.056},"cache_creation_input_token_cost":8.33333333333332e-8,"output_cost_per_reasoning_token":7.5e-6,"cache_read_input_audio_token_cost":1.5e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"},{"code":"global","name":"Global"},{"code":"us","name":"United States"}],"alias_of":"vertex/gemini-3.7-flash"},{"id":"xai/grok-latest","object":"model","created":1786548957,"owned_by":"xai","model_name":"grok-4.6","context_length":500000,"description":"Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000012,"cache_read_input_token_cost_above_200k_tokens":1e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":6e-6,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.005,"search_context_size_high":0.005,"search_context_size_medium":0.005},"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000012,"cache_read_input_token_cost_above_200k_tokens":1e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":"xai/grok-4.6"},{"id":"anthropic/claude-opus-latest","object":"model","created":1784912544,"owned_by":"anthropic","model_name":"claude-opus-5","context_length":1000000,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.000025,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":6.25e-6,"cache_creation_input_token_cost_above_1hr":0.00001},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":"anthropic/claude-opus-5"},{"id":"openai/gpt-latest","object":"model","created":1783590850,"owned_by":"openai","model_name":"gpt-5.6-sol","context_length":1050000,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.02,"search_context_size_high":0.02,"search_context_size_medium":0.02},"cache_creation_input_token_cost":6.25e-6,"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1e-6,"cache_creation_input_token_cost_above_272k_tokens":0.0000125},"list_pricing":{"input_cost_per_token":5e-6,"output_cost_per_token":0.00003,"cache_read_input_token_cost":5e-7,"search_context_cost_per_query":{"search_context_size_low":0.02,"search_context_size_high":0.02,"search_context_size_medium":0.02},"cache_creation_input_token_cost":6.25e-6,"input_cost_per_token_above_272k_tokens":0.00001,"output_cost_per_token_above_272k_tokens":0.000045,"cache_read_input_token_cost_above_272k_tokens":1e-6,"cache_creation_input_token_cost_above_272k_tokens":0.0000125},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":"openai/gpt-5.6-sol"},{"id":"anthropic/claude-sonnet-latest","object":"model","created":1782843083,"owned_by":"anthropic","model_name":"claude-sonnet-5","context_length":1000000,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-6,"cache_creation_input_token_cost_above_1hr":4e-6},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.00001,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":2.5e-6,"cache_creation_input_token_cost_above_1hr":4e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":"anthropic/claude-sonnet-5"},{"id":"anthropic/claude-fable-latest","object":"model","created":1781007515,"owned_by":"anthropic","model_name":"claude-fable-5","context_length":1000000,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","source":null,"capabilities":{"input_modalities":["text","image","file"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":false,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005,"cache_read_input_token_cost":1e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.0000125,"cache_creation_input_token_cost_above_1hr":0.00002},"list_pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005,"cache_read_input_token_cost":1e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":0.0000125,"cache_creation_input_token_cost_above_1hr":0.00002},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":"anthropic/claude-fable-5"},{"id":"openai/gpt-pro-latest","object":"model","created":1777051896,"owned_by":"openai","model_name":"gpt-5.5-pro","context_length":1050000,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"list_pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00018,"cache_read_input_token_cost":3e-6,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"input_cost_per_token_above_272k_tokens":0.00006,"output_cost_per_token_above_272k_tokens":0.00027,"cache_read_input_token_cost_above_272k_tokens":6e-6},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":"openai/gpt-5.5-pro"},{"id":"openai/gpt-mini-latest","object":"model","created":1773748178,"owned_by":"openai","model_name":"gpt-5.4-mini","context_length":400000,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":true,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"list_pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":4.5e-6,"cache_read_input_token_cost":7.5e-8,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01}},"discount":null,"regions":[{"code":"us","name":"United States"}],"alias_of":"openai/gpt-5.4-mini"},{"id":"google/gemini-pro-latest","object":"model","created":1771509627,"owned_by":"google","model_name":"gemini-3.1-pro-preview","context_length":1048576,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models","capabilities":{"input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000018,"input_cost_per_audio_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7,"cache_read_input_audio_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000018,"input_cost_per_audio_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7,"cache_read_input_audio_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"eu","name":"Europe"}],"alias_of":"google/gemini-3.1-pro-preview"},{"id":"vertex/gemini-pro-latest","object":"model","created":1771509627,"owned_by":"vertex","model_name":"gemini-3.1-pro-preview","context_length":1048576,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","source":null,"capabilities":{"input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":true,"supports_tool_choice":true,"supports_computer_use":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":false,"supports_embedding_image_input":false,"supports_parallel_function_calling":false,"reasoning":{"mandatory":true,"default_effort":null,"supported_efforts":[]}},"pricing":{"input_cost_per_token":2e-6,"output_cost_per_image":0.00012,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000018,"input_cost_per_audio_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7,"cache_creation_input_token_cost_above_200k_tokens":2.5e-7,"cache_read_input_audio_token_cost_above_200k_tokens":4e-7},"list_pricing":{"input_cost_per_token":2e-6,"output_cost_per_image":0.00012,"output_cost_per_token":0.000012,"input_cost_per_audio_token":2e-6,"input_cost_per_image_token":2e-6,"cache_read_input_token_cost":2e-7,"search_context_cost_per_query":{"search_context_size_low":0.014,"search_context_size_high":0.014,"search_context_size_medium":0.014},"cache_creation_input_token_cost":3.75e-7,"output_cost_per_reasoning_token":0.000012,"cache_read_input_audio_token_cost":2e-7,"input_cost_per_token_above_200k_tokens":4e-6,"output_cost_per_token_above_200k_tokens":0.000018,"input_cost_per_audio_token_above_200k_tokens":4e-6,"cache_read_input_token_cost_above_200k_tokens":4e-7,"cache_creation_input_token_cost_above_200k_tokens":2.5e-7,"cache_read_input_audio_token_cost_above_200k_tokens":4e-7},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":"vertex/gemini-3.1-pro-preview"},{"id":"anthropic/claude-haiku-latest","object":"model","created":1760547638,"owned_by":"anthropic","model_name":"claude-haiku-4-5","context_length":200000,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","source":null,"capabilities":{"input_modalities":["file","image","text"],"output_modalities":["text"],"supports_reasoning":true,"supports_web_search":false,"supports_tool_choice":true,"supports_computer_use":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":false,"supports_function_calling":true,"supports_native_streaming":true,"supports_assistant_prefill":true,"supports_embedding_image_input":false,"supports_parallel_function_calling":false},"pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":1.25e-6,"cache_creation_input_token_cost_above_1hr":2e-6},"list_pricing":{"input_cost_per_token":1e-6,"output_cost_per_token":5e-6,"cache_read_input_token_cost":1e-7,"search_context_cost_per_query":{"search_context_size_low":0.01,"search_context_size_high":0.01,"search_context_size_medium":0.01},"cache_creation_input_token_cost":1.25e-6,"cache_creation_input_token_cost_above_1hr":2e-6},"discount":null,"regions":[{"code":"global","name":"Global"}],"alias_of":"anthropic/claude-haiku-4-5"}]}