{"object":"list","data":[{"id":"anthropic/claude-fable-5","object":"model","created":1785525829,"owned_by":"anthropic","name":"Claude Fable 5","pricing":{"input":10.0,"output":50.0,"prompt":"0.00001","completion":"0.00005","image":"0","request":"0","input_cache_read":"0.000001"},"context_length":1000000,"max_output_length":128000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["max_tokens","stop"],"supported_features":["tools","structured_outputs","reasoning"],"description":"Anthropic's most capable widely released model for long-running agents, advanced reasoning, coding, and demanding knowledge work.","top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false}},{"id":"anthropic/claude-fable-5-1","object":"model","created":1788290016,"owned_by":"anthropic","name":"Claude Fable 5.1","pricing":{"input":10.0,"output":50.0,"prompt":"0.00001","completion":"0.00005","image":"0","request":"0","input_cache_read":"0.00000025"},"context_length":1000000,"max_output_length":128000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["max_tokens","stop"],"supported_features":["tools","structured_outputs","reasoning"],"description":"Anthropic's most capable model for real-world agents and coding.","top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false}},{"id":"anthropic/claude-haiku-4-5","object":"model","created":1778671374,"owned_by":"anthropic","name":"Claude Haiku 4.5","pricing":{"input":1.0,"output":5.0,"prompt":"0.000001","completion":"0.000005","image":"0","request":"0","input_cache_read":"0.0000001"},"context_length":200000,"max_output_length":64000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop"],"supported_features":["tools","structured_outputs","reasoning"],"is_ready":false,"description":"Anthropic's fastest model with near-frontier intelligence for high-throughput, cost-sensitive workloads.","top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":false}},{"id":"anthropic/claude-opus-4-6","object":"model","created":1770414679,"owned_by":"anthropic","name":"Claude Opus 4.6","pricing":{"input":5.0,"output":25.0,"prompt":"0.000005","completion":"0.000025","image":"0","request":"0","input_cache_read":"0.0000005"},"context_length":200000,"max_output_length":32768,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop"],"supported_features":["tools","structured_outputs","reasoning"],"is_ready":false,"description":"Anthropic's most intelligent model for building agents and coding","top_provider":{"context_length":200000,"max_completion_tokens":32768,"is_moderated":false}},{"id":"anthropic/claude-opus-4-7","object":"model","created":1778671374,"owned_by":"anthropic","name":"Claude Opus 4.7","pricing":{"input":5.0,"output":25.0,"prompt":"0.000005","completion":"0.000025","image":"0","request":"0","input_cache_read":"0.0000005"},"context_length":1000000,"max_output_length":32768,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop"],"supported_features":["tools","structured_outputs","reasoning"],"is_ready":false,"description":"Anthropic's most capable model. Next-generation built for long-running agents and complex coding tasks. 1M token context window with 128K max output.","top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false}},{"id":"anthropic/claude-opus-4-8","object":"model","created":1785525405,"owned_by":"anthropic","name":"Claude Opus 4.8","pricing":{"input":5.0,"output":25.0,"prompt":"0.000005","completion":"0.000025","image":"0","request":"0","input_cache_read":"0.0000005"},"context_length":1000000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":[],"supported_features":[],"description":"Anthropic's most capable model. Next-generation built for long-running agents and complex coding tasks. 1M token context window with 128K max output.","top_provider":{"context_length":1000000,"is_moderated":false}},{"id":"anthropic/claude-opus-5","object":"model","created":1787238785,"owned_by":"anthropic","name":"Claude Opus 5","pricing":{"input":5.0,"output":25.0,"prompt":"0.000005","completion":"0.000025","image":"0","request":"0","input_cache_read":"0.0000005"},"context_length":1000000,"max_output_length":128000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["max_tokens","stop"],"supported_features":["tools","structured_outputs","reasoning"],"description":"Anthropic model for complex agentic coding and enterprise work, with adaptive reasoning and a 1M-token context window.","top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false}},{"id":"anthropic/claude-sonnet-4-5","object":"model","created":1769639067,"owned_by":"anthropic","name":"Claude Sonnet 4.5","pricing":{"input":3.0,"output":15.0,"prompt":"0.000003","completion":"0.000015","image":"0","request":"0","input_cache_read":"0.0000003"},"context_length":200000,"max_output_length":16384,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop"],"supported_features":["tools","structured_outputs"],"is_ready":false,"description":"Anthropic's Claude Sonnet 4.5 - a powerful, efficient model balancing intelligence and speed. Excels at complex reasoning, coding, and creative tasks with 200K context window. Anonymized, not TEE-protected.","top_provider":{"context_length":200000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"anthropic/claude-sonnet-4-6","object":"model","created":1778671374,"owned_by":"anthropic","name":"Claude Sonnet 4.6","pricing":{"input":3.0,"output":15.0,"prompt":"0.000003","completion":"0.000015","image":"0","request":"0","input_cache_read":"0.0000003"},"context_length":1000000,"max_output_length":16384,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop"],"supported_features":["tools","structured_outputs","reasoning"],"is_ready":false,"description":"Anthropic's best balance of speed and intelligence. Extended thinking support with 1M token context window and 64K max output. Ideal for most production workloads.","top_provider":{"context_length":1000000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"anthropic/claude-sonnet-5","object":"model","created":1785526312,"owned_by":"anthropic","name":"Claude Sonnet 5","pricing":{"input":2.0,"output":10.0,"prompt":"0.000002","completion":"0.00001","image":"0","request":"0","input_cache_read":"0.0000002"},"context_length":1000000,"max_output_length":128000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["max_tokens","stop"],"supported_features":["tools","structured_outputs","reasoning"],"description":"Anthropic's best combination of speed and intelligence for coding, agents, and professional workloads, with adaptive reasoning.","top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false}},{"id":"black-forest-labs/FLUX.2-klein-4B","object":"model","created":1769640850,"owned_by":"nearai","name":"FLUX.2-klein-4B","hugging_face_id":"black-forest-labs/FLUX.2-klein-4B","quantization":"bf16","pricing":{"input":1.0,"output":1.0,"prompt":"0.000001","completion":"0.000001","image":"0.012","request":"0"},"context_length":128000,"max_output_length":1,"architecture":{"inputModalities":["text"],"outputModalities":["image"]},"input_modalities":["text"],"output_modalities":["image"],"supported_sampling_parameters":["seed"],"supported_features":[],"description":"The FLUX.2 [klein] model family are our fastest image models to date. FLUX.2 [klein] unifies generation and editing in a single compact architecture, delivering state-of-the-art quality with end-to-end inference in as low as under a second. Built for applications that require real-time image generation without sacrificing quality.","top_provider":{"context_length":128000,"max_completion_tokens":1,"is_moderated":false},"datacenters":[{"country_code":"US"}]},{"id":"deepseek-ai/DeepSeek-V4-Flash","object":"model","created":1780561470,"owned_by":"nearai","name":"DeepSeek V4 Flash","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash","quantization":"fp8","pricing":{"input":0.17,"output":0.35000000000000003,"prompt":"0.00000017","completion":"0.00000035","image":"0","request":"0","input_cache_read":"0.000000035"},"context_length":1048576,"max_output_length":8192,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","min_p","frequency_penalty","presence_penalty","repetition_penalty","max_tokens","stop","seed","logit_bias"],"supported_features":["tools","reasoning","json_mode","structured_outputs"],"is_ready":true,"deprecation_date":"2026-09-17T13:00:00Z","description":"DeepSeek V4 Flash — large mixture-of-experts language model from DeepSeek, FP8-quantized. Served on H200 with TP=4 and EAGLE speculative decoding in a TDX-confidential inference CVM.","top_provider":{"context_length":1048576,"max_completion_tokens":8192,"is_moderated":false},"datacenters":[{"country_code":"US"}],"openrouter":{"slug":"deepseek/deepseek-v4-flash"}},{"id":"deepseek/deepseek-v3.2","object":"model","created":1781515743,"owned_by":"attested 3p","name":"deepseek-v3.2","pricing":{"input":1.1,"output":1.1,"prompt":"0.0000011","completion":"0.0000011","image":"0","request":"0","input_cache_read":"0.00000055"},"context_length":128000,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","stop","seed","max_tokens"],"supported_features":["tools","json_mode"],"is_ready":false,"description":"Attested model served via Chutes TEE (verified end-to-end by NEAR AI).","top_provider":{"context_length":128000,"is_moderated":false}},{"id":"google/gemini-2.5-flash","object":"model","created":1778671374,"owned_by":"google","name":"Gemini 2.5 Flash","pricing":{"input":0.3,"output":2.5,"prompt":"0.0000003","completion":"0.0000025","image":"0","request":"0"},"context_length":1000000,"max_output_length":8192,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop"],"supported_features":["tools","structured_outputs","json_mode"],"is_ready":false,"description":"Google's fast hybrid reasoning model with 1M token context window. Optimized for speed and cost while maintaining strong performance across tasks.","top_provider":{"context_length":1000000,"max_completion_tokens":8192,"is_moderated":false}},{"id":"google/gemini-2.5-flash-lite","object":"model","created":1778671374,"owned_by":"google","name":"Gemini 2.5 Flash Lite","pricing":{"input":0.1,"output":0.4,"prompt":"0.0000001","completion":"0.0000004","image":"0","request":"0"},"context_length":1048576,"max_output_length":8192,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop"],"supported_features":["tools","structured_outputs","json_mode"],"is_ready":false,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","top_provider":{"context_length":1048576,"max_completion_tokens":8192,"is_moderated":false}},{"id":"google/gemini-2.5-pro","object":"model","created":1778671374,"owned_by":"google","name":"Gemini 2.5 Pro","pricing":{"input":1.25,"output":10.0,"prompt":"0.00000125","completion":"0.00001","image":"0","request":"0"},"context_length":1000000,"max_output_length":16384,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop"],"supported_features":["tools","structured_outputs","json_mode","reasoning"],"is_ready":false,"description":"Google's strongest reasoning model. Excels at coding, math, and complex analysis with 1M token context window. Supports text and image input.","top_provider":{"context_length":1000000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"google/gemini-3.1-flash-lite","object":"model","created":1778671374,"owned_by":"google","name":"Gemini 3.1 Flash Lite","pricing":{"input":0.25,"output":1.5,"prompt":"0.00000025","completion":"0.0000015","image":"0","request":"0"},"context_length":1048576,"max_output_length":8192,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop"],"supported_features":["tools","structured_outputs","json_mode"],"is_ready":false,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","top_provider":{"context_length":1048576,"max_completion_tokens":8192,"is_moderated":false}},{"id":"google/gemini-3.5-flash","object":"model","created":1779214705,"owned_by":"google","name":"Gemini 3.5 Flash","pricing":{"input":1.5,"output":9.0,"prompt":"0.0000015","completion":"0.000009","image":"0","request":"0","input_cache_read":"0.00000015"},"context_length":1000000,"max_output_length":8192,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop"],"supported_features":["tools","structured_outputs","json_mode"],"is_ready":false,"description":"Google's high-efficiency multimodal model with 1M token context. Strong agentic and coding performance, rivaling larger flagship models on many tasks.","top_provider":{"context_length":1000000,"max_completion_tokens":8192,"is_moderated":false}},{"id":"google/gemini-3.8-flash","object":"model","created":1788461610,"owned_by":"google","name":"Gemini 3.8 Flash","pricing":{"input":0.75,"output":3.75,"prompt":"0.00000075","completion":"0.00000375","image":"0","request":"0","input_cache_read":"0.000000075"},"context_length":1000000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":[],"supported_features":[],"description":"Google's stable, high-efficiency Gemini 3.8 Flash model for agentic coding, multimodal understanding, and high-throughput workloads, with a 1M-token context window.","top_provider":{"context_length":1000000,"is_moderated":false}},{"id":"moonshotai/kimi-k2.6","object":"model","created":1779449813,"owned_by":"attested 3p","name":"Kimi K2.6","hugging_face_id":"moonshotai/Kimi-K2.6","pricing":{"input":0.81,"output":3.85,"prompt":"0.00000081","completion":"0.00000385","image":"0","request":"0","input_cache_read":"0.00000041"},"context_length":262144,"max_output_length":8192,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs"],"is_ready":true,"description":"Moonshot AI's frontier MoE model with 256K context window. Excels at complex reasoning, math, coding, and multilingual tasks with native vision support.","top_provider":{"context_length":262144,"max_completion_tokens":8192,"is_moderated":false}},{"id":"moonshotai/kimi-k3","object":"model","created":1787666025,"owned_by":"attested 3p","name":"Kimi K3","quantization":"fp4","pricing":{"input":3.3000000000000003,"output":16.5,"prompt":"0.0000033","completion":"0.0000165","image":"0","request":"0","input_cache_read":"0.00000033"},"context_length":1048576,"max_output_length":65535,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","repetition_penalty","frequency_penalty","presence_penalty","stop","seed","max_tokens"],"supported_features":["tools","json_mode","structured_outputs","reasoning"],"is_ready":true,"description":"Moonshot AI native multimodal agentic model with a 1M-token context window, built for long-horizon coding, knowledge work, and reasoning.","top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false}},{"id":"openai/gpt-4.1","object":"model","created":1778671374,"owned_by":"openai","name":"OpenAI GPT-4.1","pricing":{"input":2.0,"output":8.0,"prompt":"0.000002","completion":"0.000008","image":"0","request":"0","input_cache_read":"0.0000005"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"2.00","cachedInput":"0.50","output":"8.00"}},"priority":{"short":{"uncachedInput":"3.50","cachedInput":"0.875","output":"14.00"}}}},"context_length":1000000,"max_output_length":16384,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode"],"is_ready":false,"description":"OpenAI's flagship production model with 1M token context window. Excels at instruction following, coding, and long-context tasks. 75% cheaper cached input reads.","top_provider":{"context_length":1000000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-4.1-mini","object":"model","created":1778671374,"owned_by":"openai","name":"OpenAI GPT-4.1 Mini","pricing":{"input":0.4,"output":1.6,"prompt":"0.0000004","completion":"0.0000016","image":"0","request":"0","input_cache_read":"0.0000001"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"0.40","cachedInput":"0.10","output":"1.60"}},"priority":{"short":{"uncachedInput":"0.70","cachedInput":"0.175","output":"2.80"}}}},"context_length":1000000,"max_output_length":16384,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode"],"is_ready":false,"description":"Cost-effective version of GPT-4.1 with the same 1M token context window. Great balance of capability and cost for production workloads.","top_provider":{"context_length":1000000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-4.1-nano","object":"model","created":1778671374,"owned_by":"openai","name":"OpenAI GPT-4.1 Nano","pricing":{"input":0.1,"output":0.4,"prompt":"0.0000001","completion":"0.0000004","image":"0","request":"0","input_cache_read":"0.000000025"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"0.10","cachedInput":"0.025","output":"0.40"}},"priority":{"short":{"uncachedInput":"0.20","cachedInput":"0.05","output":"0.80"}}}},"context_length":1000000,"max_output_length":16384,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode"],"is_ready":false,"description":"OpenAI's most cost-efficient model with 1M token context. Ideal for classification, extraction, and high-volume tasks where cost matters most.","top_provider":{"context_length":1000000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-5","object":"model","created":1778671374,"owned_by":"openai","name":"OpenAI GPT-5","pricing":{"input":1.25,"output":10.0,"prompt":"0.00000125","completion":"0.00001","image":"0","request":"0","input_cache_read":"0.000000125"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"1.25","cachedInput":"0.125","output":"10.00"}},"flex":{"short":{"uncachedInput":"0.625","cachedInput":"0.0625","output":"5.00"}},"priority":{"short":{"uncachedInput":"2.50","cachedInput":"0.25","output":"20.00"}}}},"context_length":400000,"max_output_length":16384,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode","reasoning"],"is_ready":false,"description":"OpenAI's next-generation model with enhanced reasoning and 400K context window. Strong performance across coding, math, and creative tasks.","top_provider":{"context_length":400000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-5.1","object":"model","created":1778671374,"owned_by":"openai","name":"GPT-5.1","pricing":{"input":1.25,"output":10.0,"prompt":"0.00000125","completion":"0.00001","image":"0","request":"0","input_cache_read":"0.000000125"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"1.25","cachedInput":"0.125","output":"10.00"}},"flex":{"short":{"uncachedInput":"0.625","cachedInput":"0.0625","output":"5.00"}},"priority":{"short":{"uncachedInput":"2.50","cachedInput":"0.25","output":"20.00"}}}},"context_length":400000,"max_output_length":16384,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode","reasoning"],"is_ready":false,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","top_provider":{"context_length":400000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-5.2","object":"model","created":1769638083,"owned_by":"openai","name":"OpenAI GPT-5.2","pricing":{"input":1.75,"output":14.0,"prompt":"0.00000175","completion":"0.000014","image":"0","request":"0","input_cache_read":"0.000000175"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"1.75","cachedInput":"0.175","output":"14.00"}},"flex":{"short":{"uncachedInput":"0.875","cachedInput":"0.0875","output":"7.00"}},"priority":{"short":{"uncachedInput":"3.50","cachedInput":"0.35","output":"28.00"}}}},"context_length":400000,"max_output_length":16384,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode","reasoning"],"is_ready":false,"description":"OpenAI GPT-5.2 with 400k context window. Anonymized endpoint optimized for deep reasoning and large-context workflows.","top_provider":{"context_length":400000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-5.4","object":"model","created":1778671374,"owned_by":"openai","name":"GPT-5.4","pricing":{"input":2.5,"output":15.0,"prompt":"0.0000025","completion":"0.000015","image":"0","request":"0","input_cache_read":"0.00000025"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","longContextThreshold":272000,"tiers":{"default":{"short":{"uncachedInput":"2.50","cachedInput":"0.25","output":"15.00"},"long":{"uncachedInput":"5.00","cachedInput":"0.50","output":"22.50"}},"flex":{"short":{"uncachedInput":"1.25","cachedInput":"0.13","output":"7.50"},"long":{"uncachedInput":"2.50","cachedInput":"0.25","output":"11.25"}},"priority":{"short":{"uncachedInput":"5.00","cachedInput":"0.50","output":"30.00"}}}},"context_length":1050000,"max_output_length":16384,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode","reasoning"],"is_ready":false,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","top_provider":{"context_length":1050000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-5.4-mini","object":"model","created":1778671374,"owned_by":"openai","name":"GPT-5.4 Mini","pricing":{"input":0.75,"output":4.5,"prompt":"0.00000075","completion":"0.0000045","image":"0","request":"0","input_cache_read":"0.000000075"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"0.75","cachedInput":"0.075","output":"4.50"}},"flex":{"short":{"uncachedInput":"0.375","cachedInput":"0.0375","output":"2.25"}},"priority":{"short":{"uncachedInput":"1.50","cachedInput":"0.15","output":"9.00"}}}},"context_length":400000,"max_output_length":16384,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode","reasoning"],"is_ready":false,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","top_provider":{"context_length":400000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-5.4-nano","object":"model","created":1778671374,"owned_by":"openai","name":"GPT-5.4 Nano","pricing":{"input":0.2,"output":1.25,"prompt":"0.0000002","completion":"0.00000125","image":"0","request":"0","input_cache_read":"0.00000002"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"0.20","cachedInput":"0.02","output":"1.25"}},"flex":{"short":{"uncachedInput":"0.10","cachedInput":"0.01","output":"0.625"}}}},"context_length":400000,"max_output_length":16384,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode","reasoning"],"is_ready":false,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","top_provider":{"context_length":400000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-5.5","object":"model","created":1778671374,"owned_by":"openai","name":"GPT-5.5","pricing":{"input":5.0,"output":30.0,"prompt":"0.000005","completion":"0.00003","image":"0","request":"0","input_cache_read":"0.0000005"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","longContextThreshold":272000,"tiers":{"default":{"short":{"uncachedInput":"5.00","cachedInput":"0.50","output":"30.00"},"long":{"uncachedInput":"10.00","cachedInput":"1.00","output":"45.00"}},"flex":{"short":{"uncachedInput":"2.50","cachedInput":"0.25","output":"15.00"},"long":{"uncachedInput":"5.00","cachedInput":"0.50","output":"22.50"}},"priority":{"short":{"uncachedInput":"12.50","cachedInput":"1.25","output":"75.00"}}}},"context_length":1050000,"max_output_length":16384,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode","reasoning"],"is_ready":false,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","top_provider":{"context_length":1050000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-5.6-luna","object":"model","created":1788306273,"owned_by":"openai","name":"GPT-5.6 Luna","pricing":{"input":0.2,"output":1.2,"prompt":"0.0000002","completion":"0.0000012","image":"0","request":"0","input_cache_read":"0.00000002"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","longContextThreshold":272000,"tiers":{"default":{"short":{"uncachedInput":"0.20","cachedInput":"0.02","cacheWrite":"0.25","output":"1.20"},"long":{"uncachedInput":"0.40","cachedInput":"0.04","cacheWrite":"0.50","output":"1.80"}},"flex":{"short":{"uncachedInput":"0.10","cachedInput":"0.01","cacheWrite":"0.125","output":"0.60"},"long":{"uncachedInput":"0.20","cachedInput":"0.02","cacheWrite":"0.25","output":"0.90"}},"priority":{"short":{"uncachedInput":"0.40","cachedInput":"0.04","cacheWrite":"0.50","output":"2.40"},"long":{"uncachedInput":"0.80","cachedInput":"0.08","cacheWrite":"1.00","output":"3.60"}}}},"context_length":1050000,"max_output_length":128000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","stop","seed","max_tokens","logit_bias"],"supported_features":["tools","json_mode","structured_outputs","logprobs","reasoning"],"description":"OpenAI GPT-5.6 model optimized for cost-sensitive workloads","top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false}},{"id":"openai/gpt-5.6-sol","object":"model","created":1787226177,"owned_by":"openai","name":"GPT-5.6 Sol","pricing":{"input":4.0,"output":20.0,"prompt":"0.000004","completion":"0.00002","image":"0","request":"0","input_cache_read":"0.0000004"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","longContextThreshold":272000,"tiers":{"default":{"short":{"uncachedInput":"4.00","cachedInput":"0.40","cacheWrite":"5.00","output":"20.00"},"long":{"uncachedInput":"8.00","cachedInput":"0.80","cacheWrite":"10.00","output":"30.00"}},"flex":{"short":{"uncachedInput":"2.00","cachedInput":"0.20","cacheWrite":"2.50","output":"10.00"},"long":{"uncachedInput":"4.00","cachedInput":"0.40","cacheWrite":"5.00","output":"15.00"}},"priority":{"short":{"uncachedInput":"8.00","cachedInput":"0.80","cacheWrite":"10.00","output":"40.00"},"long":{"uncachedInput":"16.00","cachedInput":"1.60","cacheWrite":"20.00","output":"60.00"}}}},"context_length":1050000,"max_output_length":128000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","stop","seed","max_tokens","logit_bias"],"supported_features":["tools","json_mode","structured_outputs","logprobs","reasoning"],"description":"OpenAI frontier model for complex professional work","top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false}},{"id":"openai/gpt-5-mini","object":"model","created":1778671374,"owned_by":"openai","name":"GPT-5 Mini","pricing":{"input":0.25,"output":2.0,"prompt":"0.00000025","completion":"0.000002","image":"0","request":"0","input_cache_read":"0.000000025"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"0.25","cachedInput":"0.025","output":"2.00"}},"flex":{"short":{"uncachedInput":"0.125","cachedInput":"0.0125","output":"1.00"}},"priority":{"short":{"uncachedInput":"0.45","cachedInput":"0.045","output":"3.60"}}}},"context_length":400000,"max_output_length":16384,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode","reasoning"],"is_ready":false,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","top_provider":{"context_length":400000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-5-nano","object":"model","created":1778671374,"owned_by":"openai","name":"GPT-5 Nano","pricing":{"input":0.05,"output":0.4,"prompt":"0.00000005","completion":"0.0000004","image":"0","request":"0","input_cache_read":"0.000000005"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"0.05","cachedInput":"0.005","output":"0.40"}},"flex":{"short":{"uncachedInput":"0.025","cachedInput":"0.0025","output":"0.20"}}}},"context_length":400000,"max_output_length":16384,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode","reasoning"],"is_ready":false,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","top_provider":{"context_length":400000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"openai/gpt-6-astra","object":"model","created":1788569335,"owned_by":"openai","name":"GPT-6 Astra","pricing":{"input":10.0,"output":50.0,"prompt":"0.00001","completion":"0.00005","image":"0","request":"0","input_cache_read":"0.000001"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","longContextThreshold":272000,"tiers":{"default":{"short":{"uncachedInput":"10.00","cachedInput":"1.00","cacheWrite":"12.50","output":"50.00"},"long":{"uncachedInput":"20.00","cachedInput":"2.00","cacheWrite":"25.00","output":"75.00"}},"flex":{"short":{"uncachedInput":"5.00","cachedInput":"0.50","cacheWrite":"6.25","output":"25.00"},"long":{"uncachedInput":"10.00","cachedInput":"1.00","cacheWrite":"12.50","output":"37.50"}},"priority":{"short":{"uncachedInput":"20.00","cachedInput":"2.00","cacheWrite":"25.00","output":"100.00"},"long":{"uncachedInput":"40.00","cachedInput":"4.00","cacheWrite":"50.00","output":"150.00"}}}},"context_length":1050000,"max_output_length":128000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["max_tokens"],"supported_features":["structured_outputs","json_mode","reasoning"],"description":"OpenAI model for complex reasoning, coding, research, and document creation. Tool calling is not currently supported through NEAR AI Cloud.","top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false}},{"id":"openai/o3","object":"model","created":1778671374,"owned_by":"openai","name":"OpenAI o3","pricing":{"input":2.0,"output":8.0,"prompt":"0.000002","completion":"0.000008","image":"0","request":"0","input_cache_read":"0.0000005"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"2.00","cachedInput":"0.50","output":"8.00"}},"flex":{"short":{"uncachedInput":"1.00","cachedInput":"0.25","output":"4.00"}},"priority":{"short":{"uncachedInput":"3.50","cachedInput":"0.875","output":"14.00"}}}},"context_length":200000,"max_output_length":32768,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","reasoning"],"is_ready":false,"description":"OpenAI's flagship reasoning model. Uses chain-of-thought to solve complex math, coding, and logic problems. 200K context window.","top_provider":{"context_length":200000,"max_completion_tokens":32768,"is_moderated":false}},{"id":"openai/o3-mini","object":"model","created":1778671374,"owned_by":"openai","name":"o3 Mini","pricing":{"input":1.1,"output":4.4,"prompt":"0.0000011","completion":"0.0000044","image":"0","request":"0","input_cache_read":"0.00000055"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"1.10","cachedInput":"0.55","output":"4.40"}}}},"context_length":200000,"max_output_length":32768,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","reasoning"],"is_ready":false,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","top_provider":{"context_length":200000,"max_completion_tokens":32768,"is_moderated":false}},{"id":"openai/o4-mini","object":"model","created":1778671374,"owned_by":"openai","name":"OpenAI o4 Mini","pricing":{"input":1.1,"output":4.4,"prompt":"0.0000011","completion":"0.0000044","image":"0","request":"0","input_cache_read":"0.000000275"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","tiers":{"default":{"short":{"uncachedInput":"1.10","cachedInput":"0.275","output":"4.40"}},"flex":{"short":{"uncachedInput":"0.55","cachedInput":"0.138","output":"2.20"}},"priority":{"short":{"uncachedInput":"2.00","cachedInput":"0.50","output":"8.00"}}}},"context_length":200000,"max_output_length":32768,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","reasoning"],"is_ready":false,"description":"OpenAI's cost-effective reasoning model. Strong performance on math, coding, and scientific reasoning at a fraction of o3's cost. 200K context window.","top_provider":{"context_length":200000,"max_completion_tokens":32768,"is_moderated":false}},{"id":"openai/privacy-filter","object":"model","created":1779129892,"owned_by":"nearai","name":"Privacy Filter","pricing":{"input":0.01,"output":0.0,"prompt":"0.00000001","completion":"0","image":"0","request":"0"},"context_length":512,"max_output_length":1024,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":[],"supported_features":[],"description":"PII detection (token classification) — returns spans for emails, phones, addresses, names, account numbers, secrets. NEAR AI runs this model in a TEE; prompts are not anonymized by the model itself, the cloud-api wraps it to do redaction.","top_provider":{"context_length":512,"max_completion_tokens":1024,"is_moderated":false},"datacenters":[{"country_code":"US"}]},{"id":"openai/whisper-large-v3","object":"model","created":1774015272,"owned_by":"nearai","name":"Whisper Large v3","hugging_face_id":"openai/whisper-large-v3","pricing":{"input":0.01,"output":0.01,"prompt":"0.00000001","completion":"0.00000001","image":"0","request":"0"},"context_length":448,"max_output_length":1024,"architecture":{"inputModalities":["audio"],"outputModalities":["text"]},"input_modalities":["audio"],"output_modalities":["text"],"supported_sampling_parameters":["temperature"],"supported_features":[],"description":"Whisper is a state-of-the-art model for automatic speech recognition (ASR) and speech translation.","top_provider":{"context_length":448,"max_completion_tokens":1024,"is_moderated":false},"datacenters":[{"country_code":"US"}]},{"id":"qwen/qwen3-32b","object":"model","created":1781515743,"owned_by":"attested 3p","name":"qwen3-32b","pricing":{"input":0.11,"output":0.46,"prompt":"0.00000011","completion":"0.00000046","image":"0","request":"0","input_cache_read":"0.00000006"},"context_length":128000,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","stop","seed","max_tokens"],"supported_features":["tools","json_mode"],"is_ready":false,"deprecation_date":"2026-09-14T13:00:00Z","description":"Attested model served via Chutes TEE (verified end-to-end by NEAR AI).","top_provider":{"context_length":128000,"is_moderated":false}},{"id":"qwen/qwen3.5-397b-a17b","object":"model","created":1781515743,"owned_by":"attested 3p","name":"qwen3.5-397b-a17b","pricing":{"input":0.5,"output":3.3000000000000003,"prompt":"0.0000005","completion":"0.0000033","image":"0","request":"0","input_cache_read":"0.00000025"},"context_length":128000,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","frequency_penalty","presence_penalty","stop","seed","max_tokens"],"supported_features":["tools","json_mode"],"is_ready":false,"description":"Attested model served via Chutes TEE (verified end-to-end by NEAR AI).","top_provider":{"context_length":128000,"is_moderated":false}},{"id":"Qwen/Qwen3.6-35B-A3B-FP8","object":"model","created":1778743164,"owned_by":"nearai","name":"Qwen 3.6 35B A3B FP8","hugging_face_id":"Qwen/Qwen3.6-35B-A3B-FP8","quantization":"fp8","pricing":{"input":0.17,"output":1.1,"prompt":"0.00000017","completion":"0.0000011","image":"0","request":"0","input_cache_read":"0.000000056"},"context_length":262144,"max_output_length":8192,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","min_p","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens","logit_bias"],"supported_features":["tools","json_mode","structured_outputs","logprobs","reasoning"],"is_ready":true,"description":"Qwen 3.6 35B is a fast mixture-of-experts language model with ~3B active parameters per token. Strong at reasoning, coding, and multilingual tasks with 32K context window.","top_provider":{"context_length":262144,"max_completion_tokens":8192,"is_moderated":false},"datacenters":[{"country_code":"US"}],"openrouter":{"slug":"qwen/qwen3.6-35b-a3b"}},{"id":"qwen/qwen3.7-max","object":"model","created":1779449813,"owned_by":"qwen","name":"Qwen3.7 Max","pricing":{"input":2.8000000000000003,"output":7.5,"prompt":"0.0000028","completion":"0.0000075","image":"0","request":"0"},"context_length":1000000,"max_output_length":16384,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","json_mode"],"is_ready":false,"description":"Qwen's most capable proprietary model with 1M context window. Strong at reasoning, coding, math, and multilingual tasks.","top_provider":{"context_length":1000000,"max_completion_tokens":16384,"is_moderated":false}},{"id":"Qwen/Qwen3.8-27B","object":"model","created":1787042291,"owned_by":"nearai","name":"Qwen 3.8 27B","hugging_face_id":"Qwen/Qwen3.8-27B-FP8","quantization":"fp8","pricing":{"input":0.44,"output":3.3000000000000003,"prompt":"0.00000044","completion":"0.0000033","image":"0","request":"0","input_cache_read":"0.000000044"},"context_length":262144,"max_output_length":8192,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","max_tokens","stop","seed"],"supported_features":["tools","structured_outputs","reasoning"],"is_ready":true,"description":"Qwen 3.8 27B is a dense FP8 multimodal model with reasoning, coding, tool use, vision, and a 262K context window.","top_provider":{"context_length":262144,"max_completion_tokens":8192,"is_moderated":false},"datacenters":[{"country_code":"US"}],"openrouter":{"slug":"qwen/qwen3.8-27b"}},{"id":"Qwen/Qwen3-Embedding-0.6B","object":"model","created":1774015272,"owned_by":"nearai","name":"Qwen3-Embedding-0.6B","hugging_face_id":"Qwen/Qwen3-Embedding-0.6B","quantization":"bf16","pricing":{"input":0.01,"output":0.01,"prompt":"0.00000001","completion":"0.00000001","image":"0","request":"0"},"context_length":32768,"max_output_length":1024,"architecture":{"inputModalities":["text"],"outputModalities":["embedding"]},"input_modalities":["text"],"output_modalities":["embedding"],"supported_sampling_parameters":[],"supported_features":[],"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding tasks.","top_provider":{"context_length":32768,"max_completion_tokens":1024,"is_moderated":false},"datacenters":[{"country_code":"US"}]},{"id":"Qwen/Qwen3-Reranker-0.6B","object":"model","created":1774015272,"owned_by":"nearai","name":"Qwen3-Reranker-0.6B","hugging_face_id":"Qwen/Qwen3-Reranker-0.6B","quantization":"bf16","pricing":{"input":0.01,"output":0.01,"prompt":"0.00000001","completion":"0.00000001","image":"0","request":"0"},"context_length":40960,"max_output_length":1024,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":[],"supported_features":[],"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks.","top_provider":{"context_length":40960,"max_completion_tokens":1024,"is_moderated":false},"datacenters":[{"country_code":"US"}]},{"id":"Qwen/Qwen3-VL-30B-A3B-Instruct","object":"model","created":1774015272,"owned_by":"nearai","name":"Qwen3-VL-30B-A3B-Instruct","hugging_face_id":"Qwen/Qwen3-VL-30B-A3B-Instruct","quantization":"fp8","pricing":{"input":0.15,"output":0.55,"prompt":"0.00000015","completion":"0.00000055","image":"0","request":"0","input_cache_read":"0.00000003"},"context_length":16384,"max_output_length":8192,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","min_p","frequency_penalty","presence_penalty","repetition_penalty","max_tokens","stop","seed","logit_bias"],"supported_features":["structured_outputs","logprobs"],"is_ready":true,"description":"Qwen3-VL-30B-A3B-Instruct is a vision-language model supporting text and image inputs.","top_provider":{"context_length":16384,"max_completion_tokens":8192,"is_moderated":false},"datacenters":[{"country_code":"US"}],"openrouter":{"slug":"qwen/qwen3-vl-30b-a3b-instruct"}},{"id":"x-ai/grok-4.6","object":"model","created":1789094166,"owned_by":"x-ai","name":"Grok 4.6","pricing":{"input":2.0,"output":6.0,"prompt":"0.000002","completion":"0.000006","image":"0","request":"0","input_cache_read":"0.0000005"},"textPricing":{"version":1,"currency":"USD","unit":"million_tokens","longContextThreshold":199999,"tiers":{"default":{"short":{"uncachedInput":"2.00","cachedInput":"0.50","output":"6.00"},"long":{"uncachedInput":"4.00","cachedInput":"1.00","output":"12.00"}}}},"context_length":500000,"max_output_length":450000,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","max_tokens","seed"],"supported_features":["tools","structured_outputs","json_mode","reasoning","logprobs"],"is_ready":true,"description":"xAI model for reasoning, coding, knowledge work, and image understanding.","top_provider":{"context_length":500000,"max_completion_tokens":450000,"is_moderated":false}},{"id":"z-ai/glm-5.2","object":"model","created":1781700379,"owned_by":"nearai","name":"GLM 5.2","hugging_face_id":"zai-org/GLM-5.2-FP8","quantization":"fp8","pricing":{"input":1.4000000000000001,"output":4.4,"prompt":"0.0000014","completion":"0.0000044","image":"0","request":"0","input_cache_read":"0.0000003"},"context_length":1048576,"max_output_length":131072,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","min_p","frequency_penalty","presence_penalty","repetition_penalty","max_tokens","stop","seed","logit_bias"],"supported_features":["tools","structured_outputs","reasoning","json_mode"],"deprecation_date":"2026-09-11T13:00:00Z","description":"GLM-5.2 is an open-source foundation model featuring improved MTP and IndexShare over GLM-5.1. 753B MoE architecture, FP8 precision, optimized for complex systems engineering and long-horizon agent workflows.","top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false}},{"id":"z-ai/glm-5.3-flash","object":"model","created":1788287675,"owned_by":"nearai","name":"GLM 5.3 Flash","hugging_face_id":"zai-org/GLM-5.3-Flash","quantization":"fp8","pricing":{"input":0.15,"output":0.5,"prompt":"0.00000015","completion":"0.0000005","image":"0","request":"0","input_cache_read":"0.000000035"},"context_length":1048576,"max_output_length":131072,"architecture":{"inputModalities":["text","image"],"outputModalities":["text"]},"input_modalities":["text","image"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","min_p","frequency_penalty","presence_penalty","repetition_penalty","max_tokens","stop","seed","logit_bias"],"supported_features":["tools","structured_outputs","reasoning","json_mode"],"is_ready":true,"description":"GLM-5.3-Flash is a native multimodal mixture-of-experts model for coding, agentic workflows, reasoning, tool use, and visual understanding.","top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"openrouter":{"slug":"z-ai/glm-5.3-flash"}},{"id":"zai-org/GLM-5.1-FP8","object":"model","created":1776190834,"owned_by":"nearai","name":"GLM 5.1","hugging_face_id":"zai-org/GLM-5.1-FP8","quantization":"fp8","pricing":{"input":1.4000000000000001,"output":4.4,"prompt":"0.0000014","completion":"0.0000044","image":"0","request":"0","input_cache_read":"0.00000026"},"context_length":202752,"max_output_length":16384,"architecture":{"inputModalities":["text"],"outputModalities":["text"]},"input_modalities":["text"],"output_modalities":["text"],"supported_sampling_parameters":["temperature","top_p","top_k","min_p","frequency_penalty","presence_penalty","repetition_penalty","max_tokens","stop","seed","logit_bias"],"supported_features":["tools","structured_outputs","reasoning","json_mode"],"is_ready":true,"deprecation_date":"2026-09-11T13:00:00Z","description":"GLM-5.1 is an open-source foundation model built for complex systems engineering and long-horizon agent workflows. It delivers production-grade productivity for large-scale programming tasks, with performance aligned to top closed-source models, and is designed for expert developers building at the system level.","top_provider":{"context_length":202752,"max_completion_tokens":16384,"is_moderated":false},"datacenters":[{"country_code":"US"}],"openrouter":{"slug":"z-ai/glm-5.1"}}]}