{"object":"list","data":[{"id":"anthropic/claude-opus-4-8","name":"Agent 3","description":"Anthropic's most capable model for complex reasoning and agentic coding","context_length":1000000,"max_tokens":0,"pricing":{"input_cost_per_token":0.000005,"output_cost_per_token":0.000025},"provider":{},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"anthropic/claude-fable-5","name":"AIWEBBB MASTER PRODUCTION AGENT v1.0","description":"Anthropic's most capable widely released model, for the most demanding reasoning and long-horizon agentic work.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005},"provider":{},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"anthropic/claude-fable-5","name":"assdr","description":"Anthropic's most capable widely released model, for the most demanding reasoning and long-horizon agentic work.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005},"provider":{},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"anthropic/claude-opus-4-8","name":"BRUTHA - SQUIRE","description":"Anthropic's most capable widely released model, for the most demanding reasoning and long-horizon agentic work.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005},"provider":{},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-4.1","name":"Capex and GPU Financing News","description":"OpenAI’s fast and capable model for reasoning, coding, and chat. It responds quickly, supports long context (128k tokens), and runs efficiently at scale—ideal for advanced API applications.","context_length":1047576,"max_tokens":0,"pricing":{"input_cost_per_token":0.000002,"output_cost_per_token":0.000008},"provider":{},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-4-turbo-preview","name":"Lightning SDK expert","description":"","context_length":128000,"max_tokens":0,"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00003},"provider":{},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/gpt-4-turbo-preview","name":"LitLogger helper","description":"","context_length":128000,"max_tokens":0,"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00003},"provider":{},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"anthropic/claude-opus-4-8","name":"Ovarix IA Assistant","description":"Anthropic's most capable widely released model, for the most demanding reasoning and long-horizon agentic work.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005},"provider":{},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"google/gemini-3.5-flash","name":"Realistic Agent","description":"Google's most intelligent Flash model, Gemini 3.5 Flash delivers sustained frontier performance in agentic execution, coding, and long-horizon tasks at scale.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":0.0000015,"output_cost_per_token":0.000009},"provider":{},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"anthropic/claude-fable-5","name":"Roleway","description":"Anthropic's most capable widely released model, for the most demanding reasoning and long-horizon agentic work.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00005},"provider":{},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-5.5-2026-04-23","name":"Seedance 2 Director","description":"OpenAI newest frontier model for the most complex professional work","context_length":1050000,"max_tokens":0,"pricing":{"input_cost_per_token":0.000005,"output_cost_per_token":0.00003},"provider":{},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Gemini 2.5 Flash is Google’s powerful model built for complex reasoning, coding, math, and scientific challenges. With integrated “thinking” capabilities, it delivers more accurate answers and handles context with greater nuance.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":3e-7,"output_cost_per_token":0.0000025},"provider":{"name":"Google"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"google/gemini-2.5-flash-lite-preview-06-17","name":"Gemini 2.5 Flash Lite","description":"Gemini 2.5 Flash-Lite is a model developed by Google DeepMind, designed to handle various tasks including reasoning, science, mathematics, code generation, and more. It features advanced capabilities in multilingual performance and long context understanding. It is optimized for low latency use cases, supporting multimodal input with a 1 million-token context length.\n\n","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":1e-7,"output_cost_per_token":4e-7},"provider":{"name":"Google"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Gemini 2.5 Pro is Google’s AI model built for advanced reasoning, coding, math, and scientific work. With integrated “thinking” capabilities, it delivers more accurate answers and handles context with greater nuance. It ranks at the top of multiple benchmarks, including first place on the LMArena leaderboard, showcasing strong human-preference alignment and exceptional problem-solving skills.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":0.00000125,"output_cost_per_token":0.00001},"provider":{"name":"Google"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash","description":"Gemini 3 Flash Preview is designed to deliver strong agentic capabilities (near-Pro level) at substantial speed and value.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":0.000003},"provider":{"name":"Google"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro","description":"Gemini 3.1 Pro is Google's most advanced reasoning model, delivering a major leap in core intelligence and multimodal understanding for complex agentic, coding, and long-context tasks.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":0.000002,"output_cost_per_token":0.000012},"provider":{"name":"Google"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Google's most intelligent Flash model, Gemini 3.5 Flash delivers sustained frontier performance in agentic execution, coding, and long-horizon tasks at scale.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":0.0000015,"output_cost_per_token":0.000009},"provider":{"name":"Google"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"google/gemini-3.1-flash-lite-preview","name":"gemini-3.1-flash-lite-preview","description":"Gemini 3.1 Flash-Lite Preview is Google's most cost-efficient model, optimized for high-volume agentic tasks, translation, and simple data processing","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":0.0000015},"provider":{"name":"Google"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"lightning-ai/deepseek-v4.1-flash","name":"deepseek-v4.1-flash","description":"DeepSeek V4.1 Flash is smarter, faster, and more efficient, designed for greater capability and faster inference.","context_length":1048576,"max_tokens":0,"pricing":{"input_cost_per_token":3.000000106112566e-7,"output_cost_per_token":0.0000012000000424450263},"provider":{"name":"lightning-ai"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"lightning-ai/gemma-4-31B-it","name":"gemma-4-31B-it","description":"Gemma 4 is an open-weight, multimodal model family built by Google DeepMind, supporting text, image, and audio inputs with strong reasoning and coding capabilities.\nIt spans efficient small models to large dense and mixture-of-experts architectures, offering long context windows and scalable deployment from edge devices to high-performance systems.","context_length":131072,"max_tokens":0,"pricing":{"input_cost_per_token":1.3999999737279722e-7,"output_cost_per_token":4.0000000467443897e-7},"provider":{"name":"lightning-ai"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"lightning-ai/glm-5.3","name":"glm-5.3","description":"GLM-5.3 is Z.ai's flagship frontier coding and agentic model. As the most capable open source model on the market, it sacrifices neither performance nor price.","context_length":131072,"max_tokens":0,"pricing":{"input_cost_per_token":0.0000013999999737279722,"output_cost_per_token":0.000004399999852466863},"provider":{"name":"lightning-ai"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"lightning-ai/glm-5.3-flash","name":"glm-5.3-flash","description":"GLM-5.3-Flash served on Lightning AI","context_length":253952,"max_tokens":0,"pricing":{"input_cost_per_token":1.5e-7,"output_cost_per_token":5e-7},"provider":{"name":"lightning-ai"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"lightning-ai/mimo-v2.6-pro","name":"mimo-v2.6-pro","description":"Xiaomi's MiMo-V2.6-Pro is the strongest open-source model to date, according to the Artificial Analysis Intelligence Index, surpassing Kimi K3 and Qwen3.8 Max. As the most capable open source model on the market, it sacrifices neither performance nor price.","context_length":262144,"max_tokens":0,"pricing":{"input_cost_per_token":4.400000079840538e-7,"output_cost_per_token":8.700000080352766e-7},"provider":{"name":"lightning-ai"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"lightning-ai/Qwen3.8-27B","name":"Qwen3.8-27B","description":"Qwen3.8-27B is a dense 27B parameter language model built to deliver strong general purpose reasoning, coding, and long-context performance. Competitive with large frontier models on the Artificial Analysis Intelligence Index, including GPT-5.6 Luna, DeepSeek V4 Pro, and GLM 5.2, Qwen3.8-27B punches above its weight class at an unbeatable price.","context_length":262144,"max_tokens":0,"pricing":{"input_cost_per_token":4.0000000467443897e-7,"output_cost_per_token":0.000003000000106112566},"provider":{"name":"lightning-ai"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/gpt-3.5-turbo","name":"GPT 3.5 turbo","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.","context_length":16385,"max_tokens":0,"pricing":{"input_cost_per_token":5e-7,"output_cost_per_token":0.0000015},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/gpt-4","name":"GPT 4","description":"The default GPT-4 model with an 8,192-token context window.","context_length":8192,"max_tokens":0,"pricing":{"input_cost_per_token":0.00003,"output_cost_per_token":0.00006},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/gpt-4-turbo","name":"GPT 4 turbo","description":"GPT-4 Turbo is an upgraded version of GPT-4, designed to deliver the same high intelligence while being faster and more cost-effective.","context_length":128000,"max_tokens":0,"pricing":{"input_cost_per_token":0.00001,"output_cost_per_token":0.00003},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/gpt-4.1","name":"GPT 4.1","description":"OpenAI’s fast and capable model for reasoning, coding, and chat. It responds quickly, supports long context (128k tokens), and runs efficiently at scale—ideal for advanced API applications.","context_length":1047576,"max_tokens":0,"pricing":{"input_cost_per_token":0.000002,"output_cost_per_token":0.000008},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-4o","name":"GPT 4o","description":"GPT-4o is OpenAI’s multimodal model, handling both text and image inputs with text outputs. It matches the intelligence of GPT-4 Turbo but runs twice as fast at half the cost. The model also brings stronger non-English language support and improved visual understanding.","context_length":128000,"max_tokens":0,"pricing":{"input_cost_per_token":0.0000025,"output_cost_per_token":0.00001,"base_image_tokens":85},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-5","name":"GPT 5","description":"GPT-5 is OpenAI model that is built for complex, step-by-step tasks that demand precise instruction following and high-stakes accuracy. It supports test-time routing plus prompts like “think hard about this.” It also reduces hallucinations and sycophancy while improving performance in coding, writing, and health-related tasks.\n","context_length":400000,"max_tokens":0,"pricing":{"input_cost_per_token":0.00000125,"output_cost_per_token":0.00001},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/gpt-5-mini","name":"GPT 5 mini","description":"GPT-5 Mini is a smaller, more efficient variant of GPT-5 built for lighter reasoning tasks. It maintains the strong instruction-following and safety features of GPT-5 while offering faster responses and lower costs.","context_length":400000,"max_tokens":0,"pricing":{"input_cost_per_token":2.5e-7,"output_cost_per_token":0.000002},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/gpt-5-nano","name":"GPT 5 nano","description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, built as a unified system that can decide when to respond quickly or dig deeper depending on what you ask it. It gets much better across a wide range of skills — writing, coding, math, health, and visual tasks — and is more reliable and accurate in real-world scenarios. Compared to earlier models, it hallucinates less, follows instructions more faithfully, and understands the context better.","context_length":400000,"max_tokens":0,"pricing":{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/gpt-5.2-2025-12-11","name":"GPT 5.2","description":"GPT-5.2 is OpenAI's flagship model for coding and agentic tasks across industries. ","context_length":400000,"max_tokens":0,"pricing":{"input_cost_per_token":0.00000175,"output_cost_per_token":0.000014},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-5.4-2026-03-05","name":"GPT 5.4","description":"OpenAI frontier model for complex professional work","context_length":1050000,"max_tokens":0,"pricing":{"input_cost_per_token":0.0000025,"output_cost_per_token":0.000015},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-5.4-mini-2026-03-17","name":"GPT 5.4 mini","description":"GPT-5.4 mini brings the strengths of GPT-5.4 to a faster, more efficient model designed for high-volume workloads","context_length":400000,"max_tokens":0,"pricing":{"input_cost_per_token":7.5e-7,"output_cost_per_token":0.0000045},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-5.4-nano-2026-03-17","name":"GPT 5.4 nano","description":"GPT-5.4 nano is designed for tasks where speed and cost matter most like classification, data extraction, ranking, and sub-agents","context_length":400000,"max_tokens":0,"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":0.00000125},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-5.5-2026-04-23","name":"GPT 5.5","description":"OpenAI newest frontier model for the most complex professional work","context_length":1050000,"max_tokens":0,"pricing":{"input_cost_per_token":0.000005,"output_cost_per_token":0.00003},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-5.6-luna","name":"GPT 5.6 Luna","description":"GPT-5.6 Luna is designed for cost-sensitive, high-volume workloads. It roughly corresponds to the nano model tier used in earlier GPT-5 families.","context_length":1050000,"max_tokens":0,"pricing":{"input_cost_per_token":2e-7,"output_cost_per_token":0.0000012},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-5.6-sol","name":"GPT 5.6 Sol","description":"GPT-5.6 Sol is the frontier model in the GPT-5.6 family. It roughly corresponds to the unsuffixed model tier used in earlier GPT-5 families","context_length":1050000,"max_tokens":0,"pricing":{"input_cost_per_token":0.000005,"output_cost_per_token":0.00003},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-5.6-terra","name":"GPT 5.6 Terra","description":"GPT-5.6 Terra is designed for workloads that balance intelligence and cost. It roughly corresponds to the mini model tier used in earlier GPT-5 families","context_length":1050000,"max_tokens":0,"pricing":{"input_cost_per_token":0.000002,"output_cost_per_token":0.000012},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/gpt-6-astra","name":"gpt-6-astra","description":"","context_length":1050000,"max_tokens":0,"pricing":{"input_cost_per_token":0.000009999999747378752,"output_cost_per_token":0.00004999999873689376},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/gpt-6-sol","name":"gpt-6-sol","description":"","context_length":1050000,"max_tokens":0,"pricing":{"input_cost_per_token":0.0000019999999949504854,"output_cost_per_token":0.000009999999747378752},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/gpt-6.1-sol","name":"gpt-6.1-sol","description":"","context_length":1050000,"max_tokens":0,"pricing":{"input_cost_per_token":0.000002,"output_cost_per_token":0.00001},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}},{"id":"openai/o3","name":"o3","description":"o3 is a versatile, high-performing model across many domains. It raises the bar in math, science, coding, and visual reasoning, and it’s excellent at technical writing and following instructions. Use it to tackle multi-step problems that combine text, code, and images.","context_length":200000,"max_tokens":0,"pricing":{"input_cost_per_token":0.000002,"output_cost_per_token":0.000008},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"openai/o3-mini","name":"o3 mini","description":"OpenAI o3-mini is a lightweight, cost-efficient model built for STEM reasoning tasks like math, science, and coding. It lets you adjust its reasoning effort to balance speed and depth.The model delivers strong accuracy, matching the larger o1 on tough benchmarks while running faster and cheaper.","context_length":200000,"max_tokens":0,"pricing":{"input_cost_per_token":0.0000011,"output_cost_per_token":0.0000044},"provider":{"name":"OpenAI"},"architecture":{"input_modalities":["text"],"output_modalities":["text"]}}]}