{"data":[{"context_length":1000000,"created":1715731200,"datacenters":[{"country_code":"IN"}],"description":"Vision Small: India's fast and efficient foundational AI. Trained from scratch with native multimodal understanding (text, audio, video) and deep cultural awareness. Features 1M token context for high-throughput, real-time interactions across 250+ languages.","hugging_face_id":"","id":"vispark/vision-small","input_modalities":["text","image","audio","video","file"],"max_output_length":65536,"name":"Vispark: Vision Small","object":"model","output_modalities":["text"],"owned_by":"vispark","pricing":{"completion":"0.00000316","image":"0","input_cache_read":"0","prompt":"0.00000105","request":"0"},"quantization":"bf16","supported_features":["tools","json_mode","structured_outputs","reasoning"],"supported_sampling_parameters":["temperature","top_p","top_k","stop","max_tokens","seed","frequency_penalty","presence_penalty"]},{"context_length":1000000,"created":1715731200,"datacenters":[{"country_code":"IN"}],"description":"Vision Medium: A versatile foundational model built for the world on Indian infrastructure. Balances speed with robust reasoning and 1M token context. Excels at content creation, nuanced cultural interactions, and evolving personal assistance.","hugging_face_id":"","id":"vispark/vision-medium","input_modalities":["text","image","audio","video","file"],"max_output_length":65536,"name":"Vispark: Vision Medium","object":"model","output_modalities":["text"],"owned_by":"vispark","pricing":{"completion":"0.00001263","image":"0","input_cache_read":"0","prompt":"0.00000421","request":"0"},"quantization":"bf16","supported_features":["tools","json_mode","structured_outputs","reasoning"],"supported_sampling_parameters":["temperature","top_p","top_k","stop","max_tokens","seed","frequency_penalty","presence_penalty"]},{"context_length":1000000,"created":1715731200,"datacenters":[{"country_code":"IN"}],"description":"Vision Large: India's most advanced foundational intelligence. Delivers state-of-the-art reasoning, SOTA media processing, and complex tool usage. With 1M token context, it handles deep analysis and structured data with unmatched accuracy and cultural depth.","hugging_face_id":"","id":"vispark/vision-large","input_modalities":["text","image","audio","video","file"],"max_output_length":65536,"name":"Vispark: Vision Large","object":"model","output_modalities":["text"],"owned_by":"vispark","pricing":{"completion":"0.00002211","image":"0","input_cache_read":"0","prompt":"0.00000737","request":"0"},"quantization":"bf16","supported_features":["tools","json_mode","structured_outputs","reasoning"],"supported_sampling_parameters":["temperature","top_p","top_k","stop","max_tokens","seed","frequency_penalty","presence_penalty"]}],"object":"list"}
