{"models":[{"active":true,"aggregate_sha256":"61bfc04e4016a7fa487eb10e29f79360047e302487229f298da3681984aec512","architecture":"MoE","capabilities":["chat"],"created":1779749187,"description":"OpenAI's open-weight 20.9B parameter MoE reasoning model with 3.6B active parameters per token. Supports configurable reasoning effort, full chain-of-thought, function calling, and structured outputs. Apache 2.0 licensed.","display_name":"GPT-OSS 20B","family":"gpt-oss","file_count":10,"hugging_face_id":"gpt-oss-20b","id":"gpt-oss-20b","input_modalities":["text"],"max_context_length":131072,"max_output_length":32768,"metadata":{},"min_ram_gb":24,"model_type":"text","name":"GPT-OSS 20B","output_modalities":["text"],"quantization":"fp8","r2_prefix":"v2/gpt-oss-20b--ca1fe4e9c149/2026-05-25-r1","required_provider_capabilities":[],"runtime_parameters":{},"s3_name":"v2/gpt-oss-20b--ca1fe4e9c149/2026-05-25-r1","size_gb":12.104215835,"status":"active","supported_features":null,"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":12104215835,"version":"2026-05-25-r1","weight_hash":"61bfc04e4016a7fa487eb10e29f79360047e302487229f298da3681984aec512"},{"active":true,"aggregate_sha256":"d932e96b00404b0575fff47e2dac8ed113056b3f22d0040c3c8d3f9ef25b09ed","architecture":"Qwen3.5 MoE VLM with inline MTP","capabilities":["chat","tools","vision"],"created":1786564387,"description":"Qwen3.6 35B-A3B vision-language MoE with BF16 vision and inline lossless serial-verification MTP.","display_name":"Qwen 3.6 35B A3B","family":"Qwen3.6","file_count":13,"hugging_face_id":"Qwen/Qwen3.6-35B-A3B","id":"qwen3.6-35b-a3b-vl-mtp-mxfp8","input_modalities":["text","image"],"max_context_length":262144,"max_output_length":16384,"metadata":{"canary_status":"provider_verified","hub_revision":"73a03825c2226177f3e679210965dba3508cdee8","hugging_face_id":"Qwen/Qwen3.6-35B-A3B","manifest_sha256":"d932e96b00404b0575fff47e2dac8ed113056b3f22d0040c3c8d3f9ef25b09ed","mtp_verification":"serial_target","openrouter_slug":"qwen/qwen3.6-35b-a3b","video_verified":false},"min_ram_gb":32,"model_type":"text","name":"Qwen 3.6 35B A3B","output_modalities":["text"],"quantization":"fp4","r2_prefix":"v2/qwen3.6-35b-a3b-vl-mtp-mxfp8--943d60189096/2026-08-11-r1","required_provider_capabilities":[],"runtime_parameters":{"temperature":1,"top_k":20,"top_p":0.95},"s3_name":"v2/qwen3.6-35b-a3b-vl-mtp-mxfp8--943d60189096/2026-08-11-r1","size_gb":21.308856601,"status":"active","supported_features":["tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":21308856601,"version":"2026-08-11-r1","weight_hash":"d932e96b00404b0575fff47e2dac8ed113056b3f22d0040c3c8d3f9ef25b09ed"},{"active":true,"aggregate_sha256":"45327562e9de4bdac5c2d36df675aa1d8a981f9edf5dee419609e2e51bd82fff","architecture":"Qwen3-VL MoE VLM","capabilities":["chat","tools","vision"],"created":1787904715,"description":"Qwen3-VL 30B-A3B Instruct vision-language MoE with MLX affine W4/G64 weights, BF16 vision tower, causal DeepStack image prefill, and tool calling. Text and image serving are production-verified on Darkbloom v0.8.14; video remains disabled.","display_name":"Qwen3-VL 30B A3B Instruct","family":"Qwen3-VL","file_count":16,"hugging_face_id":"Qwen/Qwen3-VL-30B-A3B-Instruct","id":"qwen3-vl-30b-a3b-instruct","input_modalities":["text","image"],"max_context_length":131072,"max_output_length":32768,"metadata":{"canary_status":"provider_verified","hub_revision":"d5ee6d41b082b7803ca762d749bf425878ca5f4b","hugging_face_id":"Qwen/Qwen3-VL-30B-A3B-Instruct","manifest_sha256":"45327562e9de4bdac5c2d36df675aa1d8a981f9edf5dee419609e2e51bd82fff","openrouter_slug":"qwen/qwen3-vl-30b-a3b-instruct","video_verified":false},"min_ram_gb":32,"model_type":"text","name":"Qwen3-VL 30B A3B Instruct","output_modalities":["text"],"quantization":"fp4","r2_prefix":"v2/qwen3-vl-30b-a3b-instruct--4e09d3bcf035/2026-08-28-r1","required_provider_capabilities":[],"runtime_parameters":{"temperature":0.7,"top_k":20,"top_p":0.8},"s3_name":"v2/qwen3-vl-30b-a3b-instruct--4e09d3bcf035/2026-08-28-r1","size_gb":18.268169822,"status":"active","supported_features":["tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":18268169822,"version":"2026-08-28-r1","weight_hash":"45327562e9de4bdac5c2d36df675aa1d8a981f9edf5dee419609e2e51bd82fff"},{"active":true,"aggregate_sha256":"a4722b6020adb1894c700b45ddcd58bc0e0f033abe7139f86cbbbfe60cba4eb6","architecture":"MoE","capabilities":["chat"],"created":1779750101,"description":"Google DeepMind's 25.2B parameter MoE model with 3.8B active parameters per token. Supports multimodal input (text + image), configurable thinking modes, function calling, and 256K context. Apache 2.0 licensed.","display_name":"Gemma 4 26B","family":"gemma","file_count":13,"hugging_face_id":"gemma-4-26b","id":"gemma-4-26b","input_modalities":["text"],"max_context_length":131072,"max_output_length":32768,"metadata":{},"min_ram_gb":36,"model_type":"text","name":"Gemma 4 26B","output_modalities":["text"],"quantization":"8bit","r2_prefix":"v2/gemma-4-26b--9e201d98f8d4/2026-05-25-r1","required_provider_capabilities":[],"runtime_parameters":{},"s3_name":"v2/gemma-4-26b--9e201d98f8d4/2026-05-25-r1","size_gb":27.986040506,"status":"active","supported_features":null,"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":27986040506,"version":"2026-05-25-r1","weight_hash":"a4722b6020adb1894c700b45ddcd58bc0e0f033abe7139f86cbbbfe60cba4eb6"},{"active":true,"aggregate_sha256":"2468a0cb3049a871f42052f4d9f9380bf12a0792f64c7a29f768559fc7d28785","architecture":"","capabilities":["chat","tools","reasoning","json_mode","vision","video"],"created":1781143034,"description":"","display_name":"Gemma 4 26B","family":"","file_count":10,"hugging_face_id":"gemma-4-26b-qat-4bit","id":"gemma-4-26b-qat-4bit","input_modalities":["text","image","video"],"max_context_length":131072,"max_output_length":32768,"metadata":{"spec_dec":{"allowed_file_types":["config","weight"],"config_sha256":"0cd54ff36e53a258532c5c1433bc44b88ba758cfe9b59bb4e6eecfd5453fabcf","file_count":2,"manifest_sha256":"8b7c00b7f131345156f5f20fa9c94a895c5340f16d9331bafb9e59628bf45bf2","max_file_count":2,"r2_prefix":"v2/mlx-community-gemma-4-26B-A4B-it-qat-assistant-4bit--2acc3ce46483/bb94eae1b70a80dac16cbf959bb4b7d56bd1fb8c","revision":"bb94eae1b70a80dac16cbf959bb4b7d56bd1fb8c","total_size_bytes":236127665}},"min_ram_gb":36,"model_type":"text","name":"Gemma 4 26B","output_modalities":["text"],"quantization":"4bit","r2_prefix":"v2/gemma-4-26b-qat-4bit--69619d858022/2026-06-08-r1","required_provider_capabilities":[],"runtime_parameters":{},"s3_name":"v2/gemma-4-26b-qat-4bit--69619d858022/2026-06-08-r1","size_gb":15.641239295,"status":"beta","supported_features":["json_mode","reasoning","tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":15641239295,"version":"2026-06-08-r1","weight_hash":"2468a0cb3049a871f42052f4d9f9380bf12a0792f64c7a29f768559fc7d28785"},{"active":true,"aggregate_sha256":"95811153b3bb2ed78bf44b3248b07b52fce637706107de8b0fddf21796ade01c","architecture":"Qwen3.5 MoE VLM with inline MTP","capabilities":["chat","tools","vision"],"created":1787700397,"description":"Qwen3.5 35B-A3B vision-language MoE with affine W4/G64 target and inline MTP modules plus BF16 vision.","display_name":"Qwen3.5 35B A3B","family":"Qwen3.5","file_count":14,"hugging_face_id":"Qwen/Qwen3.5-35B-A3B","id":"qwen3.5-35b-a3b","input_modalities":["text","image"],"max_context_length":262144,"max_output_length":16384,"metadata":{"canary_status":"text_provider_verified","hub_revision":"59d61f3ce65a6d9863b86d2e96597125219dc754","hugging_face_id":"Qwen/Qwen3.5-35B-A3B","manifest_sha256":"95811153b3bb2ed78bf44b3248b07b52fce637706107de8b0fddf21796ade01c","mtp_verification":"structural_only","video_verified":false},"min_ram_gb":36,"model_type":"text","name":"Qwen3.5 35B A3B","output_modalities":["text"],"quantization":"fp4","r2_prefix":"v2/qwen3.5-35b-a3b--f1f11460a4d1/2026-08-25-r1","required_provider_capabilities":[],"runtime_parameters":{"temperature":1,"top_k":20,"top_p":0.95},"s3_name":"v2/qwen3.5-35b-a3b--f1f11460a4d1/2026-08-25-r1","size_gb":20.893747852,"status":"beta","supported_features":["tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":20893747852,"version":"2026-08-25-r1","weight_hash":"95811153b3bb2ed78bf44b3248b07b52fce637706107de8b0fddf21796ade01c"},{"active":true,"aggregate_sha256":"a4722b6020adb1894c700b45ddcd58bc0e0f033abe7139f86cbbbfe60cba4eb6","architecture":"","capabilities":["chat"],"created":1781549910,"description":"","display_name":"Gemma 4 26B 8-bit (rollback)","family":"","file_count":13,"hugging_face_id":"gemma-4-26b-8bit","id":"gemma-4-26b-8bit","input_modalities":["text"],"max_context_length":131072,"max_output_length":32768,"metadata":{},"min_ram_gb":64,"model_type":"text","name":"Gemma 4 26B 8-bit (rollback)","output_modalities":["text"],"quantization":"8bit","r2_prefix":"v2/gemma-4-26b-8bit--0402fbf4954e/2026-05-25-r1","required_provider_capabilities":[],"runtime_parameters":{},"s3_name":"v2/gemma-4-26b-8bit--0402fbf4954e/2026-05-25-r1","size_gb":27.986040506,"status":"beta","supported_features":null,"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":27986040506,"version":"2026-05-25-r1","weight_hash":"a4722b6020adb1894c700b45ddcd58bc0e0f033abe7139f86cbbbfe60cba4eb6"}]}