{"models":[{"active":true,"aggregate_sha256":"61bfc04e4016a7fa487eb10e29f79360047e302487229f298da3681984aec512","architecture":"MoE","capabilities":["chat"],"created":1779749187,"description":"OpenAI's open-weight 20.9B parameter MoE reasoning model with 3.6B active parameters per token. Supports configurable reasoning effort, full chain-of-thought, function calling, and structured outputs. Apache 2.0 licensed.","display_name":"GPT-OSS 20B","family":"gpt-oss","file_count":10,"hugging_face_id":"gpt-oss-20b","id":"gpt-oss-20b","input_modalities":["text"],"max_context_length":131072,"max_output_length":32768,"metadata":{},"min_ram_gb":24,"model_type":"text","name":"GPT-OSS 20B","output_modalities":["text"],"quantization":"fp8","r2_prefix":"v2/gpt-oss-20b--ca1fe4e9c149/2026-05-25-r1","required_provider_capabilities":[],"runtime_parameters":{},"s3_name":"v2/gpt-oss-20b--ca1fe4e9c149/2026-05-25-r1","size_gb":12.104215835,"status":"active","supported_features":null,"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":12104215835,"version":"2026-05-25-r1","weight_hash":"61bfc04e4016a7fa487eb10e29f79360047e302487229f298da3681984aec512"},{"active":true,"aggregate_sha256":"127de76b4ef82b7aaa0acaac0ee31c784cff066eda64291f521f051469b7c24b","architecture":"Qwen3.5 dense VLM with inline MTP","capabilities":["chat","tools","vision"],"created":1788479012,"description":"Qwen 3.5 9B dense vision-language model with inline multi-token prediction (4-bit). Runs on any Apple Silicon Mac with 24 GB or more.","display_name":"Qwen 3.5 9B","family":"Qwen3.5","file_count":12,"hugging_face_id":"Qwen/Qwen3.5-9B","id":"Qwen3.5-9B","input_modalities":["text","image"],"max_context_length":262144,"max_output_length":65536,"metadata":{"activation_basis":"M4 Max B=8 text-only cells 2026-09-03; VL bound ~5.0, under the 5.5 default","activation_measured_gib":2.9,"artifact_repo":"EigenLabs/Qwen3.5-9B-MLX-4bit-mtp","hub_revision":"556dcba57d","hugging_face_id":"Qwen/Qwen3.5-9B","mtp_verification":"inline","openrouter_slug":"qwen/qwen3.5-9b","video_verified":false},"min_ram_gb":24,"model_type":"text","name":"Qwen 3.5 9B","output_modalities":["text"],"quantization":"fp4","r2_prefix":"v2/Qwen3.5-9B--3e3c04992a0d/2026-09-03-r1","required_provider_capabilities":[],"runtime_parameters":{"temperature":1,"tool_call_parser":"qwen3_5","top_k":20,"top_p":0.95},"s3_name":"v2/Qwen3.5-9B--3e3c04992a0d/2026-09-03-r1","size_gb":6.11395223,"status":"active","supported_features":["tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":6113952230,"version":"2026-09-03-r1","weight_hash":"127de76b4ef82b7aaa0acaac0ee31c784cff066eda64291f521f051469b7c24b"},{"active":true,"aggregate_sha256":"ea1e901e4946c0ba9ad70c78517548808b353db6b3a13e87a8fa20468d81244c","architecture":"prism_hadamard_qwen35","capabilities":["chat","tools","reasoning","vision"],"created":1789747707,"description":"PrismML Bonsai 2 27B, a dense Qwen3.8-derived ternary model in native MLX 2-bit packing with vision and tool support. Uses the published checkpoint without requantization; no MTP heads.","display_name":"PrismML Bonsai 2 27B","family":"Bonsai 2","file_count":8,"hugging_face_artifact":{"repo_id":"EigenLabs/Ternary-Bonsai-2-27B-MLX-2bit","revision":"0e78886223303778e33d05058b90d2d9648488f0"},"hugging_face_id":"prism-ml/Ternary-Bonsai-2-27B-gguf","id":"ternary-bonsai-2-27b","input_modalities":["text","image"],"max_context_length":262144,"max_output_length":32768,"metadata":{"canary_status":"staged_pending_signed_serving_validation","context_qualification":"262144 is the artifact architectural limit, not a per-machine full-context performance guarantee.","hugging_face_id":"prism-ml/Ternary-Bonsai-2-27B-gguf","min_ram_qualification":"24 GB is a provisional entry floor supported by default load-budget arithmetic; validate on a physical 24 GB provider before promotion. Longer context, vision and concurrency remain subject to live memory admission.","mtp_present":false,"openrouter_slug":"prism-ml/Ternary-Bonsai-2-27B","source_artifact_revision":"3f926b415992eaa2ae9dd7b573706494d6bbf787","video_verified":false},"min_ram_gb":24,"model_type":"text","name":"PrismML Bonsai 2 27B","output_modalities":["text"],"quantization":"2bit","r2_prefix":"v2/ternary-bonsai-2-27b--5ef58466aad5/2026-09-17-r1","required_provider_capabilities":[],"runtime_parameters":{"temperature":1,"tool_call_parser":"qwen3_5","top_k":20,"top_p":0.95},"s3_name":"v2/ternary-bonsai-2-27b--5ef58466aad5/2026-09-17-r1","size_gb":8.608670713,"status":"active","supported_features":["reasoning","tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":8608670713,"version":"2026-09-17-r1","weight_hash":"ea1e901e4946c0ba9ad70c78517548808b353db6b3a13e87a8fa20468d81244c"},{"active":true,"aggregate_sha256":"d932e96b00404b0575fff47e2dac8ed113056b3f22d0040c3c8d3f9ef25b09ed","architecture":"Qwen3.5 MoE VLM with inline MTP","capabilities":["chat","tools","vision"],"created":1786564387,"description":"Qwen3.6 35B-A3B vision-language MoE with BF16 vision and inline lossless serial-verification MTP.","display_name":"Qwen 3.6 35B A3B","family":"Qwen3.6","file_count":13,"hugging_face_id":"Qwen/Qwen3.6-35B-A3B","id":"qwen3.6-35b-a3b-vl-mtp-mxfp8","input_modalities":["text","image"],"max_context_length":262144,"max_output_length":32768,"metadata":{"canary_status":"provider_verified","hub_revision":"73a03825c2226177f3e679210965dba3508cdee8","hugging_face_id":"Qwen/Qwen3.6-35B-A3B","manifest_sha256":"d932e96b00404b0575fff47e2dac8ed113056b3f22d0040c3c8d3f9ef25b09ed","mtp_verification":"serial_target","openrouter_slug":"qwen/qwen3.6-35b-a3b","video_verified":false},"min_ram_gb":32,"model_type":"text","name":"Qwen 3.6 35B A3B","output_modalities":["text"],"quantization":"fp4","r2_prefix":"v2/qwen3.6-35b-a3b-vl-mtp-mxfp8--943d60189096/2026-08-11-r1","required_provider_capabilities":[],"runtime_parameters":{"temperature":1,"top_k":20,"top_p":0.95},"s3_name":"v2/qwen3.6-35b-a3b-vl-mtp-mxfp8--943d60189096/2026-08-11-r1","size_gb":21.308856601,"status":"active","supported_features":["tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":21308856601,"version":"2026-08-11-r1","weight_hash":"d932e96b00404b0575fff47e2dac8ed113056b3f22d0040c3c8d3f9ef25b09ed"},{"active":true,"aggregate_sha256":"bbd0e0adcfe74e095073fefd0b9e116e4311d606ad9989cf81f8175e8ac18463","architecture":"Qwen3.5 dense VLM with inline MTP","capabilities":["chat","tools","vision"],"created":1788473597,"description":"Qwen 3.8 27B dense vision-language model with inline multi-token prediction (4-bit). Apple M5 providers only (requires the NAX runtime).","display_name":"Qwen 3.8 27B","family":"Qwen3.8","file_count":14,"hugging_face_id":"Qwen/Qwen3.8-27B","id":"EigenLabs/Qwen3.8-27B-4bit-mtp","input_modalities":["text","image"],"max_context_length":262144,"max_output_length":32768,"metadata":{"activation_floor_basis":"M4 Max non-NAX cells 2026-09-03; re-measure on M5","activation_floor_gib_provisional":9,"hub_revision":"06d517d395dfc5588090f7f534112bee331f7b4a","hugging_face_id":"Qwen/Qwen3.8-27B","mtp_verification":"inline","openrouter_slug":"qwen/qwen3.8-27b","video_verified":false},"min_ram_gb":36,"model_type":"text","name":"Qwen 3.8 27B","output_modalities":["text"],"quantization":"fp4","r2_prefix":"v2/EigenLabs-Qwen3.8-27B-4bit-mtp--96c072dbd610/2026-09-03-r1","required_provider_capabilities":["apple_m5","mlx_nax"],"runtime_parameters":{"temperature":1,"tool_call_parser":"qwen3_5","top_k":20,"top_p":0.95},"s3_name":"v2/EigenLabs-Qwen3.8-27B-4bit-mtp--96c072dbd610/2026-09-03-r1","size_gb":16.320415757,"status":"active","supported_features":["tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":16320415757,"version":"2026-09-03-r1","weight_hash":"bbd0e0adcfe74e095073fefd0b9e116e4311d606ad9989cf81f8175e8ac18463"},{"active":true,"aggregate_sha256":"a4722b6020adb1894c700b45ddcd58bc0e0f033abe7139f86cbbbfe60cba4eb6","architecture":"MoE","capabilities":["chat"],"created":1779750101,"description":"Google DeepMind's 25.2B parameter MoE model with 3.8B active parameters per token. Supports multimodal input (text + image), configurable thinking modes, function calling, and 256K context. Apache 2.0 licensed.","display_name":"Gemma 4 26B","family":"gemma","file_count":13,"hugging_face_id":"gemma-4-26b","id":"gemma-4-26b","input_modalities":["text"],"max_context_length":131072,"max_output_length":32768,"metadata":{},"min_ram_gb":36,"model_type":"text","name":"Gemma 4 26B","output_modalities":["text"],"quantization":"8bit","r2_prefix":"v2/gemma-4-26b--9e201d98f8d4/2026-05-25-r1","required_provider_capabilities":[],"runtime_parameters":{},"s3_name":"v2/gemma-4-26b--9e201d98f8d4/2026-05-25-r1","size_gb":27.986040506,"status":"active","supported_features":null,"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":27986040506,"version":"2026-05-25-r1","weight_hash":"a4722b6020adb1894c700b45ddcd58bc0e0f033abe7139f86cbbbfe60cba4eb6"},{"active":true,"aggregate_sha256":"2468a0cb3049a871f42052f4d9f9380bf12a0792f64c7a29f768559fc7d28785","architecture":"","capabilities":["chat","tools","reasoning","json_mode","vision","video"],"created":1781143034,"description":"","display_name":"Gemma 4 26B","family":"","file_count":10,"hugging_face_id":"gemma-4-26b-qat-4bit","id":"gemma-4-26b-qat-4bit","input_modalities":["text","image","video"],"max_context_length":131072,"max_output_length":32768,"metadata":{"spec_dec":{"allowed_file_types":["config","weight"],"config_sha256":"0cd54ff36e53a258532c5c1433bc44b88ba758cfe9b59bb4e6eecfd5453fabcf","file_count":2,"manifest_sha256":"8b7c00b7f131345156f5f20fa9c94a895c5340f16d9331bafb9e59628bf45bf2","max_file_count":2,"r2_prefix":"v2/mlx-community-gemma-4-26B-A4B-it-qat-assistant-4bit--2acc3ce46483/bb94eae1b70a80dac16cbf959bb4b7d56bd1fb8c","revision":"bb94eae1b70a80dac16cbf959bb4b7d56bd1fb8c","total_size_bytes":236127665}},"min_ram_gb":36,"model_type":"text","name":"Gemma 4 26B","output_modalities":["text"],"quantization":"4bit","r2_prefix":"v2/gemma-4-26b-qat-4bit--69619d858022/2026-06-08-r1","required_provider_capabilities":[],"runtime_parameters":{},"s3_name":"v2/gemma-4-26b-qat-4bit--69619d858022/2026-06-08-r1","size_gb":15.641239295,"status":"beta","supported_features":["json_mode","reasoning","tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":15641239295,"version":"2026-06-08-r1","weight_hash":"2468a0cb3049a871f42052f4d9f9380bf12a0792f64c7a29f768559fc7d28785"},{"active":true,"aggregate_sha256":"95811153b3bb2ed78bf44b3248b07b52fce637706107de8b0fddf21796ade01c","architecture":"Qwen3.5 MoE VLM with inline MTP","capabilities":["chat","tools","vision"],"created":1787700397,"description":"Qwen3.5 35B-A3B vision-language MoE with affine W4/G64 target and inline MTP modules plus BF16 vision.","display_name":"Qwen3.5 35B A3B","family":"Qwen3.5","file_count":14,"hugging_face_id":"Qwen/Qwen3.5-35B-A3B","id":"qwen3.5-35b-a3b","input_modalities":["text","image"],"max_context_length":262144,"max_output_length":32768,"metadata":{"canary_status":"text_provider_verified","hub_revision":"59d61f3ce65a6d9863b86d2e96597125219dc754","hugging_face_id":"Qwen/Qwen3.5-35B-A3B","manifest_sha256":"95811153b3bb2ed78bf44b3248b07b52fce637706107de8b0fddf21796ade01c","mtp_verification":"structural_only","video_verified":false},"min_ram_gb":36,"model_type":"text","name":"Qwen3.5 35B A3B","output_modalities":["text"],"quantization":"fp4","r2_prefix":"v2/qwen3.5-35b-a3b--f1f11460a4d1/2026-08-25-r1","required_provider_capabilities":[],"runtime_parameters":{"temperature":1,"top_k":20,"top_p":0.95},"s3_name":"v2/qwen3.5-35b-a3b--f1f11460a4d1/2026-08-25-r1","size_gb":20.893747852,"status":"beta","supported_features":["tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":20893747852,"version":"2026-08-25-r1","weight_hash":"95811153b3bb2ed78bf44b3248b07b52fce637706107de8b0fddf21796ade01c"},{"active":true,"aggregate_sha256":"be622ff6ae88533eb31ce984ddc95e5edc3bc52de1767536f2058151383d891a","architecture":"nemotron_h","capabilities":["tools","reasoning"],"created":1789152864,"description":"Nemotron 3.5 Lightning 30B-A3B: NVIDIA hybrid Mamba2, attention, and MoE text model, converted to MLX affine 4-bit (group size 64) with embedded trained MTP weights. MTP activation depends on a compatible provider runtime.","display_name":"Nemotron 3.5 Lightning","family":"nemotron","file_count":10,"hugging_face_artifact":{"repo_id":"EigenLabs/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-MLX-4bit-mtp","revision":"153c469494b3f16a0eaa752dae25cfc9452ce5c6"},"hugging_face_id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","id":"nvidia-nemotron-3.5-lightning","input_modalities":["text"],"max_context_length":262144,"max_output_length":32768,"metadata":{"hugging_face_id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16"},"min_ram_gb":48,"model_type":"text","name":"Nemotron 3.5 Lightning","output_modalities":["text"],"quantization":"4bit","r2_prefix":"v2/nvidia-nemotron-3.5-lightning--ed925578dfaf/2026-09-09-r1","required_provider_capabilities":[],"runtime_parameters":{},"s3_name":"v2/nvidia-nemotron-3.5-lightning--ed925578dfaf/2026-09-09-r1","size_gb":18.544223673,"status":"beta","supported_features":["reasoning","tools"],"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":18544223673,"version":"2026-09-09-r1","weight_hash":"be622ff6ae88533eb31ce984ddc95e5edc3bc52de1767536f2058151383d891a"},{"active":true,"aggregate_sha256":"a4722b6020adb1894c700b45ddcd58bc0e0f033abe7139f86cbbbfe60cba4eb6","architecture":"","capabilities":["chat"],"created":1781549910,"description":"","display_name":"Gemma 4 26B 8-bit (rollback)","family":"","file_count":13,"hugging_face_id":"gemma-4-26b-8bit","id":"gemma-4-26b-8bit","input_modalities":["text"],"max_context_length":131072,"max_output_length":32768,"metadata":{},"min_ram_gb":64,"model_type":"text","name":"Gemma 4 26B 8-bit (rollback)","output_modalities":["text"],"quantization":"8bit","r2_prefix":"v2/gemma-4-26b-8bit--0402fbf4954e/2026-05-25-r1","required_provider_capabilities":[],"runtime_parameters":{},"s3_name":"v2/gemma-4-26b-8bit--0402fbf4954e/2026-05-25-r1","size_gb":27.986040506,"status":"beta","supported_features":null,"supported_sampling_parameters":["temperature","top_p","top_k","frequency_penalty","presence_penalty","repetition_penalty","stop","seed","max_tokens"],"total_size_bytes":27986040506,"version":"2026-05-25-r1","weight_hash":"a4722b6020adb1894c700b45ddcd58bc0e0f033abe7139f86cbbbfe60cba4eb6"}]}