{ "version": 2, "models": [ { "id": "qwen3.5-2b-q4", "name": "Qwen 3.5 2B", "family": "qwen", "gguf_file": "Qwen3.5-2B-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/Qwen3.5-2B-GGUF/resolve/main/Qwen3.5-2B-Q4_K_M.gguf", "gguf_sha256": "aaf42c8b7c3cab2bf3d69c355048d4a0ee9973d48f16c731c0520ee914699223", "size_mb": 1500, "vram_required_gb": 3, "context_length": 8192, "quantization": "Q4_K_M", "specialty": "Fast", "description": "Lightweight model for quick responses. Best for simple tasks and low-resource systems.", "tokens_per_sec_estimate": 120, "llm_model_name": "qwen3.5-2b", "llama_server_image": null }, { "id": "phi4-mini-q4", "name": "Phi-4 Mini", "family": "phi", "gguf_file": "Phi-4-mini-instruct-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/Phi-4-mini-instruct-GGUF/resolve/main/Phi-4-mini-instruct-Q4_K_M.gguf", "gguf_sha256": "88c00229914083cd112853aab84ed51b87bdf6b9ce42f532d8c85c7c63b1730a", "size_mb": 2490, "vram_required_gb": 4, "context_length": 128000, "quantization": "Q4_K_M", "specialty": "Balanced", "description": "Microsoft's compact model with 128K context. Punches above its weight on reasoning and code.", "tokens_per_sec_estimate": 130, "llm_model_name": "phi-4-mini", "llama_server_image": null }, { "id": "qwen3.5-4b-q4", "name": "Qwen 3.5 4B", "family": "qwen", "gguf_file": "Qwen3.5-4B-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/Qwen3.5-4B-GGUF/resolve/main/Qwen3.5-4B-Q4_K_M.gguf", "gguf_sha256": "00fe7986ff5f6b463e62455821146049db6f9313603938a70800d1fb69ef11a4", "size_mb": 2870, "vram_required_gb": 5, "context_length": 16384, "quantization": "Q4_K_M", "specialty": "Balanced", "description": "Good balance of speed and capability for Intel Arc and budget GPUs.", "tokens_per_sec_estimate": 100, "llm_model_name": "qwen3.5-4b", "llama_server_image": null }, { "id": "gemma4-e2b-q4", "name": "Gemma 4 E2B", "family": "gemma4", "gguf_file": "gemma-4-E2B-it-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/gemma-4-E2B-it-GGUF/resolve/main/gemma-4-E2B-it-Q4_K_M.gguf", "gguf_sha256": "ac0069ebccd39925d836f24a88c0f0c858d20578c29b21ab7cedce66ee576845", "size_mb": 2810, "vram_required_gb": 5, "context_length": 16384, "quantization": "Q4_K_M", "specialty": "Fast", "description": "Google's efficient small model. Good for quick tasks on budget hardware.", "tokens_per_sec_estimate": 110, "llm_model_name": "gemma-4-e2b-it", "llama_server_image": "ghcr.io/ggml-org/llama.cpp:server-cuda-b9014" }, { "id": "deepseek-r1-7b-q4", "name": "DeepSeek R1 7B", "family": "deepseek", "gguf_file": "DeepSeek-R1-Distill-Qwen-7B-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/DeepSeek-R1-Distill-Qwen-7B-GGUF/resolve/main/DeepSeek-R1-Distill-Qwen-7B-Q4_K_M.gguf", "gguf_sha256": "78272d8d32084548bd450394a560eb2d70de8232ab96a725769b1f9171235c1c", "size_mb": 4680, "vram_required_gb": 7, "context_length": 32768, "quantization": "Q4_K_M", "specialty": "Reasoning", "description": "DeepSeek R1 distilled into Qwen 7B. Chain-of-thought reasoning on a budget GPU.", "tokens_per_sec_estimate": 80, "llm_model_name": "deepseek-r1-distill-qwen-7b", "llama_server_image": null }, { "id": "gemma4-e4b-q4", "name": "Gemma 4 E4B", "family": "gemma4", "gguf_file": "gemma-4-E4B-it-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/gemma-4-E4B-it-GGUF/resolve/main/gemma-4-E4B-it-Q4_K_M.gguf", "gguf_sha256": "dff0ffba4c90b4082d70214d53ce9504a28d4d8d998276dcb3b8881a656c742a", "size_mb": 5340, "vram_required_gb": 8, "context_length": 32768, "quantization": "Q4_K_M", "specialty": "General", "description": "Google's mid-range model with strong multilingual and reasoning capabilities.", "tokens_per_sec_estimate": 85, "llm_model_name": "gemma-4-e4b-it", "llama_server_image": "ghcr.io/ggml-org/llama.cpp:server-cuda-b9014" }, { "id": "qwen3.5-9b-q4", "name": "Qwen 3.5 9B", "family": "qwen", "gguf_file": "Qwen3.5-9B-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/Qwen3.5-9B-GGUF/resolve/main/Qwen3.5-9B-Q4_K_M.gguf", "gguf_sha256": "03b74727a860a56338e042c4420bb3f04b2fec5734175f4cb9fa853daf52b7e8", "size_mb": 5760, "vram_required_gb": 8, "context_length": 32768, "quantization": "Q4_K_M", "specialty": "General", "description": "Excellent general-purpose model. Default for most systems. Great at coding, reasoning, and multilingual tasks.", "tokens_per_sec_estimate": 90, "llm_model_name": "qwen3.5-9b", "llama_server_image": null }, { "id": "phi4-q4", "name": "Phi-4 14B", "family": "phi", "gguf_file": "phi-4-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/bartowski/phi-4-GGUF/resolve/main/phi-4-Q4_K_M.gguf", "gguf_sha256": "009aba717c09d4a35890c7d35eb59d54e1dba884c7c526e7197d9c13ab5911d9", "size_mb": 9050, "vram_required_gb": 11, "context_length": 16384, "quantization": "Q4_K_M", "specialty": "General", "description": "Microsoft's MIT-licensed 14B model. Strong at STEM, coding, and structured reasoning.", "tokens_per_sec_estimate": 35, "llm_model_name": "phi-4", "llama_server_image": null }, { "id": "deepseek-r1-14b-q4", "name": "DeepSeek R1 14B", "family": "deepseek", "gguf_file": "DeepSeek-R1-Distill-Qwen-14B-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/DeepSeek-R1-Distill-Qwen-14B-GGUF/resolve/main/DeepSeek-R1-Distill-Qwen-14B-Q4_K_M.gguf", "gguf_sha256": "67a7933cf2ad596a393c8e13b30bc4da2d50b283e250b78554aed18817eca31c", "size_mb": 8990, "vram_required_gb": 12, "context_length": 32768, "quantization": "Q4_K_M", "specialty": "Reasoning", "description": "DeepSeek R1 distilled into Qwen 14B. Strong chain-of-thought reasoning for mid-range GPUs.", "tokens_per_sec_estimate": 50, "llm_model_name": "deepseek-r1-distill-qwen-14b", "llama_server_image": null }, { "id": "qwen3.5-27b-q4", "name": "Qwen 3.5 27B", "family": "qwen", "gguf_file": "Qwen3.5-27B-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/Qwen3.5-27B-GGUF/resolve/main/Qwen3.5-27B-Q4_K_M.gguf", "gguf_sha256": "84b5f7f112156d63836a01a69dc3f11a6ba63b10a23b8ca7a7efaf52d5a2d806", "size_mb": 16700, "vram_required_gb": 20, "context_length": 32768, "quantization": "Q4_K_M", "specialty": "Quality", "description": "Dense 27B powerhouse. Top-tier quality for 24GB GPUs. Excels at complex reasoning and coding.", "tokens_per_sec_estimate": 25, "llm_model_name": "qwen3.5-27b", "llama_server_image": null }, { "id": "gemma4-26b-a4b-q4", "name": "Gemma 4 26B-A4B", "family": "gemma4", "gguf_file": "gemma-4-26B-A4B-it-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/ggml-org/gemma-4-26B-A4B-it-GGUF/resolve/main/gemma-4-26B-A4B-it-Q4_K_M.gguf", "gguf_sha256": "23c6997912cb7fa36147fe05877de73ddbb2a80ff69b18ff171b354dccf2b5b5", "size_mb": 18000, "vram_required_gb": 22, "context_length": 16384, "quantization": "Q4_K_M", "specialty": "Reasoning", "description": "Google's large MoE model. Strong at reasoning and analysis tasks.", "tokens_per_sec_estimate": 50, "llm_model_name": "gemma-4-26b-a4b-it", "llama_server_image": "ghcr.io/ggml-org/llama.cpp:server-cuda-b9014" }, { "id": "qwen3-30b-a3b-q4", "name": "Qwen 3 30B-A3B", "family": "qwen", "gguf_file": "Qwen3-30B-A3B-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/Qwen3-30B-A3B-GGUF/resolve/main/Qwen3-30B-A3B-Q4_K_M.gguf", "gguf_sha256": "9f1a24700a339b09c06009b729b5c809e0b64c213b8af5b711b3dbdfd0c5ba48", "size_mb": 18600, "vram_required_gb": 22, "context_length": 131072, "quantization": "Q4_K_M", "specialty": "Quality", "description": "High-quality MoE model with 131K context. Excellent for complex reasoning and long documents.", "tokens_per_sec_estimate": 55, "llm_model_name": "qwen3-30b-a3b", "llama_server_image": null }, { "id": "gemma4-31b-q4", "name": "Gemma 4 31B", "family": "gemma4", "gguf_file": "gemma-4-31B-it-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/ggml-org/gemma-4-31B-it-GGUF/resolve/main/gemma-4-31B-it-Q4_K_M.gguf", "gguf_sha256": "a20deaf2f8fc27c501f32fadfd538f8f31a76f10f47d5d3eb895f3d1112d752c", "size_mb": 19800, "vram_required_gb": 24, "context_length": 131072, "quantization": "Q4_K_M", "specialty": "Quality", "description": "Google's flagship dense model with 131K context. Top-tier quality for enterprise GPUs.", "tokens_per_sec_estimate": 40, "llm_model_name": "gemma-4-31b-it", "llama_server_image": "ghcr.io/ggml-org/llama.cpp:server-cuda-b9014" }, { "id": "deepseek-r1-32b-q4", "name": "DeepSeek R1 32B", "family": "deepseek", "gguf_file": "DeepSeek-R1-Distill-Qwen-32B-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/DeepSeek-R1-Distill-Qwen-32B-GGUF/resolve/main/DeepSeek-R1-Distill-Qwen-32B-Q4_K_M.gguf", "gguf_sha256": "ca171ca03554ee20cf67ad6b540610ae7eabb95af00c0abd36bb73542e140fb5", "size_mb": 19900, "vram_required_gb": 24, "context_length": 32768, "quantization": "Q4_K_M", "specialty": "Reasoning", "description": "DeepSeek R1 distilled into Qwen 32B. The best open-source reasoning model at this size.", "tokens_per_sec_estimate": 25, "llm_model_name": "deepseek-r1-distill-qwen-32b", "llama_server_image": null }, { "id": "qwen3.5-35b-a3b-q4", "name": "Qwen 3.5 35B-A3B", "family": "qwen", "gguf_file": "Qwen3.5-35B-A3B-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/Qwen3.5-35B-A3B-GGUF/resolve/main/Qwen3.5-35B-A3B-Q4_K_M.gguf", "gguf_sha256": "3b46d1066bc91cc2d613e3bc22ce691dd77e6f0d33c9060690d24ce6de494375", "size_mb": 22000, "vram_required_gb": 24, "context_length": 131072, "quantization": "Q4_K_M", "specialty": "Quality", "description": "Successor to Qwen3 30B-A3B. MoE with 131K context — top scores on coding, math, and reasoning benchmarks.", "tokens_per_sec_estimate": 50, "llm_model_name": "qwen3.5-35b-a3b", "llama_server_image": null }, { "id": "qwen3.6-35b-a3b-ud-q4", "name": "Qwen 3.6 35B-A3B", "family": "qwen", "gguf_file": "Qwen3.6-35B-A3B-UD-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/Qwen3.6-35B-A3B-GGUF/resolve/main/Qwen3.6-35B-A3B-UD-Q4_K_M.gguf", "gguf_sha256": "ac0e2c1189e055faa36eff361580e79c5bd6f8e76bffb4ce547f167d53e31a61", "size_mb": 21110, "vram_required_gb": 24, "context_length": 131072, "quantization": "UD-Q4_K_M", "specialty": "Quality", "description": "Current A3B MoE successor for Spark-class unified-memory hosts. Verified on DGX Spark with llama.cpp b9014.", "tokens_per_sec_estimate": 59, "llm_model_name": "qwen3.6-35b-a3b", "llama_server_image": null }, { "id": "deepseek-r1-70b-q4", "name": "DeepSeek R1 70B", "family": "deepseek", "gguf_file": "DeepSeek-R1-Distill-Llama-70B-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/DeepSeek-R1-Distill-Llama-70B-GGUF/resolve/main/DeepSeek-R1-Distill-Llama-70B-Q4_K_M.gguf", "gguf_sha256": "952ff479c48ac3ece81fb6d9a5a03bfeac215e6caac780fbe91ff8cb0e05bcf3", "size_mb": 42500, "vram_required_gb": 48, "context_length": 32768, "quantization": "Q4_K_M", "specialty": "Reasoning", "description": "The largest DeepSeek R1 distill. State-of-the-art reasoning for dual-GPU or 48GB+ setups.", "tokens_per_sec_estimate": 15, "llm_model_name": "deepseek-r1-distill-llama-70b", "llama_server_image": null }, { "id": "qwen3-coder-next-q4", "name": "Qwen 3 Coder Next", "family": "qwen", "gguf_file": "qwen3-coder-next-Q4_K_M.gguf", "gguf_url": "https://huggingface.co/unsloth/Qwen3-Coder-Next-GGUF/resolve/main/Qwen3-Coder-Next-Q4_K_M.gguf", "gguf_sha256": "9e6032d2f3b50a60f17ce8bf5a1d85c71af9b53b89c7978020ae7c660f29b090", "size_mb": 48500, "vram_required_gb": 52, "context_length": 131072, "quantization": "Q4_K_M", "specialty": "Code", "description": "Flagship coding model for enterprise GPUs and Strix Halo. 131K context for large codebases.", "tokens_per_sec_estimate": 30, "llm_model_name": "qwen3-coder-next", "llama_server_image": null }, { "id": "llama4-scout-q4", "name": "Llama 4 Scout", "family": "llama4", "gguf_file": "Llama-4-Scout-17B-16E-Instruct-Q4_K_M-00001-of-00002.gguf", "gguf_url": "", "gguf_sha256": "", "gguf_parts": [ { "file": "Llama-4-Scout-17B-16E-Instruct-Q4_K_M-00001-of-00002.gguf", "url": "https://huggingface.co/unsloth/Llama-4-Scout-17B-16E-Instruct-GGUF/resolve/main/Q4_K_M/Llama-4-Scout-17B-16E-Instruct-Q4_K_M-00001-of-00002.gguf", "sha256": "fbe956902467171ed7c0c326e5d868771a84d46468d407abecd0f289297313f9" }, { "file": "Llama-4-Scout-17B-16E-Instruct-Q4_K_M-00002-of-00002.gguf", "url": "https://huggingface.co/unsloth/Llama-4-Scout-17B-16E-Instruct-GGUF/resolve/main/Q4_K_M/Llama-4-Scout-17B-16E-Instruct-Q4_K_M-00002-of-00002.gguf", "sha256": "e7330ae14527f08e8a44a6a573b45084052b8822c4b8a5179c5cad9cd6f6f795" } ], "size_mb": 65300, "vram_required_gb": 70, "context_length": 131072, "quantization": "Q4_K_M", "specialty": "General", "description": "Meta's flagship MoE with 16 experts. 131K context, strong multimodal capabilities. Needs 80GB+ VRAM.", "tokens_per_sec_estimate": 20, "llm_model_name": "llama-4-scout", "llama_server_image": null }, { "id": "qwen3.5-122b-a10b-q4", "name": "Qwen 3.5 122B-A10B", "family": "qwen", "gguf_file": "Qwen3.5-122B-A10B-Q4_K_M-00001-of-00003.gguf", "gguf_url": "", "gguf_sha256": "", "gguf_parts": [ { "file": "Qwen3.5-122B-A10B-Q4_K_M-00001-of-00003.gguf", "url": "https://huggingface.co/unsloth/Qwen3.5-122B-A10B-GGUF/resolve/main/Q4_K_M/Qwen3.5-122B-A10B-Q4_K_M-00001-of-00003.gguf", "sha256": "467c9bd92ea518539cf75bf5a5fbfbd35e9a0b40d766ccaa67bf120e12041df3" }, { "file": "Qwen3.5-122B-A10B-Q4_K_M-00002-of-00003.gguf", "url": "https://huggingface.co/unsloth/Qwen3.5-122B-A10B-GGUF/resolve/main/Q4_K_M/Qwen3.5-122B-A10B-Q4_K_M-00002-of-00003.gguf", "sha256": "90db14846413aebdac365b57206441437cac5f7e5037d94b325f0167f902e6e7" }, { "file": "Qwen3.5-122B-A10B-Q4_K_M-00003-of-00003.gguf", "url": "https://huggingface.co/unsloth/Qwen3.5-122B-A10B-GGUF/resolve/main/Q4_K_M/Qwen3.5-122B-A10B-Q4_K_M-00003-of-00003.gguf", "sha256": "e3c24b8ebec070bb4f69ea0aca25a16531da7440cd515529953e046882901f97" } ], "size_mb": 76500, "vram_required_gb": 80, "context_length": 131072, "quantization": "Q4_K_M", "specialty": "Quality", "description": "The largest Qwen 3.5 MoE. 122B params, 10B active. Top-tier quality for multi-GPU or 96GB+ unified memory systems.", "tokens_per_sec_estimate": 15, "llm_model_name": "qwen3.5-122b-a10b", "llama_server_image": null } ] }