grug-27b-v2 / gguf_conversion.json
ProCreations's picture
Release validated Grug 27B v2 with integrated MTP and measured results
f572b85 verified
Raw History Blame Contribute Delete
5.22 kB
{
"conversion_transformers": "5.16.1",
"source_model": "ProCreations/grug-27b-v2-candidate-20260912-qwen",
"source_revision": "79a2b2b81bf6ee0123da02467cf187db04e03708",
"llama_cpp_commit": "2a3005c23f60cb38dab70b8ea2ddbd969bcf3e87",
"mtp_embedded_in_every_text_gguf": true,
"artifacts": [
{
"file": "grug-27b-v2-Q4_K_M.gguf",
"bytes": 16998721440,
"sha256": "5db0549d718c0accbe5979af0a959099f7165e3b0bea7a8604c7a1bd6872363f",
"mtp_tensor_count": 15,
"mtp_matrix_quantization": "Q8_0",
"smoke_stdout": "| model | size | params | backend | threads | test | t/s |\n| ------------------------------ | ---------: | ---------: | ---------- | ------: | --------------: | -------------------: |\n| qwen35 27B Q4_K - Medium | 15.82 GiB | 27.32 B | CPU | 24 | pp16 | 21.50 ± 0.00 |\n| qwen35 27B Q4_K - Medium | 15.82 GiB | 27.32 B | CPU | 24 | tg8 | 3.16 ± 0.00 |\n\nbuild: 2a3005c (1)\n",
"smoke_stderr": ""
},
{
"file": "grug-27b-v2-Q5_K_M.gguf",
"bytes": 19682420640,
"sha256": "bf02d822be68618f9dca224065aa47e0cbb478de8b432a0ccc672a28d06438f8",
"mtp_tensor_count": 15,
"mtp_matrix_quantization": "Q8_0",
"smoke_stdout": "| model | size | params | backend | threads | test | t/s |\n| ------------------------------ | ---------: | ---------: | ---------- | ------: | --------------: | -------------------: |\n| qwen35 27B Q5_K - Medium | 18.32 GiB | 27.32 B | CPU | 24 | pp16 | 13.27 ± 0.00 |\n| qwen35 27B Q5_K - Medium | 18.32 GiB | 27.32 B | CPU | 24 | tg8 | 2.81 ± 0.00 |\n\nbuild: 2a3005c (1)\n",
"smoke_stderr": ""
},
{
"file": "grug-27b-v2-Q6_K.gguf",
"bytes": 22533851040,
"sha256": "c110ad0b8863c06dacd753c41c301ba3d7c4901e5776d0d80e2396b6c184fd0c",
"mtp_tensor_count": 15,
"mtp_matrix_quantization": "Q8_0",
"smoke_stdout": "| model | size | params | backend | threads | test | t/s |\n| ------------------------------ | ---------: | ---------: | ---------- | ------: | --------------: | -------------------: |\n| qwen35 27B Q6_K | 20.98 GiB | 27.32 B | CPU | 24 | pp16 | 15.91 ± 0.00 |\n| qwen35 27B Q6_K | 20.98 GiB | 27.32 B | CPU | 24 | tg8 | 2.53 ± 0.00 |\n\nbuild: 2a3005c (1)\n",
"smoke_stderr": ""
},
{
"file": "grug-27b-v2-Q8_0.gguf",
"bytes": 29047084960,
"sha256": "01431cb3864ee4cd7ef5a9e5d73079194f6c3bca66d0cac5ce19502e2e948f44",
"mtp_tensor_count": 15,
"mtp_matrix_quantization": "Q8_0",
"smoke_stdout": "| model | size | params | backend | threads | test | t/s |\n| ------------------------------ | ---------: | ---------: | ---------- | ------: | --------------: | -------------------: |\n| qwen35 27B Q8_0 | 27.04 GiB | 27.32 B | CPU | 24 | pp16 | 17.03 ± 0.00 |\n| qwen35 27B Q8_0 | 27.04 GiB | 27.32 B | CPU | 24 | tg8 | 1.92 ± 0.00 |\n\nbuild: 2a3005c (1)\n",
"smoke_stderr": ""
},
{
"file": "grug-27b-v2-Q3_K_M.gguf",
"bytes": 13752764320,
"sha256": "3b36b201c4c7053bea301533d2804aa41912f701c19e320bffe0cfa5e72e07c4",
"mtp_tensor_count": 15,
"mtp_matrix_quantization": "Q8_0",
"smoke_stdout": "| model | size | params | backend | threads | test | t/s |\n| ------------------------------ | ---------: | ---------: | ---------- | ------: | --------------: | -------------------: |\n| qwen35 27B Q3_K - Medium | 12.80 GiB | 27.32 B | CPU | 24 | pp16 | 17.12 ± 0.00 |\n| qwen35 27B Q3_K - Medium | 12.80 GiB | 27.32 B | CPU | 24 | tg8 | 4.01 ± 0.00 |\n\nbuild: 2a3005c (1)\n",
"smoke_stderr": ""
},
{
"file": "mmproj-grug-27b-v2-F16.gguf",
"bytes": 927606976,
"sha256": "1335119dd8d9f9e8c4b0a65de12b7d218cfc47282c706a0c6be0dd85be209da6"
}
],
"elapsed_seconds": 1267.8098618984222,
"conversion_time_remaining_validation": "GPU generation and embedded MTP parity must pass before public release",
"post_conversion_validation": {
"completed": true,
"record": "results/runtime_validation.json",
"basic_functional_checks": "12/12 for every quantization with MTP off and on",
"native_api_checks": "14/15 for every mode; reasoning alias is unsupported by the pinned llama-server parser",
"token_parity": "Measured and reported per quantization; not a bit-identical decoding guarantee",
"coding_grader": "Corrected final-answer replay on HF CPU Jobs; see results/coding_rescore.json"
}
}