Winnow-12B / release-manifest.json
EldanRing's picture
Add verified NVFP4 and matching MTP assistant assets
3439960 verified
Raw History Blame Contribute Delete
5.7 kB
{
"schema_version": 1,
"model_name": "Winnow-12B",
"repository": "EldanRing/Winnow-12B",
"status": "GGUF release: BF16, Q8_0 and NVFP4 language models, optional F16 vision projector and optional matching12B BF16 GGUF MTP assistant",
"dataset_released": false,
"base_model": "google/gemma-4-12B-it",
"base_revision": "707f0a3b8a3c7ad586ed01e27eafbad8a27dd0f7",
"training": {
"method": "LoRA merged into the weights, exported as BF16/Q8_0 GGUF and quantized as NVFP4 GGUF; quantization adds no training",
"rank": 32,
"alpha": 64,
"dropout": 0.0,
"target_modules": [
"q_proj",
"k_proj",
"v_proj",
"o_proj",
"gate_proj",
"up_proj",
"down_proj"
],
"data_description": "Private mixture of synthetic decision scenarios, teacher-supervised examples and labeled semantic tasks; data and training pipeline are not released."
},
"license": "apache-2.0",
"inference_repository": "https://github.com/EldanRing/winnow-inference",
"conversion_repository": "https://github.com/ggml-org/llama.cpp",
"conversion_revision": "911f6cdc8ab8a530b2bee09ee61471a6f3178eeb",
"calibration": {
"decision_temperature": 1.0,
"fitted_posthoc_map": false
},
"artifacts": [
{
"path": "chat_template.jinja",
"sha256": "ae53464bf3be25802b3a5b37def7fd89667067d7577049b3b2d74c4d8de4c6d4",
"size_bytes": 18683,
"role": "model_asset"
},
{
"path": "config.json",
"sha256": "b50e760a9107b629e1a140ab7a4392e875c37baefee022f27c98b90b9d947cf0",
"size_bytes": 4953,
"role": "model_asset"
},
{
"path": "generation_config.json",
"sha256": "a8349d9bd64cc5841297fcb5002f0fdc4749c473c8f1b10ea337f9ce4ee7014e",
"size_bytes": 260,
"role": "model_asset"
},
{
"path": "processor_config.json",
"sha256": "6b938e76555b3e9946890770e1abcd442a4718f34041a58e8139dc8ad34545c9",
"size_bytes": 1382,
"role": "model_asset"
},
{
"path": "tokenizer.json",
"sha256": "cc8d3a0ce36466ccc1278bf987df5f71db1719b9ca6b4118264f45cb627bfe0f",
"size_bytes": 32169626,
"role": "model_asset"
},
{
"path": "tokenizer_config.json",
"sha256": "a62f4e85a47c0c136edaaa3a4f591fd6783717299a9def47e5ad03a49f6a5eb9",
"size_bytes": 3089,
"role": "model_asset"
},
{
"path": "gguf/Winnow-12B-Q8_0.gguf",
"sha256": "b710efc4c0d048ee61eed92c5fef5ce323a4d17e7c51f9f0533cc72ae50818ea",
"size_bytes": 12669646592,
"role": "q8_0_gguf"
},
{
"path": "gguf/mmproj-Winnow-12B.gguf",
"sha256": "91f086971e56d7a7d8d39e271873fccdb49541bd259d6e02c401a4f1cb7a219e",
"size_bytes": 175115840,
"role": "vision_projector",
"precision": "F16",
"companion_for": [
"gguf/Winnow-12B-BF16.gguf",
"gguf/Winnow-12B-Q8_0.gguf",
"gguf/Winnow-12B-NVFP4.gguf"
]
},
{
"path": "gguf/Winnow-12B-BF16.gguf",
"sha256": "10e6b41e00a3fc6668b53a92fb5d9c85179f47145e0786e7a3bc7b5b09ba8968",
"size_bytes": 23832065792,
"role": "bf16_gguf"
},
{
"path": "gguf/Winnow-12B-NVFP4.gguf",
"sha256": "9aaf25d04b8ab8259be54b153c99e2ab799948256a10abf51b2cd79e665b1f2f",
"size_bytes": 8163448416,
"role": "nvfp4_gguf"
},
{
"path": "gguf/Gemma-4-12B-IT-Assistant-BF16.gguf",
"sha256": "dc630b2f3f3cb8c3170d865d200f414d2da91676ae37c1738e898f7ba92fbeae",
"size_bytes": 861519840,
"role": "mtp_assistant_bf16_gguf",
"source_model": "google/gemma-4-12B-it-assistant",
"source_revision": "46d4c6f13f0ac0ad827b915669b8df9b81c64c51",
"license": "apache-2.0",
"documentation": "docs/assistants/README.md"
}
],
"evaluation_reports": [
"docs/BENCHMARKS.md",
"https://github.com/EldanRing/winnow-inference/blob/main/docs/BENCHMARKS.md"
],
"bf16_benchmark_format": "BF16 GGUF; current download and recorded benchmark SHA256 are compared in bf16_export_verification",
"projector_description": "Optional F16 vision projector for BF16, Q8_0 or NVFP4; a companion file, not another language model.",
"local_test_profile": {
"gpu": "NVIDIA GeForce RTX 5070 Ti",
"gpu_memory_gb": 16,
"host_ram_gb": 96,
"context_positions": 65536,
"kv_cache": "q8_0",
"decision_parallel": 4,
"chat_parallel": 1,
"memory_policy": "exclusive"
},
"inference_revision": "6c2b3c04e248a319f2cb43832628eba03e55fe38",
"weight_format": "GGUF",
"available_model_precisions": [
"BF16",
"Q8_0",
"NVFP4"
],
"bf16_export_verification": {
"source_shards_match_previous_release": true,
"converter_revision": "911f6cdc8ab8a530b2bee09ee61471a6f3178eeb",
"previous_benchmark_sha256": "a7ec058c65b4e73da9f6066a0239b449c74d644c234a5f901aebb19e56de3a22",
"current_sha256": "10e6b41e00a3fc6668b53a92fb5d9c85179f47145e0786e7a3bc7b5b09ba8968",
"matches_previous_benchmark_file": false
},
"previous_model_artifact_revision": "b2b14213dfa252e6d6b543c8b334762e51772d29",
"training_pipeline_released": false,
"optional_inference": {
"repository": "https://github.com/EldanRing/winnow-inference",
"runtime_lock_sha256": "9560b74c9fd736c80880c2733569075e066e7e8f49a7b465c8a4589604c4f65e",
"context": 8192,
"native_parallel": 4,
"chat_parallel": 1,
"cache": "q8_0",
"scope": "Linux/CUDA singleGPU tested RTX5070Ti16GB; experimental adaptive calibration text only",
"supported_presets": [
"12b-nvfp4-vision8k-mtp",
"12b-q8-text8k-mtp"
],
"adaptive_policies": [
"nvfp4-entropy-v1",
"q8-fixed50-v1"
]
}
}