{ "schema_version": 1, "model_name": "Winnow-12B", "repository": "EldanRing/Winnow-12B", "status": "GGUF release: BF16, Q8_0 and NVFP4 language models, optional F16 vision projector and optional matching12B BF16 GGUF MTP assistant", "dataset_released": false, "base_model": "google/gemma-4-12B-it", "base_revision": "707f0a3b8a3c7ad586ed01e27eafbad8a27dd0f7", "training": { "method": "LoRA merged into the weights, exported as BF16/Q8_0 GGUF and quantized as NVFP4 GGUF; quantization adds no training", "rank": 32, "alpha": 64, "dropout": 0.0, "target_modules": [ "q_proj", "k_proj", "v_proj", "o_proj", "gate_proj", "up_proj", "down_proj" ], "data_description": "Private mixture of synthetic decision scenarios, teacher-supervised examples and labeled semantic tasks; data and training pipeline are not released." }, "license": "apache-2.0", "inference_repository": "https://github.com/EldanRing/winnow-inference", "conversion_repository": "https://github.com/ggml-org/llama.cpp", "conversion_revision": "911f6cdc8ab8a530b2bee09ee61471a6f3178eeb", "calibration": { "decision_temperature": 1.0, "fitted_posthoc_map": false }, "artifacts": [ { "path": "chat_template.jinja", "sha256": "ae53464bf3be25802b3a5b37def7fd89667067d7577049b3b2d74c4d8de4c6d4", "size_bytes": 18683, "role": "model_asset" }, { "path": "config.json", "sha256": "b50e760a9107b629e1a140ab7a4392e875c37baefee022f27c98b90b9d947cf0", "size_bytes": 4953, "role": "model_asset" }, { "path": "generation_config.json", "sha256": "a8349d9bd64cc5841297fcb5002f0fdc4749c473c8f1b10ea337f9ce4ee7014e", "size_bytes": 260, "role": "model_asset" }, { "path": "processor_config.json", "sha256": "6b938e76555b3e9946890770e1abcd442a4718f34041a58e8139dc8ad34545c9", "size_bytes": 1382, "role": "model_asset" }, { "path": "tokenizer.json", "sha256": "cc8d3a0ce36466ccc1278bf987df5f71db1719b9ca6b4118264f45cb627bfe0f", "size_bytes": 32169626, "role": "model_asset" }, { "path": "tokenizer_config.json", "sha256": "a62f4e85a47c0c136edaaa3a4f591fd6783717299a9def47e5ad03a49f6a5eb9", "size_bytes": 3089, "role": "model_asset" }, { "path": "gguf/Winnow-12B-Q8_0.gguf", "sha256": "b710efc4c0d048ee61eed92c5fef5ce323a4d17e7c51f9f0533cc72ae50818ea", "size_bytes": 12669646592, "role": "q8_0_gguf" }, { "path": "gguf/mmproj-Winnow-12B.gguf", "sha256": "91f086971e56d7a7d8d39e271873fccdb49541bd259d6e02c401a4f1cb7a219e", "size_bytes": 175115840, "role": "vision_projector", "precision": "F16", "companion_for": [ "gguf/Winnow-12B-BF16.gguf", "gguf/Winnow-12B-Q8_0.gguf", "gguf/Winnow-12B-NVFP4.gguf" ] }, { "path": "gguf/Winnow-12B-BF16.gguf", "sha256": "10e6b41e00a3fc6668b53a92fb5d9c85179f47145e0786e7a3bc7b5b09ba8968", "size_bytes": 23832065792, "role": "bf16_gguf" }, { "path": "gguf/Winnow-12B-NVFP4.gguf", "sha256": "9aaf25d04b8ab8259be54b153c99e2ab799948256a10abf51b2cd79e665b1f2f", "size_bytes": 8163448416, "role": "nvfp4_gguf" }, { "path": "gguf/Gemma-4-12B-IT-Assistant-BF16.gguf", "sha256": "dc630b2f3f3cb8c3170d865d200f414d2da91676ae37c1738e898f7ba92fbeae", "size_bytes": 861519840, "role": "mtp_assistant_bf16_gguf", "source_model": "google/gemma-4-12B-it-assistant", "source_revision": "46d4c6f13f0ac0ad827b915669b8df9b81c64c51", "license": "apache-2.0", "documentation": "docs/assistants/README.md" } ], "evaluation_reports": [ "docs/BENCHMARKS.md", "https://github.com/EldanRing/winnow-inference/blob/main/docs/BENCHMARKS.md" ], "bf16_benchmark_format": "BF16 GGUF; current download and recorded benchmark SHA256 are compared in bf16_export_verification", "projector_description": "Optional F16 vision projector for BF16, Q8_0 or NVFP4; a companion file, not another language model.", "local_test_profile": { "gpu": "NVIDIA GeForce RTX 5070 Ti", "gpu_memory_gb": 16, "host_ram_gb": 96, "context_positions": 65536, "kv_cache": "q8_0", "decision_parallel": 4, "chat_parallel": 1, "memory_policy": "exclusive" }, "inference_revision": "6c2b3c04e248a319f2cb43832628eba03e55fe38", "weight_format": "GGUF", "available_model_precisions": [ "BF16", "Q8_0", "NVFP4" ], "bf16_export_verification": { "source_shards_match_previous_release": true, "converter_revision": "911f6cdc8ab8a530b2bee09ee61471a6f3178eeb", "previous_benchmark_sha256": "a7ec058c65b4e73da9f6066a0239b449c74d644c234a5f901aebb19e56de3a22", "current_sha256": "10e6b41e00a3fc6668b53a92fb5d9c85179f47145e0786e7a3bc7b5b09ba8968", "matches_previous_benchmark_file": false }, "previous_model_artifact_revision": "b2b14213dfa252e6d6b543c8b334762e51772d29", "training_pipeline_released": false, "optional_inference": { "repository": "https://github.com/EldanRing/winnow-inference", "runtime_lock_sha256": "9560b74c9fd736c80880c2733569075e066e7e8f49a7b465c8a4589604c4f65e", "context": 8192, "native_parallel": 4, "chat_parallel": 1, "cache": "q8_0", "scope": "Linux/CUDA singleGPU tested RTX5070Ti16GB; experimental adaptive calibration text only", "supported_presets": [ "12b-nvfp4-vision8k-mtp", "12b-q8-text8k-mtp" ], "adaptive_policies": [ "nvfp4-entropy-v1", "q8-fixed50-v1" ] } }