fabryka-english-base-250m-e01 / evaluation /bananamind-base-bench-1.1.json
kacperwikiel's picture
Add verified official BananaMind Base Bench 1.1 results
484393c verified
Raw History Blame Contribute Delete
5.31 kB
{
"created_at": "2026-09-13T00:16:25.063883+00:00",
"benchmark": "BananaMind Base Bench 1.1",
"benchmark_version": "1.1",
"dataset_id": "BananaMind/BananaMind-Base-Bench-1.1",
"dataset_revision": "d4aade51312889e8580963e1ce960c6eaef1a450",
"dataset_sha256": "2f563bb46df778ca494fa20f994a8d3045d4c51fbbffeee433764e2813abea21",
"model": "SlayerLab/fabryka-english-base-250m-e01",
"model_revision": "1e9760c8a1cc0a41f2461d49b5f12d7e4148a2d5",
"tokenizer": "SlayerLab/fabryka-english-base-250m-e01",
"device": "cpu",
"dtype": "float32",
"model_context_length": 2048,
"scoring": {
"task": "four-choice base-text continuation",
"choice_score": "mean conditional token log-probability",
"chat_template": false,
"generation": false,
"add_special_tokens": false,
"add_bos": false
},
"elo": {
"scale": 400.0,
"prior_rating": 1000.0,
"prior_weight": 4.0,
"method": "weighted fixed-item logistic maximum-likelihood rating"
},
"summary": {
"official_complete_run": true,
"cases": 350,
"passed": 106,
"accuracy": 0.3028571428571429,
"weighted_points": 192.5625,
"possible_weighted_points": 668.1875,
"weighted_accuracy": 0.28818632494621643,
"overall_elo": 842,
"overall_elo_unrounded": 841.5314663792717,
"categories": {
"language_completion": {
"cases": 50,
"passed": 30,
"accuracy": 0.6,
"weighted_points": 49.0,
"possible_weighted_points": 78.5,
"weighted_accuracy": 0.6242038216560509,
"elo": 993,
"elo_unrounded": 993.4680335899088
},
"commonsense": {
"cases": 50,
"passed": 15,
"accuracy": 0.3,
"weighted_points": 23.1,
"possible_weighted_points": 87.17500000000001,
"weighted_accuracy": 0.2649842271293375,
"elo": 772,
"elo_unrounded": 772.4284853146141
},
"world_knowledge": {
"cases": 50,
"passed": 15,
"accuracy": 0.3,
"weighted_points": 25.025000000000002,
"possible_weighted_points": 87.72500000000001,
"weighted_accuracy": 0.2852664576802508,
"elo": 791,
"elo_unrounded": 791.3189921625742
},
"context_tracking": {
"cases": 50,
"passed": 11,
"accuracy": 0.22,
"weighted_points": 17.7,
"possible_weighted_points": 94.19999999999999,
"weighted_accuracy": 0.18789808917197454,
"elo": 740,
"elo_unrounded": 740.3310741220985
},
"quantitative": {
"cases": 50,
"passed": 16,
"accuracy": 0.32,
"weighted_points": 34.45,
"possible_weighted_points": 103.025,
"weighted_accuracy": 0.33438485804416407,
"elo": 925,
"elo_unrounded": 925.4521956942147
},
"logical_reasoning": {
"cases": 50,
"passed": 14,
"accuracy": 0.28,
"weighted_points": 31.387500000000003,
"possible_weighted_points": 107.66250000000001,
"weighted_accuracy": 0.29153605015673983,
"elo": 939,
"elo_unrounded": 938.8608126904453
},
"code_completion": {
"cases": 50,
"passed": 5,
"accuracy": 0.1,
"weighted_points": 11.899999999999999,
"possible_weighted_points": 109.89999999999999,
"weighted_accuracy": 0.10828025477707005,
"elo": 729,
"elo_unrounded": 728.9371753478358
}
},
"difficulties": {
"easy": {
"cases": 117,
"passed": 39,
"accuracy": 0.3333333333333333,
"weighted_points": 44.25,
"possible_weighted_points": 141.2,
"weighted_accuracy": 0.3133852691218131,
"elo": 731,
"elo_unrounded": 730.9701408690678
},
"medium": {
"cases": 117,
"passed": 32,
"accuracy": 0.27350427350427353,
"weighted_points": 57.300000000000004,
"possible_weighted_points": 211.875,
"weighted_accuracy": 0.2704424778761062,
"elo": 791,
"elo_unrounded": 790.8983609544082
},
"hard": {
"cases": 116,
"passed": 35,
"accuracy": 0.3017241379310345,
"weighted_points": 91.0125,
"possible_weighted_points": 315.1125,
"weighted_accuracy": 0.2888254194930382,
"elo": 953,
"elo_unrounded": 953.4654883221033
}
},
"context_truncations": 0
},
"verification": {
"complete": true,
"report_sha256": "befd8ed7c3b5d2862a12eca06bcb443a49d415f3386e70cb7f1480283f0b9f2f",
"runner_sha256": "973a81d09d1c4075d031e1369b4278c52a7813d1ab3b11b33eef665d3247bf2c",
"verifier_sha256": "a0d441f3e633abd1b282d8728f435c777bbc9317a3410865a1d9d31598802d40",
"runtime": {
"torch": "2.14.0+cpu",
"transformers": "5.3.0",
"tokenizers": "0.22.2"
},
"checks": [
"350 unique original records and all choices",
"mean-logprob predictions and item weights",
"overall/category/difficulty accuracy and weighted accuracy",
"fixed-item Elo likelihood equation with official prior",
"reported context truncations"
],
"scope": "Result integrity and official-runner aggregation; not an independent model rerun or contamination guarantee."
}
}