Other
Transformers
TensorBoard
Safetensors
English
diffusion_lm
fill-mask
custom_code
tiny-llm-ablation
from-scratch
diffusion
masked-language-modeling
Eval Results (legacy)
Instructions to use d0rj/diffusion-51M-base with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use d0rj/diffusion-51M-base with Transformers:
# pip install -U transformers accelerate # Load model directly from transformers import AutoModelForMaskedLM model = AutoModelForMaskedLM.from_pretrained("d0rj/diffusion-51M-base", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download evaluation/comparison-20261001/results.json from d0rj/diffusion-51M-base: direct link, hf CLI and curl.
- Browser
- Download file 25.9 kB
-
https://huggingface.co/d0rj/diffusion-51M-base/resolve/main/evaluation/comparison-20261001/results.json
- Command line
-
hf download hf://d0rj/diffusion-51M-base/evaluation/comparison-20261001/results.json
-
curl -L -o results.json https://huggingface.co/d0rj/diffusion-51M-base/resolve/main/evaluation/comparison-20261001/results.json
25.9 kB
| { | |
| "date": "2026-10-01", | |
| "model": "d0rj/diffusion-51M-base", | |
| "protocol": "Full selected splits; lm-eval 0.4.12; seed 1234; BF16 on RTX 5070 Ti;\ncontext cap 2048 (ArithMark 1024), TF32 disabled, no chat template and no added\nfew-shot examples. TruthfulQA retains the harness's fixed six-QA preamble.\nArithMark and BananaMind normalize by continuation token count; ordinary harness\nacc_norm uses its own length normalization. BananaMind is raw accuracy, not Elo.\nSciQ includes the support passage. Balanced COPA uses the mirrored 1000-item\ntrain-named evaluation split; cRia's split was inferred, not confirmed.\nMMLU scores full answer continuations across 57 subjects, weighted by item count;\nBLiMP averages 67 equal-sized minimal-pair subsets. Standard errors are retained\nfrom each evaluator. Wilson intervals are reported only where the runner logged\nbinary item accuracy; MC2 is probability mass, not binary accuracy. These\nintervals do not model dependence between paired/templated examples or training\nseed variation. UL2, PrefixLM and experimental diffusion PLL use their documented\nconditional scoring protocols; PLL exposes the other answer tokens and is not\nautoregressive likelihood. cRia's published scores used a different precision\nand benchmark-adapted checkpoint; this completes our comparison coverage, not\nan independent reproduction of cRia or an official leaderboard submission.", | |
| "results": { | |
| "checkpoint": "/mnt/d/Projects/tiny_llm/runs/eval/diffusion-v2-step15000-20260923", | |
| "checkpoint_sha256": { | |
| "config.json": "8e3daed6864d9b1f39be95c7b311a93c47efd4e76ee33fb4dbeede6d4701343d", | |
| "model.safetensors": "d0d24e0cc492cd5c9a613710c454af9a6e49265d2bae7abb69a673a53efe2bfb" | |
| }, | |
| "core": "/mnt/d/Projects/tiny_llm/runs/eval/diffusion-v2-core-full-20260923", | |
| "training_run": "diffusion-gpu-v2", | |
| "training_step": 15000, | |
| "hf": "d0rj/diffusion-51M-base", | |
| "tokenizer": "/mnt/d/Projects/tiny_llm/tokenizer", | |
| "tb": "runs/diffusion-gpu-v2/tb", | |
| "protocol": "EXPERIMENTAL: continuation PLL; single-mask left-to-right greedy generation; NOT AR likelihood", | |
| "kind": "diffusion", | |
| "tasks": { | |
| "arithmark3": { | |
| "label": "ArithMark-3", | |
| "dataset": "AxiomicLabs/Arithmark-3.0", | |
| "config": "default", | |
| "split": "train", | |
| "samples": 1000, | |
| "shots": 0, | |
| "primary": "acc_norm", | |
| "metrics": { | |
| "acc": { | |
| "value": 0.332, | |
| "stderr": 0.014899597242811565, | |
| "ci95": [ | |
| 0.30350363489239063, | |
| 0.36178215595875596 | |
| ], | |
| "ci_method": "Wilson 95%; item independence approximation" | |
| }, | |
| "acc_norm": { | |
| "value": 0.332, | |
| "stderr": 0.014899597242811565, | |
| "ci95": [ | |
| 0.30350363489239063, | |
| 0.36178215595875596 | |
| ], | |
| "ci_method": "Wilson 95%; item independence approximation" | |
| } | |
| }, | |
| "result_path": "runs/eval/cria-backfill-20261001/diffusion-v2/attempt-001/arithmark3.json", | |
| "result_sha256": "68430957620991dec536d3717f684c1b92d6921b9e324940f542ceb62109ec0f", | |
| "categories": { | |
| "elementary_school_math_continuation::addition::grades_1_2::easy": { | |
| "count": 128, | |
| "acc_norm,none": 0.21875, | |
| "acc_norm_stderr,none": 0.03668319762764659 | |
| }, | |
| "elementary_school_math_continuation::comparison::grades_2_3::medium": { | |
| "count": 44, | |
| "acc_norm,none": 0.18181818181818182, | |
| "acc_norm_stderr,none": 0.05881787629278459 | |
| }, | |
| "elementary_school_math_continuation::comparison_difference::grades_2_3::medium": { | |
| "count": 48, | |
| "acc_norm,none": 0.2708333333333333, | |
| "acc_norm_stderr,none": 0.06482097094483913 | |
| }, | |
| "elementary_school_math_continuation::data::grades_2_3::easy": { | |
| "count": 43, | |
| "acc_norm,none": 0.16279069767441862, | |
| "acc_norm_stderr,none": 0.056964877739143674 | |
| }, | |
| "elementary_school_math_continuation::division::grades_3_4::medium": { | |
| "count": 54, | |
| "acc_norm,none": 0.37037037037037035, | |
| "acc_norm_stderr,none": 0.06633194954624339 | |
| }, | |
| "elementary_school_math_continuation::fractions_counting::grades_3_4::medium": { | |
| "count": 50, | |
| "acc_norm,none": 0.08, | |
| "acc_norm_stderr,none": 0.03875617133214439 | |
| }, | |
| "elementary_school_math_continuation::geometry_area::grades_4_5::medium": { | |
| "count": 52, | |
| "acc_norm,none": 0.46153846153846156, | |
| "acc_norm_stderr,none": 0.06980655484407924 | |
| }, | |
| "elementary_school_math_continuation::geometry_perimeter::grades_4_5::medium": { | |
| "count": 45, | |
| "acc_norm,none": 0.4666666666666667, | |
| "acc_norm_stderr,none": 0.0752101433090355 | |
| }, | |
| "elementary_school_math_continuation::measurement::grades_2_3::easy": { | |
| "count": 76, | |
| "acc_norm,none": 0.32894736842105265, | |
| "acc_norm_stderr,none": 0.05425138981075869 | |
| }, | |
| "elementary_school_math_continuation::money::grades_3_4::medium": { | |
| "count": 64, | |
| "acc_norm,none": 0.375, | |
| "acc_norm_stderr,none": 0.060993754559283325 | |
| }, | |
| "elementary_school_math_continuation::multiplication::grades_3_4::medium": { | |
| "count": 74, | |
| "acc_norm,none": 0.5675675675675675, | |
| "acc_norm_stderr,none": 0.057983774751431 | |
| }, | |
| "elementary_school_math_continuation::patterns::grades_3_4::medium": { | |
| "count": 53, | |
| "acc_norm,none": 0.16981132075471697, | |
| "acc_norm_stderr,none": 0.05206789873629056 | |
| }, | |
| "elementary_school_math_continuation::subtraction::grades_1_2::easy": { | |
| "count": 117, | |
| "acc_norm,none": 0.23931623931623933, | |
| "acc_norm_stderr,none": 0.03961495460787807 | |
| }, | |
| "elementary_school_math_continuation::time::grades_2_3::easy": { | |
| "count": 55, | |
| "acc_norm,none": 0.9818181818181818, | |
| "acc_norm_stderr,none": 0.018181818181818184 | |
| }, | |
| "elementary_school_math_continuation::two_step_add_subtract::grades_2_3::medium": { | |
| "count": 46, | |
| "acc_norm,none": 0.21739130434782608, | |
| "acc_norm_stderr,none": 0.06148754619013457 | |
| }, | |
| "elementary_school_math_continuation::two_step_addition::grades_2_3::medium": { | |
| "count": 19, | |
| "acc_norm,none": 0.3157894736842105, | |
| "acc_norm_stderr,none": 0.10956136839295436 | |
| }, | |
| "elementary_school_math_continuation::two_step_subtraction::grades_2_3::medium": { | |
| "count": 32, | |
| "acc_norm,none": 0.28125, | |
| "acc_norm_stderr,none": 0.08075219711382271 | |
| } | |
| } | |
| }, | |
| "balanced_copa": { | |
| "label": "Balanced COPA", | |
| "dataset": "pkavumba/balanced-copa", | |
| "config": "default", | |
| "split": "train", | |
| "samples": 1000, | |
| "shots": 0, | |
| "primary": "acc", | |
| "metrics": { | |
| "acc": { | |
| "value": 0.522, | |
| "stderr": 0.01580397942816194, | |
| "ci95": [ | |
| 0.4910152521271681, | |
| 0.5528163704994674 | |
| ], | |
| "ci_method": "Wilson 95%; item independence approximation" | |
| } | |
| }, | |
| "result_path": "runs/eval/cria-backfill-20261001/diffusion-v2/attempt-001/balanced_copa.json", | |
| "result_sha256": "1375e8cff071a9464a7c0eded68c0afd7b944bb716d680493dfbe3301963b482", | |
| "categories": {} | |
| }, | |
| "commonsense_qa": { | |
| "label": "CommonsenseQA", | |
| "dataset": "tau/commonsense_qa", | |
| "config": "default", | |
| "split": "validation", | |
| "samples": 1221, | |
| "shots": 0, | |
| "primary": "acc", | |
| "metrics": { | |
| "acc": { | |
| "value": 0.21457821457821458, | |
| "stderr": 0.011753423094216953, | |
| "ci95": [ | |
| 0.19246524693116476, | |
| 0.2384815135817527 | |
| ], | |
| "ci_method": "Wilson 95%; item independence approximation" | |
| } | |
| }, | |
| "result_path": "runs/eval/cria-backfill-20261001/diffusion-v2/attempt-001/commonsense_qa.json", | |
| "result_sha256": "28a06065e6b827e085a8a03ce1bba594ae9edf1ae0cc7d5814b01be992f064bb", | |
| "categories": {} | |
| }, | |
| "sciq": { | |
| "label": "SciQ (with support)", | |
| "dataset": "allenai/sciq", | |
| "config": "default", | |
| "split": "test", | |
| "samples": 1000, | |
| "shots": 0, | |
| "primary": "acc_norm", | |
| "metrics": { | |
| "acc": { | |
| "value": 0.723, | |
| "stderr": 0.014158794845306273, | |
| "ci95": [ | |
| 0.6944497560637553, | |
| 0.7498435096516871 | |
| ], | |
| "ci_method": "Wilson 95%; item independence approximation" | |
| }, | |
| "acc_norm": { | |
| "value": 0.688, | |
| "stderr": 0.014658474370509057, | |
| "ci95": [ | |
| 0.6586108249018996, | |
| 0.7159503139075316 | |
| ], | |
| "ci_method": "Wilson 95%; item independence approximation" | |
| } | |
| }, | |
| "result_path": "runs/eval/cria-backfill-20261001/diffusion-v2/attempt-001/sciq.json", | |
| "result_sha256": "0f1d7377097498ae7ea47e0c1b70941aedbcea59f3e2b3ca0c0101083e6af4ee", | |
| "categories": {} | |
| }, | |
| "truthfulqa_mc2": { | |
| "label": "TruthfulQA MC2", | |
| "dataset": "truthfulqa/truthful_qa", | |
| "config": "multiple_choice", | |
| "split": "validation", | |
| "samples": 817, | |
| "shots": 0, | |
| "primary": "acc", | |
| "metrics": { | |
| "acc": { | |
| "value": 0.45836334441298165, | |
| "stderr": 0.0158524239677764 | |
| } | |
| }, | |
| "result_path": "runs/eval/cria-backfill-20261001/diffusion-v2/attempt-001/truthfulqa_mc2.json", | |
| "result_sha256": "bfc8d62f669c951f26cc2d300dbc4bd7c7787001573e0fc310c2db27c9405541", | |
| "categories": {} | |
| }, | |
| "bananamind_base": { | |
| "label": "BananaMind Base 1.1", | |
| "dataset": "BananaMind/BananaMind-Base-Bench-1.1", | |
| "config": "default", | |
| "split": "test", | |
| "samples": 350, | |
| "shots": 0, | |
| "primary": "raw_accuracy", | |
| "metrics": { | |
| "raw_accuracy": { | |
| "value": 0.39714285714285713, | |
| "stderr": 0.02619195222772307, | |
| "ci95": [ | |
| 0.34726441786546686, | |
| 0.4492546213676228 | |
| ], | |
| "ci_method": "Wilson 95%; item independence approximation" | |
| } | |
| }, | |
| "result_path": "runs/eval/cria-backfill-20261001/diffusion-v2/attempt-001/bananamind_base.json", | |
| "result_sha256": "5d9eb7e2d3daa94d14dd7d35a35003c17ab1f376931e739ca027198bb566dcb7", | |
| "categories": { | |
| "code_completion": { | |
| "count": 50, | |
| "raw_accuracy,none": 0.28, | |
| "raw_accuracy_stderr,none": 0.06414269805898186 | |
| }, | |
| "commonsense": { | |
| "count": 50, | |
| "raw_accuracy,none": 0.26, | |
| "raw_accuracy_stderr,none": 0.06266203485560375 | |
| }, | |
| "context_tracking": { | |
| "count": 50, | |
| "raw_accuracy,none": 0.48, | |
| "raw_accuracy_stderr,none": 0.07137140569598169 | |
| }, | |
| "language_completion": { | |
| "count": 50, | |
| "raw_accuracy,none": 0.4, | |
| "raw_accuracy_stderr,none": 0.06998542122237651 | |
| }, | |
| "logical_reasoning": { | |
| "count": 50, | |
| "raw_accuracy,none": 0.44, | |
| "raw_accuracy_stderr,none": 0.07091242083423346 | |
| }, | |
| "quantitative": { | |
| "count": 50, | |
| "raw_accuracy,none": 0.36, | |
| "raw_accuracy_stderr,none": 0.06857142857142857 | |
| }, | |
| "world_knowledge": { | |
| "count": 50, | |
| "raw_accuracy,none": 0.56, | |
| "raw_accuracy_stderr,none": 0.07091242083423346 | |
| } | |
| } | |
| }, | |
| "mmlu_continuation": { | |
| "label": "MMLU continuation", | |
| "dataset": "cais/mmlu", | |
| "config": "57 subjects", | |
| "split": "test", | |
| "samples": 14042, | |
| "shots": 0, | |
| "primary": "acc", | |
| "metrics": { | |
| "acc": { | |
| "value": 0.24483691781797465, | |
| "stderr": 0.0036208715458871405 | |
| } | |
| }, | |
| "result_path": "runs/eval/cria-backfill-20261001/diffusion-v2/attempt-001/mmlu_continuation.json", | |
| "result_sha256": "423e99cb8501f47e672d81cfa27934e7e8986de4b85e52c5248b7ac37a01370a", | |
| "categories": {} | |
| }, | |
| "blimp": { | |
| "label": "BLiMP", | |
| "dataset": "nyu-mll/blimp", | |
| "config": "67 minimal-pair subsets", | |
| "split": "train", | |
| "samples": 67000, | |
| "shots": 0, | |
| "primary": "acc", | |
| "metrics": { | |
| "acc": { | |
| "value": 0.6420746268656716, | |
| "stderr": 0.0016890890018278603 | |
| } | |
| }, | |
| "result_path": "runs/eval/cria-backfill-20261001/diffusion-v2/attempt-001/blimp.json", | |
| "result_sha256": "585fbe3ad36be3a0616a31f078a27c405f136571da4859f6f6cd216cad0b0f09", | |
| "categories": {} | |
| } | |
| }, | |
| "batch_size": 1, | |
| "label": "Diffusion v2 (PLL)", | |
| "manifests": { | |
| "runs/eval/cria-backfill-20261001/diffusion-v2/attempt-001": { | |
| "checkpoint": "/mnt/d/Projects/tiny_llm/runs/eval/diffusion-v2-step15000-20260923", | |
| "local_model": "diffusion", | |
| "kind": "diffusion", | |
| "tokenizer": "/mnt/d/Projects/tiny_llm/tokenizer", | |
| "revision": "main", | |
| "tokenizer_revision": null, | |
| "protocol": "EXPERIMENTAL: continuation PLL; single-mask left-to-right greedy generation; NOT AR likelihood", | |
| "harness_version": "0.4.12", | |
| "smoke_only": false, | |
| "tasks": [ | |
| { | |
| "task": "arithmark3", | |
| "shots": 0, | |
| "description": "ArithMark-3.0 train-named evaluation split; token-normalized acc_norm", | |
| "unsafe_code": false, | |
| "rolling": false, | |
| "engine": "token_continuation" | |
| }, | |
| { | |
| "task": "balanced_copa", | |
| "shots": 0, | |
| "description": "Balanced COPA mirrored 1000-item train split; acc; cRia split inferred", | |
| "unsafe_code": false, | |
| "rolling": false, | |
| "engine": "lm_eval" | |
| }, | |
| { | |
| "task": "commonsense_qa", | |
| "shots": 0, | |
| "description": "CommonsenseQA; acc", | |
| "unsafe_code": false, | |
| "rolling": false, | |
| "engine": "lm_eval" | |
| }, | |
| { | |
| "task": "sciq", | |
| "shots": 0, | |
| "description": "SciQ test with support passage; acc and acc_norm", | |
| "unsafe_code": false, | |
| "rolling": false, | |
| "engine": "lm_eval" | |
| }, | |
| { | |
| "task": "truthfulqa_mc2", | |
| "shots": 0, | |
| "description": "TruthfulQA multiple choice; mc2", | |
| "unsafe_code": false, | |
| "rolling": false, | |
| "engine": "lm_eval" | |
| }, | |
| { | |
| "task": "bananamind_base", | |
| "shots": 0, | |
| "description": "BananaMind Base Bench 1.1 test; mean-token raw accuracy; gated", | |
| "unsafe_code": false, | |
| "rolling": false, | |
| "engine": "token_continuation" | |
| }, | |
| { | |
| "task": "mmlu_continuation", | |
| "shots": 0, | |
| "description": "MMLU cloze: full answer continuations; acc", | |
| "unsafe_code": false, | |
| "rolling": false, | |
| "engine": "lm_eval" | |
| }, | |
| { | |
| "task": "blimp", | |
| "shots": 0, | |
| "description": "Full BLiMP group; grammatical minimal pairs", | |
| "unsafe_code": false, | |
| "rolling": false, | |
| "engine": "lm_eval" | |
| } | |
| ], | |
| "options": { | |
| "checkpoint": "/mnt/d/Projects/tiny_llm/runs/eval/diffusion-v2-step15000-20260923", | |
| "model": null, | |
| "tokenizer": "/mnt/d/Projects/tiny_llm/tokenizer", | |
| "revision": "main", | |
| "tokenizer_revision": null, | |
| "suite": "core", | |
| "tasks": "arithmark3,balanced_copa,commonsense_qa,sciq,truthfulqa_mc2,bananamind_base,mmlu_continuation,blimp", | |
| "num_fewshot": null, | |
| "device": "cuda:0", | |
| "dtype": "bfloat16", | |
| "allow_tf32": false, | |
| "batch_size": 1, | |
| "max_length": 2048, | |
| "max_gen_toks": null, | |
| "limit": null, | |
| "seed": 1234, | |
| "threads": 4, | |
| "bootstrap_iters": 1000, | |
| "trust_remote_code": false, | |
| "allow_code_execution": false, | |
| "apply_chat_template": false, | |
| "log_samples": false, | |
| "output": "/mnt/d/Projects/tiny_llm/runs/eval/cria-backfill-20261001/diffusion-v2/attempt-001", | |
| "tb_dir": "runs/diffusion-gpu-v2/tb", | |
| "training_step": 15000, | |
| "list": false, | |
| "dry_run": false | |
| }, | |
| "checkpoint_stamp": { | |
| "model.safetensors": [ | |
| 205580928, | |
| 1790122976166062900 | |
| ], | |
| "config.json": [ | |
| 677, | |
| 1790122975189825600 | |
| ] | |
| }, | |
| "checkpoint_sha256": { | |
| "config.json": "8e3daed6864d9b1f39be95c7b311a93c47efd4e76ee33fb4dbeede6d4701343d", | |
| "model.safetensors": "d0d24e0cc492cd5c9a613710c454af9a6e49265d2bae7abb69a673a53efe2bfb" | |
| }, | |
| "source_sha256": { | |
| "src/tiny_llm/__init__.py": "8c7f075d51006a53fe1d645ef5a84d65db58ff4251e7544074fe29ee7549c14c", | |
| "src/tiny_llm/benchmarks.py": "5d9f461366389bd1f926c6a6468192f3e2313c87caf5b1de27f407d563a858a4", | |
| "src/tiny_llm/checkpoint.py": "ac9df5f8c981a47eb1f1e59d0687ff41a1f6c38ca74ced50f71ac446a58d6ef5", | |
| "src/tiny_llm/comet_sync.py": "ea802bfb58dacad66f83b9e5ac5ccb330c190204a81c90c9e14aafe44c893c37", | |
| "src/tiny_llm/comet_views.py": "3e03f4ae69f8edc2eaf5f946f5cc90f373ca1b9cc91f9dd2e9adae6d6a672689", | |
| "src/tiny_llm/continuation.py": "4f4143c828d4f94a851bcbf8a466d3e1b3e9eaec1027970a426208cc0ff4f4db", | |
| "src/tiny_llm/data.py": "1a52e38afef9af32463779cc2f471d9b7174a7c4c73753e74f485d526d3c6733", | |
| "src/tiny_llm/diffusion_diagnostics.py": "2ae667a076b1d3f057a29d20f00559068eb73a073b3f99030e1d5b4c2f478fa5", | |
| "src/tiny_llm/evaluation/__init__.py": "e49e80f7be3ab6e200f0db14b028e465de9caa8f6f705430eabce0c41b8322ba", | |
| "src/tiny_llm/evaluation/adapters.py": "8e64e5542bbcb0f9ad448b320e712916a9559823987e298b22321a8d8e2f727f", | |
| "src/tiny_llm/evaluation/catalog.py": "2fd29084ec780539023a38bcbfd7302931fa6be48624e227875f13518a1b986f", | |
| "src/tiny_llm/evaluation/cli.py": "a322e39bf58ba52132249dae524e9772c3a8e1a5ba388f5c0267f8dc1ea18882", | |
| "src/tiny_llm/evaluation/tasks/balanced_copa.yaml": "f47baa6a713e15f7a5600ed257f316f8c54252bf4977bbc716a6da7aa580f05d", | |
| "src/tiny_llm/evaluation/tasks/balanced_copa_test.yaml": "dce21ad70b395d8d576c584be1b87ebae846e360e59878030645215a387b28e6", | |
| "src/tiny_llm/evaluation/telemetry.py": "f4dad30574753739ea10ba71ef9a7ed9d9118a89a8202dee224961be27f13a48", | |
| "src/tiny_llm/evaluation/token_continuation.py": "e8d88cd4aca51ca2fda07f2b799a54612d0e2964a98e05824a135bea1a533550", | |
| "src/tiny_llm/losses.py": "592a05287369f6e5077b3faca404a9392de4772cf06389fb8aa7e36924e74f86", | |
| "src/tiny_llm/models/__init__.py": "b94ab167b6eb9f9b455823eea382e67b5ac2cc7f9ef5f6de183a09b9650b9718", | |
| "src/tiny_llm/models/diffusion/__init__.py": "c92eaf5986d604d8aaf1894a862a217cefa32ab25359175b88110e45a71c1b25", | |
| "src/tiny_llm/models/diffusion/configuration_diffusion_lm.py": "3b303aba60f0a831c966f5cfa7f5441daa2584814e335646195bd6ceecfa6d0c", | |
| "src/tiny_llm/models/diffusion/modeling_diffusion_lm.py": "4fc21c2aed1d5a231afe6f7634d9d8c3682b4e2abfc8fecef27b52136de77b2f", | |
| "src/tiny_llm/models/looped/__init__.py": "4b46b426021c59ed313cb2f49afc1926ad2793d771ef90117e3fef0462957cff", | |
| "src/tiny_llm/models/looped/configuration_looped_lm.py": "b8d99c52b98e0a97a4cae76ca4c20affd8f4d019ec38e0a9beefd5fad864dc25", | |
| "src/tiny_llm/models/looped/modeling_looped_lm.py": "9bdb3c5e3aaae663d09e98055f95df1adbc42e0a4bf5dffb7f6c39242af41b9e", | |
| "src/tiny_llm/models/moe_t5/__init__.py": "18e33fe38b2049521f573512ac8796afa209e1b25c2052784e1c927a7160d66c", | |
| "src/tiny_llm/models/moe_t5/configuration_alicet5_moe.py": "4647b1e9ab9ce07887bffc1abc3beb3e691962a380b4cc237aae376490bb56b7", | |
| "src/tiny_llm/models/moe_t5/modeling_alicet5_moe.py": "5f899c4726b120eed6c9bf6258ad319bb1ccea734645f27e7e6ccbc5057222ff", | |
| "src/tiny_llm/models/prefixlm/__init__.py": "c94c3c861acba8d2147a3a6c46bccfc71ac8adc8d70757bede009598029d31ea", | |
| "src/tiny_llm/models/prefixlm/configuration_prefix_lm.py": "20b3a62ae3628b5a64566ef19d2947c90f10c6159daad00dd13d18fc5910c42c", | |
| "src/tiny_llm/models/prefixlm/modeling_prefix_lm.py": "f62a38211df175696f64945173d1e41d9f7fcf6152a7648020da8c91788e98fd", | |
| "src/tiny_llm/models/q50m/__init__.py": "9d0dfec6fa0cfa6967e2f17a73d8fbeca69ba93b4a93651eda4d95626d5f5eb5", | |
| "src/tiny_llm/models/q50m/configuration_q50m.py": "796806b2d5d982b633796bc7de53c3073b1b6e52d259f44ef798b63f55eaea92", | |
| "src/tiny_llm/models/q50m/modeling_q50m.py": "b68abce1f19a36deb39c4ac4e520ffaab0c879d8a7cc2abef35f014bacc011f2", | |
| "src/tiny_llm/models/qmod/__init__.py": "e8c3e10710ac1cceac72ea329db480e680e8bbaae2a92ec26c36c92a778552ad", | |
| "src/tiny_llm/models/qmod/configuration_qmod.py": "0dfc9a4540697a9b15e423b9592df9fc435dcfc437ddad20dc6b6377ebebea40", | |
| "src/tiny_llm/models/qmod/modeling_qmod.py": "3831e34a86309b91a7a0787b9471eaceb91aa4c267f8030eb2e6722a0fcab6f8", | |
| "src/tiny_llm/models/qmoe/__init__.py": "cfb2bb24bc2f3d5b72b3890a393b0386cb25f127e375f5454255478a36dec32f", | |
| "src/tiny_llm/models/qmoe/configuration_qmoe.py": "bba7c9ac16f444146a78c99e7a26b4a11864b215d2b673bf658d9930f2f4de2f", | |
| "src/tiny_llm/models/qmoe/modeling_qmoe.py": "1d713e5e287f46dbf13b7cf9c562b4539efbf6231f7ceaa261f8cea1acd2768c", | |
| "src/tiny_llm/models/qt5/__init__.py": "df943b42a9f339e006687bcd30a666838c1dec15c79751b944e5f7a2bc69186c", | |
| "src/tiny_llm/models/qt5/configuration_qt5.py": "59709fe2c9fe08ccecea6633b29b3b47170fd30abcbee37952507d676446cd44", | |
| "src/tiny_llm/models/qt5/modeling_qt5.py": "6269829cca6631824622df034ff7b544dbbe47790556c60b26e356b5279cceff", | |
| "src/tiny_llm/models/t5/__init__.py": "e4d8838253800edbcac1f9a9b6dc875bcb3ffa2962a0c9044d066448ea10488e", | |
| "src/tiny_llm/models/t5/configuration_alicet5.py": "81b1f594139f201af9afb25ba58df862432c15dedf57f6e340ee2af80ed0d44f", | |
| "src/tiny_llm/models/t5/modeling_alicet5.py": "bd52ce0022668f2defb57770df229f148e070ea73becca9a2e681325114bfc20", | |
| "src/tiny_llm/models/zarya/__init__.py": "651b19cdba422680ff16bf2b2f63be0c3e9045c44c410fb17dbf51df2a23bd55", | |
| "src/tiny_llm/models/zarya/configuration_zarya_lm.py": "42a3e7a1b11b687bac34e00f9cd3cb59d1cd573eec7492e1c55834662d37ed3f", | |
| "src/tiny_llm/models/zarya/modeling_zarya_lm.py": "73ab033860288f7367d5b912c79de908fd81a4e88f369a658b96dd2bcfcd7bfe", | |
| "src/tiny_llm/telemetry.py": "7ee8f14f4545bfa0e5ac632440d49c7f86fd6812a16429f862ebe3cfe09e3545", | |
| "src/tiny_llm/ul2.py": "f4d948d89d5e106965612971dd473a54b57719f811761f18457c6221d763b876", | |
| "src/tiny_llm/zarya.py": "8b5084ab5086c75745115d3a9ba4f2fa9cfda8ad713f65d3f8a22e77db8c0941" | |
| }, | |
| "packages": { | |
| "lm_eval": "0.4.12", | |
| "torch": "2.11.0+cu128", | |
| "transformers": "5.17.0", | |
| "datasets": "5.0.1", | |
| "accelerate": "1.15.0" | |
| }, | |
| "git_commit": "fb65467388c90a7d94d50fbcb711abfd8a568102", | |
| "git_dirty": true | |
| } | |
| }, | |
| "tokenizer_sha256": { | |
| "tokenizer.json": "6155a4212832ddd656acd1cd45ab4acb327dbb080e3acdbec84f086adf17414e", | |
| "tokenizer_config.json": "cccfc292e0846e8ab8aa04dafae32483a831bcf1df6f15100f82d306b4a9f4c6" | |
| }, | |
| "tensorboard_event_files": { | |
| "runs/diffusion-gpu-v2/tb/events.out.tfevents.1790869605.DESKTOP-OU3A33R.22736.0.eval": "09883b49cdd62a5c8d42822fc29a35820226c1a3dd208b40f9d979c36f869bea", | |
| "runs/diffusion-gpu-v2/tb/events.out.tfevents.1790869629.DESKTOP-OU3A33R.22736.1.eval": "2e57b508c1a857eaa1ed999eed15938a410f7c121a210bad83fa69eeada93658", | |
| "runs/diffusion-gpu-v2/tb/events.out.tfevents.1790869675.DESKTOP-OU3A33R.22736.2.eval": "2bae9a58f6c351803b0b319919a9d8cc7186ac3e2f2260ae30345a026a3e480c", | |
| "runs/diffusion-gpu-v2/tb/events.out.tfevents.1790869703.DESKTOP-OU3A33R.22736.3.eval": "899e1194dc8114940e256510c7c05ae80ff677e26d37ae4bf993febf50dc7f34", | |
| "runs/diffusion-gpu-v2/tb/events.out.tfevents.1790869755.DESKTOP-OU3A33R.22736.4.eval": "e6b697b446f79c28d37039be65f7b2c25ceb1229f2613fd0c14ae810bab0e69a", | |
| "runs/diffusion-gpu-v2/tb/events.out.tfevents.1790869765.DESKTOP-OU3A33R.22736.5.eval": "f1a8ba94bd9321bfe0e1c89de68577390172de30d014932a2cba5e418058bb5d", | |
| "runs/diffusion-gpu-v2/tb/events.out.tfevents.1790870472.DESKTOP-OU3A33R.22736.6.eval": "9fc3d6b6a242981d27e6de0c0ad4a6f3ebed56d14127b78b2d0d8ec4706378b3", | |
| "runs/diffusion-gpu-v2/tb/events.out.tfevents.1790871766.DESKTOP-OU3A33R.22736.7.eval": "1d195bb7e90934ea6960c362e14f3d097376fa93db6bd3b141bf4b35f5eba525" | |
| }, | |
| "tensorboard_verified_points": 1166 | |
| } | |
| } | |