Download bench_hypernix3_mini.py from ray0rf1re/HyperNix.3-mini: direct link, hf CLI and curl.
- Browser
- Download file 20.8 kB
-
https://huggingface.co/ray0rf1re/HyperNix.3-mini/resolve/main/bench_hypernix3_mini.py
- Command line
-
hf download hf://ray0rf1re/HyperNix.3-mini/bench_hypernix3_mini.py
-
curl -L -o bench_hypernix3_mini.py https://huggingface.co/ray0rf1re/HyperNix.3-mini/resolve/main/bench_hypernix3_mini.py
20.8 kB
| #!/usr/bin/env python3 | |
| """ | |
| bench_hypernix3_mini.py — measure hypernix.3-mini against real baselines. | |
| hypernix.3-mini 48.7M params 1.21B tokens (this repo) | |
| Qwen/Qwen3.5-0.8B-Base ~800M params (~16x larger) | |
| LiquidAI/LFM2.5-350M-Base ~350M params (~7x larger) | |
| LiquidAI/LFM2.5-230M-Base 229.7M params (~4.7x larger) | |
| Everything is scored by EleutherAI's lm-evaluation-harness, so the task | |
| formatting, the length normalisation and the metric definitions are the same | |
| ones behind every published number for the baselines — the only thing this file | |
| adds is a `TemplateLM` subclass that lets the harness talk to a hyperNix0x-v2 | |
| model through NeoOven. The baselines go through the harness's own `HFLM`. | |
| Nothing is copied from anyone's model card: every row in the output table is | |
| produced on your machine, by this script, or it is not in the table. | |
| python bench_hypernix3_mini.py --model-dir ./runs/hypernix.3-mini/final | |
| python bench_hypernix3_mini.py --model-dir ... --limit 200 # quick pass | |
| python bench_hypernix3_mini.py --model-dir ... --only hypernix # skip baselines | |
| Outputs (next to the model, unless --out says otherwise): | |
| bench_results.json full harness output, per-task, with stderr + config | |
| bench_table.md the markdown table the model card embeds | |
| A 48.7M base model is not going to beat a 0.8B instruct-lineage model on | |
| knowledge tasks, and this script is not built to make it look like it does. It | |
| reports accuracy *and* the things a small model actually wins on — parameters, | |
| training compute, VRAM, tokens/s on the card you trained it on — so the | |
| comparison is legible instead of flattering. | |
| Requires: lm-eval>=0.4.9, transformers, torch, hypernix. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| import math | |
| import os | |
| import platform | |
| import subprocess | |
| import sys | |
| import time | |
| from pathlib import Path | |
| from typing import Any | |
| os.environ.setdefault("TOKENIZERS_PARALLELISM", "false") | |
| os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True") | |
| import torch | |
| MODEL_NAME = "hypernix.3-mini" | |
| #: Base (not instruct) checkpoints, because hypernix.3-mini is a base model and | |
| #: comparing it to an instruction-tuned one measures the tuning, not the pretrain. | |
| BASELINES = { | |
| "Qwen3.5-0.8B-Base": "Qwen/Qwen3.5-0.8B-Base", # ~800M, ~16x | |
| "LFM2.5-350M-Base": "LiquidAI/LFM2.5-350M-Base", # ~350M, ~7x | |
| "LFM2.5-230M-Base": "LiquidAI/LFM2.5-230M-Base", # 229.7M, ~4.7x | |
| } | |
| #: Zero-shot unless noted. These are the tasks a *base* model of this size is | |
| #: normally reported on; winogrande and lambada are the two that separate a | |
| #: model that learned something from one that learned n-gram statistics. | |
| TASKS = [ | |
| "hellaswag", | |
| "arc_easy", | |
| "arc_challenge", | |
| "piqa", | |
| "winogrande", | |
| "openbookqa", | |
| "lambada_openai", | |
| ] | |
| #: Reported separately: perplexity, not accuracy, and lower is better. | |
| PPL_TASKS = ["wikitext"] | |
| #: Which metric to pull for the summary table. acc_norm (length-normalised) is | |
| #: the convention for the multiple-choice tasks that have it. | |
| HEADLINE = { | |
| "hellaswag": "acc_norm,none", | |
| "arc_easy": "acc_norm,none", | |
| "arc_challenge": "acc_norm,none", | |
| "piqa": "acc_norm,none", | |
| "openbookqa": "acc_norm,none", | |
| "winogrande": "acc,none", | |
| "lambada_openai": "acc,none", | |
| "wikitext": "word_perplexity,none", | |
| } | |
| RANDOM_BASELINE = { | |
| "hellaswag": 25.0, "arc_easy": 25.0, "arc_challenge": 25.0, | |
| "piqa": 50.0, "winogrande": 50.0, "openbookqa": 25.0, "lambada_openai": 0.0, | |
| } | |
| def log(msg: str) -> None: | |
| print(f"[bench] {msg}", flush=True) | |
| # --------------------------------------------------------------------------- | |
| # hyperNix0x-v2 -> lm-eval | |
| # --------------------------------------------------------------------------- | |
| def make_hypernix_lm(model_dir: Path, tokenizer_dir: Path | None, device: str, | |
| batch_size: int, max_length: int, patch: bool = True): | |
| """A TemplateLM the harness can score, backed by NeoOven. | |
| TemplateLM already implements `loglikelihood` (and the request plumbing) on | |
| top of three primitives: tokenise, score a batch of token pairs, and say | |
| what the EOT id is. That is the whole adapter — the scoring maths below is | |
| the same teacher-forced gather every harness backend does, so a number from | |
| this class means what the same number means for Qwen. | |
| """ | |
| from lm_eval.api.instance import Instance # noqa: F401 (typing only) | |
| from lm_eval.api.model import TemplateLM | |
| from hypernix.models.neo_oven import preheat_brewed | |
| if patch: | |
| try: | |
| sys.path.insert(0, str(Path(__file__).resolve().parent)) | |
| from train_hypernix3_mini import patch_brewer_attention | |
| patch_brewer_attention() | |
| except Exception as exc: # pragma: no cover | |
| log(f"could not apply the attention patch ({exc}); " | |
| f"continuing with stock hypernix") | |
| oven = preheat_brewed(model_dir, tokenizer_source=tokenizer_dir or model_dir, | |
| device=device, dtype="float32") | |
| oven.model.eval() | |
| if oven.tokenizer_kind != "hf": | |
| raise SystemExit( | |
| f"{model_dir} has no usable tokenizer (kind={oven.tokenizer_kind!r}). " | |
| f"Pass --tokenizer-dir pointing at the run's tokenizer/ directory." | |
| ) | |
| class HypernixLM(TemplateLM): | |
| def __init__(self) -> None: | |
| super().__init__() | |
| self.oven = oven | |
| self._device = torch.device(device) | |
| self._batch = batch_size | |
| self._max_len = min(max_length, oven.model.config.max_seq_len) | |
| # -- primitives the harness needs -------------------------------- | |
| def eot_token_id(self) -> int: | |
| tid = getattr(self.oven.tokenizer, "eos_token_id", None) | |
| return int(tid) if tid is not None else 0 | |
| def max_length(self) -> int: | |
| return self._max_len | |
| def tok_encode(self, string: str, add_special_tokens: bool = False, **kw): | |
| return self.oven.tokenizer.encode(string, add_special_tokens=False) | |
| def tok_decode(self, tokens, **kw) -> str: | |
| return self.oven.tokenizer.decode(tokens, skip_special_tokens=True) | |
| # -- scoring ------------------------------------------------------ | |
| def _score_batch(self, chunk): | |
| """(logprob of continuation, was it the argmax everywhere) per item.""" | |
| inputs, spans = [], [] | |
| for _, ctx_enc, cont_enc in chunk: | |
| ids = (ctx_enc + cont_enc)[-(self._max_len + 1):] | |
| inp = ids[:-1] # teacher forcing | |
| spans.append((len(inp) - len(cont_enc), len(cont_enc), ids[-len(cont_enc):])) | |
| inputs.append(inp) | |
| width = max(len(i) for i in inputs) | |
| # Right padding is safe in a causal model: a padded tail cannot | |
| # reach the positions we read back. | |
| batch = torch.zeros(len(inputs), width, dtype=torch.long) | |
| for row, ids in enumerate(inputs): | |
| batch[row, :len(ids)] = torch.tensor(ids, dtype=torch.long) | |
| batch = batch.to(self._device) | |
| logits = self.oven.model(batch)["logits"].float() | |
| logprobs = torch.log_softmax(logits, dim=-1) | |
| out = [] | |
| for row, (start, length, targets) in enumerate(spans): | |
| tgt = torch.tensor(targets, device=self._device).unsqueeze(-1) | |
| sl = logprobs[row, start:start + length] | |
| picked = torch.gather(sl, 1, tgt).squeeze(-1) | |
| greedy = bool((sl.argmax(dim=-1) == tgt.squeeze(-1)).all()) | |
| out.append((float(picked.sum()), greedy)) | |
| return out | |
| def _loglikelihood_tokens(self, requests, disable_tqdm: bool = False, **kw): | |
| # Length-sorted so padding waste stays small, then restored. | |
| order = sorted(range(len(requests)), | |
| key=lambda i: -(len(requests[i][1]) + len(requests[i][2]))) | |
| results: list[Any] = [None] * len(requests) | |
| done = 0 | |
| for i in range(0, len(order), self._batch): | |
| idx = order[i:i + self._batch] | |
| for j, res in zip(idx, self._score_batch([requests[j] for j in idx])): | |
| results[j] = res | |
| done += len(idx) | |
| if not disable_tqdm and done % (self._batch * 50) < self._batch: | |
| log(f" scored {done}/{len(requests)}") | |
| return results | |
| def loglikelihood_rolling(self, requests, disable_tqdm: bool = False): | |
| """Whole-document logprob, for the perplexity tasks.""" | |
| out = [] | |
| for req in requests: | |
| (string,) = req.args | |
| ids = self.tok_encode(string) | |
| total, window = 0.0, self._max_len | |
| # Non-overlapping windows, each conditioned on an EOT prefix: | |
| # the harness's own default rolling strategy. | |
| for start in range(0, len(ids), window): | |
| chunk = ids[start:start + window] | |
| if not chunk: | |
| continue | |
| inp = [self.eot_token_id] + chunk[:-1] | |
| t = torch.tensor([inp], device=self._device) | |
| lp = torch.log_softmax(self.oven.model(t)["logits"].float(), dim=-1) | |
| tgt = torch.tensor(chunk, device=self._device).unsqueeze(-1) | |
| total += float(torch.gather(lp[0], 1, tgt).sum()) | |
| out.append(total) | |
| return out | |
| def generate_until(self, requests, disable_tqdm: bool = False): | |
| outs = [] | |
| for req in requests: | |
| ctx, kwargs = req.args | |
| stops = tuple(kwargs.get("until", []) or []) | |
| text = self.oven.complete( | |
| ctx, max_new_tokens=int(kwargs.get("max_gen_toks", 128)), | |
| temperature=float(kwargs.get("temperature", 0.0) or 0.01), top_k=1) | |
| for s in stops: | |
| if s and s in text: | |
| text = text.split(s)[0] | |
| outs.append(text) | |
| return outs | |
| return HypernixLM(), oven | |
| # --------------------------------------------------------------------------- | |
| # Baselines -> lm-eval | |
| # --------------------------------------------------------------------------- | |
| def make_hf_lm(repo: str, device: str, batch_size: int, dtype: str): | |
| from lm_eval.models.huggingface import HFLM | |
| return HFLM(pretrained=repo, device=device, batch_size=batch_size, dtype=dtype, | |
| trust_remote_code=True) | |
| # --------------------------------------------------------------------------- | |
| # Throughput / footprint — where a 48.7M model actually competes | |
| # --------------------------------------------------------------------------- | |
| def measure_speed(forward, n_params: int, vocab_size: int, device: str, ctx: int = 256, | |
| new_tokens: int = 64, reps: int = 3) -> dict[str, float]: | |
| """Single-stream decode tokens/s and peak VRAM, same protocol for every model.""" | |
| if device.startswith("cuda"): | |
| torch.cuda.synchronize() | |
| torch.cuda.reset_peak_memory_stats() | |
| ids = torch.randint(0, max(2, min(vocab_size, 1000)), (1, ctx), device=device) | |
| forward(ids) # warm up kernels | |
| if device.startswith("cuda"): | |
| torch.cuda.synchronize() | |
| best = math.inf | |
| for _ in range(reps): | |
| seq = ids | |
| t0 = time.perf_counter() | |
| for _ in range(new_tokens): | |
| logits = forward(seq) | |
| nxt = logits[:, -1:, :].argmax(dim=-1) | |
| seq = torch.cat([seq, nxt], dim=1) | |
| if device.startswith("cuda"): | |
| torch.cuda.synchronize() | |
| best = min(best, time.perf_counter() - t0) | |
| peak = (torch.cuda.max_memory_allocated() / 1e9) if device.startswith("cuda") else 0.0 | |
| return { | |
| "decode_tokens_per_s": new_tokens / best, | |
| "peak_vram_gb": peak, | |
| "params": n_params, | |
| "fp32_weights_gb": n_params * 4 / 1e9, | |
| } | |
| # --------------------------------------------------------------------------- | |
| # Running the suite | |
| # --------------------------------------------------------------------------- | |
| def run_suite(lm, tasks: list[str], limit: int | None, seed: int) -> dict: | |
| from lm_eval import simple_evaluate | |
| return simple_evaluate(model=lm, tasks=tasks, limit=limit, batch_size=None, | |
| random_seed=seed, numpy_random_seed=seed, | |
| torch_random_seed=seed, verbosity="WARNING") | |
| def headline(results: dict, task: str) -> tuple[float | None, float | None]: | |
| row = (results.get("results") or {}).get(task) | |
| if not row: | |
| return None, None | |
| key = HEADLINE.get(task) | |
| if key not in row: # fall back to whatever acc it reported | |
| key = next((k for k in row if k.startswith(("acc_norm", "acc", "word_perplexity"))), None) | |
| if key is None: | |
| return None, None | |
| err = row.get(key.replace(",", "_stderr,"), None) | |
| value = row[key] | |
| scale = 100.0 if not key.startswith("word_perplexity") else 1.0 | |
| return (float(value) * scale, | |
| float(err) * scale if isinstance(err, (int, float)) else None) | |
| def gpu_name() -> str: | |
| if torch.cuda.is_available(): | |
| return torch.cuda.get_device_name(0) | |
| try: | |
| return subprocess.run(["uname", "-m"], capture_output=True, text=True).stdout.strip() | |
| except Exception: | |
| return platform.machine() | |
| def build_table(payload: dict) -> str: | |
| models = list(payload["models"]) | |
| lines = [] | |
| lines.append("| Task (0-shot) | " + " | ".join(models) + " | random |") | |
| lines.append("|---|" + "---|" * (len(models) + 1)) | |
| for task in TASKS: | |
| cells = [] | |
| for m in models: | |
| v, e = payload["models"][m]["tasks"].get(task, (None, None)) | |
| cells.append("—" if v is None else | |
| (f"{v:.1f}" + (f" ±{e:.1f}" if e else ""))) | |
| rnd = RANDOM_BASELINE.get(task) | |
| lines.append(f"| {task} | " + " | ".join(cells) + " | " + | |
| (f"{rnd:.0f}" if rnd else "—") + " |") | |
| for task in PPL_TASKS: | |
| cells = [] | |
| for m in models: | |
| v, _ = payload["models"][m]["tasks"].get(task, (None, None)) | |
| cells.append("—" if v is None else f"{v:.1f}") | |
| lines.append(f"| {task} ppl ↓ | " + " | ".join(cells) + " | — |") | |
| lines.append("") | |
| lines.append("| Footprint | " + " | ".join(models) + " |") | |
| lines.append("|---|" + "---|" * len(models)) | |
| for label, key, fmt in ( | |
| ("Parameters", "params", lambda v: f"{v / 1e6:.1f}M"), | |
| ("Weights (fp32)", "fp32_weights_gb", lambda v: f"{v:.2f} GB"), | |
| ("Peak VRAM, 1-stream decode", "peak_vram_gb", lambda v: f"{v:.2f} GB"), | |
| ("Decode tok/s", "decode_tokens_per_s", lambda v: f"{v:.1f}"), | |
| ): | |
| cells = [] | |
| for m in models: | |
| entry = payload["models"][m] | |
| speed = entry.get("speed") or {} | |
| v = speed.get(key) | |
| if v is None and key == "params": | |
| v = entry.get("params") | |
| if v is None and key == "fp32_weights_gb" and entry.get("params"): | |
| v = entry["params"] * 4 / 1e9 | |
| cells.append("—" if v is None else fmt(v)) | |
| lines.append(f"| {label} | " + " | ".join(cells) + " |") | |
| return "\n".join(lines) | |
| def main() -> None: | |
| ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0], | |
| formatter_class=argparse.RawDescriptionHelpFormatter) | |
| ap.add_argument("--model-dir", required=True, | |
| help="the trained hypernix.3-mini export (config.json + model.pt)") | |
| ap.add_argument("--tokenizer-dir", default=None) | |
| ap.add_argument("--out", default=None, help="defaults to <model-dir>/..") | |
| ap.add_argument("--device", default="cuda" if torch.cuda.is_available() else "cpu") | |
| ap.add_argument("--batch-size", type=int, default=8) | |
| ap.add_argument("--baseline-batch-size", type=int, default=4) | |
| ap.add_argument("--baseline-dtype", default="float32", | |
| help="float32 keeps it fair (and fast) on Pascal") | |
| ap.add_argument("--max-length", type=int, default=1024) | |
| ap.add_argument("--limit", type=int, default=None, | |
| help="docs per task; None = the full set (what you publish)") | |
| ap.add_argument("--tasks", default=",".join(TASKS + PPL_TASKS)) | |
| ap.add_argument("--only", default="all", | |
| choices=["all", "hypernix", "baselines"]) | |
| ap.add_argument("--skip-speed", action="store_true") | |
| ap.add_argument("--seed", type=int, default=1234) | |
| ap.add_argument("--hypernix-src", default=None) | |
| args = ap.parse_args() | |
| if args.hypernix_src: | |
| sys.path.insert(0, args.hypernix_src) | |
| model_dir = Path(args.model_dir).expanduser().resolve() | |
| out_dir = Path(args.out).expanduser() if args.out else model_dir.parent | |
| out_dir.mkdir(parents=True, exist_ok=True) | |
| tasks = [t.strip() for t in args.tasks.split(",") if t.strip()] | |
| payload: dict[str, Any] = { | |
| "benchmark_version": 1, | |
| "generated_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), | |
| "harness": "EleutherAI/lm-evaluation-harness", | |
| "limit": args.limit, | |
| "seed": args.seed, | |
| "device": args.device, | |
| "gpu": gpu_name(), | |
| "torch": torch.__version__, | |
| "tasks": tasks, | |
| "models": {}, | |
| } | |
| try: | |
| import lm_eval | |
| payload["harness_version"] = getattr(lm_eval, "__version__", "unknown") | |
| except Exception: | |
| payload["harness_version"] = "unknown" | |
| # -- our model ---------------------------------------------------------- | |
| if args.only in ("all", "hypernix"): | |
| log(f"loading {MODEL_NAME} from {model_dir}") | |
| lm, oven = make_hypernix_lm(model_dir, | |
| Path(args.tokenizer_dir) if args.tokenizer_dir else None, | |
| args.device, args.batch_size, args.max_length) | |
| n_params = oven.model.num_params() | |
| log(f"{MODEL_NAME}: {n_params:,} params — running {len(tasks)} tasks") | |
| res = run_suite(lm, tasks, args.limit, args.seed) | |
| entry = {"repo": "local", "tasks": {t: headline(res, t) for t in tasks}, | |
| "raw": res.get("results"), "params": n_params} | |
| if not args.skip_speed: | |
| entry["speed"] = measure_speed(lambda x: oven.model(x)["logits"], n_params, | |
| oven.model.config.vocab_size, args.device) | |
| entry["speed"]["params"] = n_params | |
| payload["models"][MODEL_NAME] = entry | |
| del lm, oven | |
| if args.device.startswith("cuda"): | |
| torch.cuda.empty_cache() | |
| # -- baselines ---------------------------------------------------------- | |
| if args.only in ("all", "baselines"): | |
| for label, repo in BASELINES.items(): | |
| log(f"loading baseline {repo}") | |
| try: | |
| lm = make_hf_lm(repo, args.device, args.baseline_batch_size, | |
| args.baseline_dtype) | |
| except Exception as exc: | |
| log(f"SKIPPING {repo}: {type(exc).__name__}: {exc}") | |
| payload["models"][label] = {"repo": repo, "error": str(exc), "tasks": {}} | |
| continue | |
| res = run_suite(lm, tasks, args.limit, args.seed) | |
| n_params = sum(p.numel() for p in lm.model.parameters()) | |
| entry = {"repo": repo, "tasks": {t: headline(res, t) for t in tasks}, | |
| "raw": res.get("results"), "params": n_params} | |
| if not args.skip_speed: | |
| entry["speed"] = measure_speed( | |
| lambda x: lm.model(input_ids=x).logits, n_params, | |
| int(getattr(lm.model.config, "vocab_size", 32000)), args.device) | |
| payload["models"][label] = entry | |
| del lm | |
| if args.device.startswith("cuda"): | |
| torch.cuda.empty_cache() | |
| table = build_table(payload) | |
| (out_dir / "bench_results.json").write_text(json.dumps(payload, indent=2, default=str)) | |
| (out_dir / "bench_table.md").write_text(table + "\n") | |
| print() | |
| print(table) | |
| print() | |
| log(f"wrote {out_dir / 'bench_results.json'} and {out_dir / 'bench_table.md'}") | |
| if args.limit: | |
| log(f"NOTE: --limit {args.limit} was set. These are a sanity pass, not " | |
| f"publishable numbers — re-run without --limit before you upload.") | |
| if __name__ == "__main__": | |
| main() | |