HyperNix.3-mini / bench_hypernix3_mini.py
ray0rf1re's picture
Add hypernix.3-mini weights, card and benchmarks
8bdb1c4 verified
Raw History Blame Contribute Delete
20.8 kB
#!/usr/bin/env python3
"""
bench_hypernix3_mini.py — measure hypernix.3-mini against real baselines.
hypernix.3-mini 48.7M params 1.21B tokens (this repo)
Qwen/Qwen3.5-0.8B-Base ~800M params (~16x larger)
LiquidAI/LFM2.5-350M-Base ~350M params (~7x larger)
LiquidAI/LFM2.5-230M-Base 229.7M params (~4.7x larger)
Everything is scored by EleutherAI's lm-evaluation-harness, so the task
formatting, the length normalisation and the metric definitions are the same
ones behind every published number for the baselines — the only thing this file
adds is a `TemplateLM` subclass that lets the harness talk to a hyperNix0x-v2
model through NeoOven. The baselines go through the harness's own `HFLM`.
Nothing is copied from anyone's model card: every row in the output table is
produced on your machine, by this script, or it is not in the table.
python bench_hypernix3_mini.py --model-dir ./runs/hypernix.3-mini/final
python bench_hypernix3_mini.py --model-dir ... --limit 200 # quick pass
python bench_hypernix3_mini.py --model-dir ... --only hypernix # skip baselines
Outputs (next to the model, unless --out says otherwise):
bench_results.json full harness output, per-task, with stderr + config
bench_table.md the markdown table the model card embeds
A 48.7M base model is not going to beat a 0.8B instruct-lineage model on
knowledge tasks, and this script is not built to make it look like it does. It
reports accuracy *and* the things a small model actually wins on — parameters,
training compute, VRAM, tokens/s on the card you trained it on — so the
comparison is legible instead of flattering.
Requires: lm-eval>=0.4.9, transformers, torch, hypernix.
"""
from __future__ import annotations
import argparse
import json
import math
import os
import platform
import subprocess
import sys
import time
from pathlib import Path
from typing import Any
os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True")
import torch
MODEL_NAME = "hypernix.3-mini"
#: Base (not instruct) checkpoints, because hypernix.3-mini is a base model and
#: comparing it to an instruction-tuned one measures the tuning, not the pretrain.
BASELINES = {
"Qwen3.5-0.8B-Base": "Qwen/Qwen3.5-0.8B-Base", # ~800M, ~16x
"LFM2.5-350M-Base": "LiquidAI/LFM2.5-350M-Base", # ~350M, ~7x
"LFM2.5-230M-Base": "LiquidAI/LFM2.5-230M-Base", # 229.7M, ~4.7x
}
#: Zero-shot unless noted. These are the tasks a *base* model of this size is
#: normally reported on; winogrande and lambada are the two that separate a
#: model that learned something from one that learned n-gram statistics.
TASKS = [
"hellaswag",
"arc_easy",
"arc_challenge",
"piqa",
"winogrande",
"openbookqa",
"lambada_openai",
]
#: Reported separately: perplexity, not accuracy, and lower is better.
PPL_TASKS = ["wikitext"]
#: Which metric to pull for the summary table. acc_norm (length-normalised) is
#: the convention for the multiple-choice tasks that have it.
HEADLINE = {
"hellaswag": "acc_norm,none",
"arc_easy": "acc_norm,none",
"arc_challenge": "acc_norm,none",
"piqa": "acc_norm,none",
"openbookqa": "acc_norm,none",
"winogrande": "acc,none",
"lambada_openai": "acc,none",
"wikitext": "word_perplexity,none",
}
RANDOM_BASELINE = {
"hellaswag": 25.0, "arc_easy": 25.0, "arc_challenge": 25.0,
"piqa": 50.0, "winogrande": 50.0, "openbookqa": 25.0, "lambada_openai": 0.0,
}
def log(msg: str) -> None:
print(f"[bench] {msg}", flush=True)
# ---------------------------------------------------------------------------
# hyperNix0x-v2 -> lm-eval
# ---------------------------------------------------------------------------
def make_hypernix_lm(model_dir: Path, tokenizer_dir: Path | None, device: str,
batch_size: int, max_length: int, patch: bool = True):
"""A TemplateLM the harness can score, backed by NeoOven.
TemplateLM already implements `loglikelihood` (and the request plumbing) on
top of three primitives: tokenise, score a batch of token pairs, and say
what the EOT id is. That is the whole adapter — the scoring maths below is
the same teacher-forced gather every harness backend does, so a number from
this class means what the same number means for Qwen.
"""
from lm_eval.api.instance import Instance # noqa: F401 (typing only)
from lm_eval.api.model import TemplateLM
from hypernix.models.neo_oven import preheat_brewed
if patch:
try:
sys.path.insert(0, str(Path(__file__).resolve().parent))
from train_hypernix3_mini import patch_brewer_attention
patch_brewer_attention()
except Exception as exc: # pragma: no cover
log(f"could not apply the attention patch ({exc}); "
f"continuing with stock hypernix")
oven = preheat_brewed(model_dir, tokenizer_source=tokenizer_dir or model_dir,
device=device, dtype="float32")
oven.model.eval()
if oven.tokenizer_kind != "hf":
raise SystemExit(
f"{model_dir} has no usable tokenizer (kind={oven.tokenizer_kind!r}). "
f"Pass --tokenizer-dir pointing at the run's tokenizer/ directory."
)
class HypernixLM(TemplateLM):
def __init__(self) -> None:
super().__init__()
self.oven = oven
self._device = torch.device(device)
self._batch = batch_size
self._max_len = min(max_length, oven.model.config.max_seq_len)
# -- primitives the harness needs --------------------------------
@property
def eot_token_id(self) -> int:
tid = getattr(self.oven.tokenizer, "eos_token_id", None)
return int(tid) if tid is not None else 0
@property
def max_length(self) -> int:
return self._max_len
def tok_encode(self, string: str, add_special_tokens: bool = False, **kw):
return self.oven.tokenizer.encode(string, add_special_tokens=False)
def tok_decode(self, tokens, **kw) -> str:
return self.oven.tokenizer.decode(tokens, skip_special_tokens=True)
# -- scoring ------------------------------------------------------
@torch.no_grad()
def _score_batch(self, chunk):
"""(logprob of continuation, was it the argmax everywhere) per item."""
inputs, spans = [], []
for _, ctx_enc, cont_enc in chunk:
ids = (ctx_enc + cont_enc)[-(self._max_len + 1):]
inp = ids[:-1] # teacher forcing
spans.append((len(inp) - len(cont_enc), len(cont_enc), ids[-len(cont_enc):]))
inputs.append(inp)
width = max(len(i) for i in inputs)
# Right padding is safe in a causal model: a padded tail cannot
# reach the positions we read back.
batch = torch.zeros(len(inputs), width, dtype=torch.long)
for row, ids in enumerate(inputs):
batch[row, :len(ids)] = torch.tensor(ids, dtype=torch.long)
batch = batch.to(self._device)
logits = self.oven.model(batch)["logits"].float()
logprobs = torch.log_softmax(logits, dim=-1)
out = []
for row, (start, length, targets) in enumerate(spans):
tgt = torch.tensor(targets, device=self._device).unsqueeze(-1)
sl = logprobs[row, start:start + length]
picked = torch.gather(sl, 1, tgt).squeeze(-1)
greedy = bool((sl.argmax(dim=-1) == tgt.squeeze(-1)).all())
out.append((float(picked.sum()), greedy))
return out
def _loglikelihood_tokens(self, requests, disable_tqdm: bool = False, **kw):
# Length-sorted so padding waste stays small, then restored.
order = sorted(range(len(requests)),
key=lambda i: -(len(requests[i][1]) + len(requests[i][2])))
results: list[Any] = [None] * len(requests)
done = 0
for i in range(0, len(order), self._batch):
idx = order[i:i + self._batch]
for j, res in zip(idx, self._score_batch([requests[j] for j in idx])):
results[j] = res
done += len(idx)
if not disable_tqdm and done % (self._batch * 50) < self._batch:
log(f" scored {done}/{len(requests)}")
return results
@torch.no_grad()
def loglikelihood_rolling(self, requests, disable_tqdm: bool = False):
"""Whole-document logprob, for the perplexity tasks."""
out = []
for req in requests:
(string,) = req.args
ids = self.tok_encode(string)
total, window = 0.0, self._max_len
# Non-overlapping windows, each conditioned on an EOT prefix:
# the harness's own default rolling strategy.
for start in range(0, len(ids), window):
chunk = ids[start:start + window]
if not chunk:
continue
inp = [self.eot_token_id] + chunk[:-1]
t = torch.tensor([inp], device=self._device)
lp = torch.log_softmax(self.oven.model(t)["logits"].float(), dim=-1)
tgt = torch.tensor(chunk, device=self._device).unsqueeze(-1)
total += float(torch.gather(lp[0], 1, tgt).sum())
out.append(total)
return out
def generate_until(self, requests, disable_tqdm: bool = False):
outs = []
for req in requests:
ctx, kwargs = req.args
stops = tuple(kwargs.get("until", []) or [])
text = self.oven.complete(
ctx, max_new_tokens=int(kwargs.get("max_gen_toks", 128)),
temperature=float(kwargs.get("temperature", 0.0) or 0.01), top_k=1)
for s in stops:
if s and s in text:
text = text.split(s)[0]
outs.append(text)
return outs
return HypernixLM(), oven
# ---------------------------------------------------------------------------
# Baselines -> lm-eval
# ---------------------------------------------------------------------------
def make_hf_lm(repo: str, device: str, batch_size: int, dtype: str):
from lm_eval.models.huggingface import HFLM
return HFLM(pretrained=repo, device=device, batch_size=batch_size, dtype=dtype,
trust_remote_code=True)
# ---------------------------------------------------------------------------
# Throughput / footprint — where a 48.7M model actually competes
# ---------------------------------------------------------------------------
@torch.no_grad()
def measure_speed(forward, n_params: int, vocab_size: int, device: str, ctx: int = 256,
new_tokens: int = 64, reps: int = 3) -> dict[str, float]:
"""Single-stream decode tokens/s and peak VRAM, same protocol for every model."""
if device.startswith("cuda"):
torch.cuda.synchronize()
torch.cuda.reset_peak_memory_stats()
ids = torch.randint(0, max(2, min(vocab_size, 1000)), (1, ctx), device=device)
forward(ids) # warm up kernels
if device.startswith("cuda"):
torch.cuda.synchronize()
best = math.inf
for _ in range(reps):
seq = ids
t0 = time.perf_counter()
for _ in range(new_tokens):
logits = forward(seq)
nxt = logits[:, -1:, :].argmax(dim=-1)
seq = torch.cat([seq, nxt], dim=1)
if device.startswith("cuda"):
torch.cuda.synchronize()
best = min(best, time.perf_counter() - t0)
peak = (torch.cuda.max_memory_allocated() / 1e9) if device.startswith("cuda") else 0.0
return {
"decode_tokens_per_s": new_tokens / best,
"peak_vram_gb": peak,
"params": n_params,
"fp32_weights_gb": n_params * 4 / 1e9,
}
# ---------------------------------------------------------------------------
# Running the suite
# ---------------------------------------------------------------------------
def run_suite(lm, tasks: list[str], limit: int | None, seed: int) -> dict:
from lm_eval import simple_evaluate
return simple_evaluate(model=lm, tasks=tasks, limit=limit, batch_size=None,
random_seed=seed, numpy_random_seed=seed,
torch_random_seed=seed, verbosity="WARNING")
def headline(results: dict, task: str) -> tuple[float | None, float | None]:
row = (results.get("results") or {}).get(task)
if not row:
return None, None
key = HEADLINE.get(task)
if key not in row: # fall back to whatever acc it reported
key = next((k for k in row if k.startswith(("acc_norm", "acc", "word_perplexity"))), None)
if key is None:
return None, None
err = row.get(key.replace(",", "_stderr,"), None)
value = row[key]
scale = 100.0 if not key.startswith("word_perplexity") else 1.0
return (float(value) * scale,
float(err) * scale if isinstance(err, (int, float)) else None)
def gpu_name() -> str:
if torch.cuda.is_available():
return torch.cuda.get_device_name(0)
try:
return subprocess.run(["uname", "-m"], capture_output=True, text=True).stdout.strip()
except Exception:
return platform.machine()
def build_table(payload: dict) -> str:
models = list(payload["models"])
lines = []
lines.append("| Task (0-shot) | " + " | ".join(models) + " | random |")
lines.append("|---|" + "---|" * (len(models) + 1))
for task in TASKS:
cells = []
for m in models:
v, e = payload["models"][m]["tasks"].get(task, (None, None))
cells.append("—" if v is None else
(f"{v:.1f}" + (f" ±{e:.1f}" if e else "")))
rnd = RANDOM_BASELINE.get(task)
lines.append(f"| {task} | " + " | ".join(cells) + " | " +
(f"{rnd:.0f}" if rnd else "—") + " |")
for task in PPL_TASKS:
cells = []
for m in models:
v, _ = payload["models"][m]["tasks"].get(task, (None, None))
cells.append("—" if v is None else f"{v:.1f}")
lines.append(f"| {task} ppl ↓ | " + " | ".join(cells) + " | — |")
lines.append("")
lines.append("| Footprint | " + " | ".join(models) + " |")
lines.append("|---|" + "---|" * len(models))
for label, key, fmt in (
("Parameters", "params", lambda v: f"{v / 1e6:.1f}M"),
("Weights (fp32)", "fp32_weights_gb", lambda v: f"{v:.2f} GB"),
("Peak VRAM, 1-stream decode", "peak_vram_gb", lambda v: f"{v:.2f} GB"),
("Decode tok/s", "decode_tokens_per_s", lambda v: f"{v:.1f}"),
):
cells = []
for m in models:
entry = payload["models"][m]
speed = entry.get("speed") or {}
v = speed.get(key)
if v is None and key == "params":
v = entry.get("params")
if v is None and key == "fp32_weights_gb" and entry.get("params"):
v = entry["params"] * 4 / 1e9
cells.append("—" if v is None else fmt(v))
lines.append(f"| {label} | " + " | ".join(cells) + " |")
return "\n".join(lines)
def main() -> None:
ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0],
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--model-dir", required=True,
help="the trained hypernix.3-mini export (config.json + model.pt)")
ap.add_argument("--tokenizer-dir", default=None)
ap.add_argument("--out", default=None, help="defaults to <model-dir>/..")
ap.add_argument("--device", default="cuda" if torch.cuda.is_available() else "cpu")
ap.add_argument("--batch-size", type=int, default=8)
ap.add_argument("--baseline-batch-size", type=int, default=4)
ap.add_argument("--baseline-dtype", default="float32",
help="float32 keeps it fair (and fast) on Pascal")
ap.add_argument("--max-length", type=int, default=1024)
ap.add_argument("--limit", type=int, default=None,
help="docs per task; None = the full set (what you publish)")
ap.add_argument("--tasks", default=",".join(TASKS + PPL_TASKS))
ap.add_argument("--only", default="all",
choices=["all", "hypernix", "baselines"])
ap.add_argument("--skip-speed", action="store_true")
ap.add_argument("--seed", type=int, default=1234)
ap.add_argument("--hypernix-src", default=None)
args = ap.parse_args()
if args.hypernix_src:
sys.path.insert(0, args.hypernix_src)
model_dir = Path(args.model_dir).expanduser().resolve()
out_dir = Path(args.out).expanduser() if args.out else model_dir.parent
out_dir.mkdir(parents=True, exist_ok=True)
tasks = [t.strip() for t in args.tasks.split(",") if t.strip()]
payload: dict[str, Any] = {
"benchmark_version": 1,
"generated_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
"harness": "EleutherAI/lm-evaluation-harness",
"limit": args.limit,
"seed": args.seed,
"device": args.device,
"gpu": gpu_name(),
"torch": torch.__version__,
"tasks": tasks,
"models": {},
}
try:
import lm_eval
payload["harness_version"] = getattr(lm_eval, "__version__", "unknown")
except Exception:
payload["harness_version"] = "unknown"
# -- our model ----------------------------------------------------------
if args.only in ("all", "hypernix"):
log(f"loading {MODEL_NAME} from {model_dir}")
lm, oven = make_hypernix_lm(model_dir,
Path(args.tokenizer_dir) if args.tokenizer_dir else None,
args.device, args.batch_size, args.max_length)
n_params = oven.model.num_params()
log(f"{MODEL_NAME}: {n_params:,} params — running {len(tasks)} tasks")
res = run_suite(lm, tasks, args.limit, args.seed)
entry = {"repo": "local", "tasks": {t: headline(res, t) for t in tasks},
"raw": res.get("results"), "params": n_params}
if not args.skip_speed:
entry["speed"] = measure_speed(lambda x: oven.model(x)["logits"], n_params,
oven.model.config.vocab_size, args.device)
entry["speed"]["params"] = n_params
payload["models"][MODEL_NAME] = entry
del lm, oven
if args.device.startswith("cuda"):
torch.cuda.empty_cache()
# -- baselines ----------------------------------------------------------
if args.only in ("all", "baselines"):
for label, repo in BASELINES.items():
log(f"loading baseline {repo}")
try:
lm = make_hf_lm(repo, args.device, args.baseline_batch_size,
args.baseline_dtype)
except Exception as exc:
log(f"SKIPPING {repo}: {type(exc).__name__}: {exc}")
payload["models"][label] = {"repo": repo, "error": str(exc), "tasks": {}}
continue
res = run_suite(lm, tasks, args.limit, args.seed)
n_params = sum(p.numel() for p in lm.model.parameters())
entry = {"repo": repo, "tasks": {t: headline(res, t) for t in tasks},
"raw": res.get("results"), "params": n_params}
if not args.skip_speed:
entry["speed"] = measure_speed(
lambda x: lm.model(input_ids=x).logits, n_params,
int(getattr(lm.model.config, "vocab_size", 32000)), args.device)
payload["models"][label] = entry
del lm
if args.device.startswith("cuda"):
torch.cuda.empty_cache()
table = build_table(payload)
(out_dir / "bench_results.json").write_text(json.dumps(payload, indent=2, default=str))
(out_dir / "bench_table.md").write_text(table + "\n")
print()
print(table)
print()
log(f"wrote {out_dir / 'bench_results.json'} and {out_dir / 'bench_table.md'}")
if args.limit:
log(f"NOTE: --limit {args.limit} was set. These are a sanity pass, not "
f"publishable numbers — re-run without --limit before you upload.")
if __name__ == "__main__":
main()