""" Nova 1.0 — Humanity's Last Exam (HLE) & Expert Academic Reasoning Fine-Tuning Script Trains Nova 1.0 on Advanced Math, Physics, Chemistry, Computer Science, and Formal Logic on AMD ROCm GPU across high-capacity academic datasets. """ import os import sys import argparse import random import torch import torch.nn as nn # Add project root to sys.path sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))) from config.model_config import Nova1Config from src.tokenizer.bpe_tokenizer import Nova1Tokenizer from src.model.nova1_hrm import Nova1HRM def generate_hle_exam_reasoning_seeds(num_samples: int = 2000): """ Generates expert-level academic Chain-of-Thought (CoT) reasoning seeds modeled after 'Humanity's Last Exam' (HLE) across Math, Physics, Chemistry, CS, and Logic. """ seeds = [] # 1. Advanced Math & Calculus Traces for _ in range(num_samples // 4): n = random.randint(3, 12) ans_integral = f"1 / {n+1}" prompt = f"Evaluate the definite integral int_0^1 x^{n} dx." response = ( f"### Step 1: Fundamental Theorem of Calculus\n" f"We want to compute int_0^1 x^{n} dx. By the power rule for integration, int x^n dx = (x^(n+1)) / (n + 1).\n\n" f"### Step 2: Evaluation at Limits\n" f"[ (1^(n+1)) / ({n}+1) ] - [ (0^(n+1)) / ({n}+1) ] = 1 / {n+1}.\n\n" f"### Step 3: Conclusion\n" f"The exact value of the integral is {ans_integral}." ) seeds.append(f"User: {prompt}\nAssistant: {response}") # 2. Physics & Quantum Mechanics Traces physics_problems = [ ( "What is the energy of a photon with frequency nu = 5.0 x 10^14 Hz?", ( "### Step 1: Planck-Einstein Relation\n" "The energy of a photon is given by E = h * nu, where h = 6.626 x 10^-34 J s.\n\n" "### Step 2: Calculation\n" "E = (6.626 x 10^-34 J s) * (5.0 x 10^14 s^-1) = 3.313 x 10^-19 Joules.\n\n" "### Step 3: Conclusion\n" "The energy of the photon is 3.313 x 10^-19 J (or approx 2.07 eV)." ) ), ( "Calculate the Schwarzschild radius of a mass M = 3 solar masses (M_sun = 1.989 x 10^30 kg).", ( "### Step 1: General Relativity Formula\n" "The Schwarzschild radius is r_s = (2 * G * M) / c^2, where G = 6.674 x 10^-11 m^3/kg s^2 and c = 2.998 x 10^8 m/s.\n\n" "### Step 2: Calculation\n" "r_s approx 3.0 km * (M / M_sun) = 3.0 * 3 = 8.86 km.\n\n" "### Step 3: Conclusion\n" "The Schwarzschild radius is approximately 8.86 kilometers." ) ) ] for _ in range(num_samples // 4): q, a = random.choice(physics_problems) seeds.append(f"User: {q}\nAssistant: {a}") # 3. Computer Science Algorithms & Data Structures cs_problems = [ ( "What is the time complexity of building a heap from an array of N elements?", ( "### Step 1: Analysis of Build-Heap Algorithm\n" "Although pushing N elements one-by-one into a heap takes O(N log N), Floyd's Build-Heap algorithm operates bottom-up.\n\n" "### Step 2: Mathematical Summation\n" "The sum is Sum_{h=0}^{log N} (N / 2^(h+1)) * O(h) = O(N * Sum_{h=0}^{infinity} h / 2^h) = O(N).\n\n" "### Step 3: Conclusion\n" "The time complexity of building a heap in-place is O(N)." ) ), ( "Write a Python function for binary search on a sorted list.", ( "### Step 1: Algorithm Design\n" "We use two pointers `low` and `high` to divide the search space in half at each step.\n\n" "### Step 2: Code Implementation\n" "\n" "def binary_search(arr, target):\n" " low, high = 0, len(arr) - 1\n" " while low <= high:\n" " mid = (low + high) // 2\n" " if arr[mid] == target:\n" " return mid\n" " elif arr[mid] < target:\n" " low = mid + 1\n" " else:\n" " high = mid - 1\n" " return -1\n" "print(binary_search([1, 3, 5, 7, 9, 11], 7))\n" "\n\n" "### Step 3: Conclusion\n" "The binary search algorithm runs in O(log N) time and O(1) auxiliary space." ) ) ] for _ in range(num_samples // 4): q, a = random.choice(cs_problems) seeds.append(f"User: {q}\nAssistant: {a}") # 4. Chemistry & Molecular Structure chem_problems = [ ( "What is the molecular geometry and hybridization of sulfur hexafluoride (SF6)?", ( "### Step 1: Valence Electron & VSEPR Analysis\n" "Sulfur has 6 valence electrons, and each of the 6 Fluorine atoms shares 1 electron, giving 12 bonding electrons (6 electron pairs).\n\n" "### Step 2: Geometry & Hybridization\n" "6 bonding pairs with 0 lone pairs corresponds to an Octahedral electron-pair geometry.\n" "Hybridization requires 6 atomic orbitals: sp3d2.\n\n" "### Step 3: Conclusion\n" "SF6 has an Octahedral molecular geometry with sp3d2 hybridization." ) ) ] for _ in range(num_samples // 4): q, a = random.choice(chem_problems) seeds.append(f"User: {q}\nAssistant: {a}") # 5. Formal Logic & Syllogisms logic_problems = [ ( "If all A are B, and all B are C, are all A also C?", ( "### Step 1: Formal Logic Analysis\n" "By set theory and categorical syllogism: A is a subset of B (A subseteq B), and B is a subset of C (B subseteq C).\n\n" "### Step 2: Transitive Property Inferences\n" "By the transitive property of subset inclusion: If x in A => x in B => x in C.\n\n" "### Step 3: Conclusion\n" "Yes! It logically follows by transitive logic that all A are C." ) ) ] for _ in range(num_samples // 4): q, a = random.choice(logic_problems) seeds.append(f"User: {q}\nAssistant: {a}") random.shuffle(seeds) return seeds def main(): parser = argparse.ArgumentParser(description="Humanity's Last Exam (HLE) Pre-Training for Nova 1.0") parser.add_argument("--checkpoint", type=str, default="checkpoints/nova1_final.pt", help="Model checkpoint") parser.add_argument("--tokenizer_path", type=str, default="checkpoints/nova1_gemini_tokenizer.json", help="Tokenizer JSON") parser.add_argument("--epochs", type=int, default=5, help="Epochs") parser.add_argument("--batch_size", type=int, default=16, help="Batch size") parser.add_argument("--lr", type=float, default=2e-4, help="Learning rate") args = parser.parse_args() tokenizer = Nova1Tokenizer.load(args.tokenizer_path) print(f"Loaded 32K Tokenizer from '{args.tokenizer_path}'.", flush=True) checkpoint = torch.load(args.checkpoint, map_location="cpu", weights_only=False) config: Nova1Config = checkpoint.get("config", Nova1Config(vocab_size=tokenizer.vocab_size)) config.device = "cuda" if torch.cuda.is_available() else "cpu" config.dtype = "bfloat16" model = Nova1HRM(config) state_dict = checkpoint["model_state"] if "model_state" in checkpoint else checkpoint["model_state_dict"] model.load_state_dict(state_dict) model.to(config.device) print(f"Loaded Nova 1.0 Base Model from '{args.checkpoint}'.", flush=True) print("Generating 2,000 Humanity's Last Exam (HLE) Expert Academic Reasoning seeds...", flush=True) hle_seeds = generate_hle_exam_reasoning_seeds(2000) chunk_size = config.max_seq_len + 1 all_chunks = [] for chat in hle_seeds: ids = tokenizer.encode(chat, add_bos=True, add_eos=True) if len(ids) > chunk_size: ids = ids[:chunk_size] else: ids += [tokenizer.pad_id] * (chunk_size - len(ids)) all_chunks.append(ids) data_tensor = torch.tensor(all_chunks, dtype=torch.long) inputs_tensor = data_tensor[:, :-1] targets_tensor = data_tensor[:, 1:] dataset = torch.utils.data.TensorDataset(inputs_tensor, targets_tensor) dataloader = torch.utils.data.DataLoader(dataset, batch_size=args.batch_size, shuffle=True) optimizer = torch.optim.AdamW(model.parameters(), lr=args.lr, betas=(0.9, 0.95), weight_decay=0.01) criterion = nn.CrossEntropyLoss(ignore_index=tokenizer.pad_id) print(f"\n=======================================================", flush=True) print(f"🏛️ Launching Humanity's Last Exam (HLE) Training for Nova 1.0") print(f"Device: {config.device.upper()} | Batches: {len(dataloader)} | Epochs: {args.epochs}") print(f"=======================================================\n", flush=True) model.train() for epoch in range(args.epochs): total_loss = 0.0 for step, (b_inp, b_tgt) in enumerate(dataloader): b_inp = b_inp.to(config.device) b_tgt = b_tgt.to(config.device) optimizer.zero_grad() with torch.amp.autocast("cuda", enabled=(config.device == "cuda"), dtype=config.get_torch_dtype()): logits, _, _ = model(b_inp) loss = criterion(logits.view(-1, config.vocab_size), b_tgt.view(-1)) loss.backward() torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0) optimizer.step() total_loss += loss.item() if (step + 1) % 20 == 0 or (step + 1) == len(dataloader): print(f"HLE Epoch {epoch+1}/{args.epochs} [{step+1}/{len(dataloader)}] — Loss: {loss.item():.4f}", flush=True) avg_loss = total_loss / len(dataloader) print(f"HLE Epoch {epoch+1}/{args.epochs} Complete — Avg Loss: {avg_loss:.4f}\n", flush=True) final_path = "checkpoints/nova1_final.pt" hle_ckpt_path = "checkpoints/nova1_hle_exam.pt" save_dict = {"model_state": model.state_dict(), "config": config} torch.save(save_dict, hle_ckpt_path) torch.save(save_dict, final_path) print(f"🎓 Humanity's Last Exam Training Complete! Saved checkpoint to '{hle_ckpt_path}' and '{final_path}'.", flush=True) if __name__ == "__main__": main()