From bf4833e6dbe0758a4c9165c8329053fc2791b5f1 Mon Sep 17 00:00:00 2001 From: Charles Martin Date: Tue, 29 Sep 2026 22:17:45 -0700 Subject: [PATCH 1/8] Add high-dose raw-alpha memorization protocol --- .../high_dose_memorization_raw_alpha.yaml | 140 ++++++++++++++++++ 1 file changed, 140 insertions(+) create mode 100644 baseline/experiments/nanogpt_one_head_2026_08_21_baseline/configs/high_dose_memorization_raw_alpha.yaml diff --git a/baseline/experiments/nanogpt_one_head_2026_08_21_baseline/configs/high_dose_memorization_raw_alpha.yaml b/baseline/experiments/nanogpt_one_head_2026_08_21_baseline/configs/high_dose_memorization_raw_alpha.yaml new file mode 100644 index 0000000..c2b641b --- /dev/null +++ b/baseline/experiments/nanogpt_one_head_2026_08_21_baseline/configs/high_dose_memorization_raw_alpha.yaml @@ -0,0 +1,140 @@ +protocol: + name: fineweb_high_dose_memorization_raw_alpha + version: 1 + description: Five-seed high-dose tracked-canary study. Exposure dose is the independent variable; raw WeightWatcher alpha is the primary spectral variable. No additional harmful random bank. + +dataset: + name: HuggingFaceFW/fineweb-edu + config: sample-10BT + split: train + revision: 593b3a867298afb8ce42625a270ef20ddcad28f9 + tokenizer: gpt2 + train_tokens: 80000000 + val_tokens: 1000000 + test_tokens: 1000000 + +model: + vocab_size: 50257 + block_size: 256 + n_layer: 1 + n_head: 1 + n_embd: 128 + dropout: 0.0 + bias: false + tie_weights: true + +training: + seeds: [1337, 2027, 4099, 31415, 271828] + batch_size: 4 + grad_accum_steps: 8 + target_epochs: 4.0 + epoch_interval: 0.125 + eval_interval_steps: 500 + eval_batches: 64 + checkpoint_interval_steps: 500 + grad_clip: 1.0 + +optimizer_profiles: + sgd_momentum: + display_name: SGD + Nesterov momentum (not a campaign arm) + family: sgd + learning_rate: 0.05 + min_learning_rate: 0.005 + warmup_fraction: 0.10 + lr_schedule_epochs: 1.0 + schedule: warmup_cosine + momentum: 0.90 + dampening: 0.0 + nesterov: true + weight_decay: 0.01 + adamw: + display_name: AdamW + family: adamw + learning_rate: 0.0006 + min_learning_rate: 0.00006 + warmup_fraction: 0.01 + lr_schedule_epochs: 1.0 + schedule: warmup_cosine + beta1: 0.90 + beta2: 0.95 + epsilon: 1.0e-8 + weight_decay: 0.10 + muon: + display_name: Muon + auxiliary AdamW (not a campaign arm) + family: muon + matrix_learning_rate: 0.02 + matrix_min_learning_rate: 0.002 + aux_learning_rate: 0.0003 + aux_min_learning_rate: 0.00003 + warmup_fraction: 0.05 + lr_schedule_epochs: 1.0 + schedule: warmup_cosine + momentum: 0.95 + nesterov: true + newton_schulz_steps: 5 + muon_epsilon: 1.0e-7 + matrix_weight_decay: 0.01 + beta1: 0.90 + beta2: 0.95 + epsilon: 1.0e-8 + aux_weight_decay: 0.01 + muon_clip: + display_name: MuonClip + auxiliary AdamW + family: muon_clip + learning_rate: 0.0002 + min_learning_rate: 0.00002 + warmup_fraction: 0.0512 + lr_schedule_epochs: 1.0 + schedule: warmup_cosine + momentum: 0.95 + nesterov: false + newton_schulz_steps: 5 + muon_epsilon: 1.0e-7 + weight_decay: 0.10 + update_rms_scale: 0.20 + qk_clip_threshold: 100.0 + qk_clip_balance: 0.50 + qk_diagnostics_interval: 500 + beta1: 0.90 + beta2: 0.95 + epsilon: 1.0e-8 + +evaluation: + train_probe_seed: 21001 + validation_probe_seed: 22001 + test_probe_seed: 23001 + bleu_probe_seed: 24001 + bleu_examples: 64 + bleu_prompt_tokens: 64 + bleu_continuation_tokens: 32 + bleu_batch_size: 4 + +weightwatcher: + enabled: true + ERG: true + randomize: true + strict: true + min_evals: 20 + fix_fingers: clip_xmax + max_fingers: 10 + require_raw_alpha: true + +runtime: + matmul_precision: highest + allow_tf32: false + cudnn_benchmark: false + mps_fallback: true + deterministic_algorithms: true + deterministic_warn_only: false + empty_mps_cache_after_weightwatcher: true + +memorization: + enabled: true + data_seed: 20260925 + doses: [0, 64, 128, 256, 512, 1024] + canaries_per_dose: 8 + prefix_tokens: 64 + suffix_tokens: 32 + acquisition_fraction: 0.50 + harmful_load_fraction: 0 + harmful_load_dose: 64 From 223ce39ee35f73e416a9deefee9fa91295d9162c Mon Sep 17 00:00:00 2001 From: Charles Martin Date: Tue, 29 Sep 2026 22:17:55 -0700 Subject: [PATCH 2/8] Add five-seed high-dose TPU sweep launcher --- .../tpu_high_dose_sweep.py | 85 +++++++++++++++++++ 1 file changed, 85 insertions(+) create mode 100644 baseline/nanogpt_one_head/src/rg_nanogpt_one_head/tpu_high_dose_sweep.py diff --git a/baseline/nanogpt_one_head/src/rg_nanogpt_one_head/tpu_high_dose_sweep.py b/baseline/nanogpt_one_head/src/rg_nanogpt_one_head/tpu_high_dose_sweep.py new file mode 100644 index 0000000..ace8c1e --- /dev/null +++ b/baseline/nanogpt_one_head/src/rg_nanogpt_one_head/tpu_high_dose_sweep.py @@ -0,0 +1,85 @@ +"""Five-seed high-dose canary experiment on four isolated TPU chips. + +Additive launcher built on the already-qualified TPU sweep supervisor. It creates +one condition ("highdose"), assigns the five seeds round-robin to chips 0..3, +and therefore runs four seeds concurrently followed by the fifth on chip 0. +""" +from __future__ import annotations +import argparse, json +from pathlib import Path +from copy import deepcopy + +from .config import load_config +from .data import validate_prepared_data +from . import tpu_memorization_sweep as base + +SEEDS = [1337, 2027, 4099, 31415, 271828] +CHIPS = [0, 1, 2, 3] +CONFIG_REL = Path("baseline/experiments/nanogpt_one_head_2026_08_21_baseline/configs/high_dose_memorization_raw_alpha.yaml") + +def make(args): + code = args.code.resolve() + data = args.data_root.resolve() + root = args.root.resolve() + cfg = load_config(code / CONFIG_REL) + if list(map(int, cfg["training"]["seeds"])) != SEEDS: + raise ValueError("registered seed set changed") + mem = cfg["memorization"] + if list(map(int, mem["doses"])) != [0,64,128,256,512,1024]: + raise ValueError("registered dose set changed") + if float(mem.get("harmful_load_fraction", -1)) != 0: + raise ValueError("high-dose study must not contain an additional random bank") + if float(cfg["training"]["epoch_interval"]) != 0.125: + raise ValueError("spectral checkpoint interval must be 0.125 epoch") + if cfg["weightwatcher"].get("require_raw_alpha") is not True: + raise ValueError("raw alpha is required") + metadata = validate_prepared_data(data, cfg) + base.check_root(root, code, False) + tasks = base.build_tasks(SEEDS, CHIPS, ["highdose"], ["muon_clip"]) + source = base.source_info(code) + plan = dict( + schema_version=1, executor="high-dose-raw-alpha-v1", + root=str(root), code=str(code), data_root=str(data), source=source, + hardware_block=args.hardware_block, chips=CHIPS, seeds=SEEDS, + loads=["highdose"], optimizers=["muon_clip"], canary_batch_size=48, + allow_ephemeral=False, corpus=metadata, + configs={"highdose": deepcopy(cfg)}, + tasks=[base.asdict(t) for t in tasks], + scientific_hypothesis={ + "primary_spectral_variable":"alpha_raw", + "threshold":2.0, + "independent_variable":"tracked_canary_exposure_dose", + "doses":[0,64,128,256,512,1024], + "acquisition_fraction":0.5, + "note":"alpha_clip_xmax is retained only as a secondary diagnostic", + }, + ) + return plan + +def main(argv=None): + p=argparse.ArgumentParser() + sub=p.add_subparsers(dest="cmd",required=True) + for cmd in ("plan","run"): + q=sub.add_parser(cmd) + q.add_argument("--code",type=Path,required=True) + q.add_argument("--data-root",type=Path,required=True) + q.add_argument("--root",type=Path,required=True) + q.add_argument("--hardware-block",required=True) + q.add_argument("--retries",type=int,default=2) + q=sub.add_parser("status"); q.add_argument("--root",type=Path,required=True) + args=p.parse_args(argv) + if args.cmd=="status": + return base.show_status(args.root.resolve()) + plan=make(args) + if args.cmd=="plan": + print(json.dumps({k:v for k,v in plan.items() if k!="configs"},indent=2)) + print("HIGH-DOSE STUDY: 5 runs x 39063 steps; doses 0,64,128,256,512,1024; RAW alpha primary; no training started") + return 0 + root=Path(plan["root"]); root.mkdir(parents=True,exist_ok=True) + with base.exclusive_lock(root/".sweep.lock"): + pf=root/"tpu_sweep_plan.json" + base.freeze_plan(pf,plan) + return base.run_sweep(pf,args.retries) + +if __name__=="__main__": + raise SystemExit(main()) From 0d8a1df8ccccc3af0737b7dd58a478505b115330 Mon Sep 17 00:00:00 2001 From: Charles Martin Date: Tue, 29 Sep 2026 22:18:12 -0700 Subject: [PATCH 3/8] Fix high-dose launcher config extension and task construction From 80c56eaaa23e352ae449c7adc6db0e2b6156da78 Mon Sep 17 00:00:00 2001 From: Charles Martin Date: Tue, 29 Sep 2026 22:36:52 -0700 Subject: [PATCH 4/8] Validate and run high-dose raw-alpha study through qualified TPU workers without changing old runs --- .../tpu_high_dose_sweep.py | 253 +++++++++++++----- 1 file changed, 182 insertions(+), 71 deletions(-) diff --git a/baseline/nanogpt_one_head/src/rg_nanogpt_one_head/tpu_high_dose_sweep.py b/baseline/nanogpt_one_head/src/rg_nanogpt_one_head/tpu_high_dose_sweep.py index ace8c1e..3df9c5e 100644 --- a/baseline/nanogpt_one_head/src/rg_nanogpt_one_head/tpu_high_dose_sweep.py +++ b/baseline/nanogpt_one_head/src/rg_nanogpt_one_head/tpu_high_dose_sweep.py @@ -1,85 +1,196 @@ -"""Five-seed high-dose canary experiment on four isolated TPU chips. +"""High-dose MuonClip study using the qualified independent-chip TPU executor. -Additive launcher built on the already-qualified TPU sweep supervisor. It creates -one condition ("highdose"), assigns the five seeds round-robin to chips 0..3, -and therefore runs four seeds concurrently followed by the fifth on chip 0. +No historical training/optimizer/evaluator code is patched. The old executor's +worker consumes an immutable plan, so it can execute this separately validated +protocol without going through the old load-sweep configuration builder. """ from __future__ import annotations -import argparse, json -from pathlib import Path + +import argparse from copy import deepcopy +from dataclasses import asdict +from datetime import datetime, timezone +import json +from pathlib import Path +import sys +import threading -from .config import load_config -from .data import validate_prepared_data from . import tpu_memorization_sweep as base SEEDS = [1337, 2027, 4099, 31415, 271828] CHIPS = [0, 1, 2, 3] -CONFIG_REL = Path("baseline/experiments/nanogpt_one_head_2026_08_21_baseline/configs/high_dose_memorization_raw_alpha.yaml") +DOSES = [0, 64, 128, 256, 512, 1024] +VERSION = "high-dose-raw-alpha-v1" +CONFIG_DIR = Path("baseline/experiments/nanogpt_one_head_2026_08_21_baseline/configs") +CONFIG_REL = CONFIG_DIR / "high_dose_memorization_raw_alpha.yaml" +DEFAULT_PREVIOUS = Path("/mnt/disks/rg-data/fineweb-memorization-tpu-v1") + + +def validate_study(cfg: dict, reference: dict) -> None: + """Allow only the explicitly agreed scientific changes from the prior study.""" + expected = deepcopy(reference) + expected["protocol"] = deepcopy(cfg["protocol"]) + expected["training"]["epoch_interval"] = 0.125 + expected["memorization"]["doses"] = DOSES.copy() + # The optional ordinary-Adam arm is not present in this MuonClip-only YAML. + expected["optimizer_profiles"].pop("adam", None) + if cfg != expected: + bad = sorted(k for k in set(cfg) | set(expected) if cfg.get(k) != expected.get(k)) + raise ValueError("unregistered high-dose protocol change in: " + ", ".join(bad)) + if cfg["protocol"]["name"] != "fineweb_high_dose_memorization_raw_alpha": + raise ValueError("wrong protocol name") + if cfg["protocol"]["version"] != 1 or cfg["training"]["seeds"] != SEEDS: + raise ValueError("wrong protocol version or seed inventory") + -def make(args): - code = args.code.resolve() - data = args.data_root.resolve() - root = args.root.resolve() +def load_study(code: Path) -> dict: + # Install before loading YAML: the historical validator otherwise does not + # register MuonClip. Worker processes independently install the same extension. + from .muonclip import install_muonclip_extension + install_muonclip_extension() + from .config import load_config cfg = load_config(code / CONFIG_REL) - if list(map(int, cfg["training"]["seeds"])) != SEEDS: - raise ValueError("registered seed set changed") - mem = cfg["memorization"] - if list(map(int, mem["doses"])) != [0,64,128,256,512,1024]: - raise ValueError("registered dose set changed") - if float(mem.get("harmful_load_fraction", -1)) != 0: - raise ValueError("high-dose study must not contain an additional random bank") - if float(cfg["training"]["epoch_interval"]) != 0.125: - raise ValueError("spectral checkpoint interval must be 0.125 epoch") - if cfg["weightwatcher"].get("require_raw_alpha") is not True: - raise ValueError("raw alpha is required") + reference = load_config(code / CONFIG_DIR / "harmful_memorization_0pct.yaml") + validate_study(cfg, reference) + return cfg + + +def tasks() -> list: + # Do not call base.build_tasks(..., loads=["highdose"]): that public helper + # deliberately accepts only the five historical additional-bank load names. + return [base.Task(i, "highdose", seed, "muon_clip", CHIPS[i % 4]) + for i, seed in enumerate(SEEDS)] + + +def require_previous_complete(root: Path) -> None: + """Read-only guard: all 25 previous tasks must have matching completion receipts.""" + plan = base.json_read(root / "tpu_sweep_plan.json") + previous_tasks = plan["tasks"] + if len(previous_tasks) != 25: + raise ValueError("--previous-root is not the 25-run campaign") + count = 0 + for task in previous_tasks: + receipt = root / "receipts" / f"task_{task['index']:03d}.json" + completion = (root / f"load_{task['load']}" / "results" / task["optimizer"] + / f"seed_{task['seed']}" / "run_complete.json") + if not receipt.is_file() or not completion.is_file(): + continue + record, done = base.json_read(receipt), base.json_read(completion) + saved = record.get("completion", {}) + if (record.get("task") == task and done.get("completed") is True + and done.get("optimizer_steps") == 39063 + and done.get("seed") == task["seed"] + and done.get("optimizer") == task["optimizer"] + and done.get("fingerprint") + and saved == done): + count += 1 + if count != 25: + raise RuntimeError(f"previous campaign has {count}/25 matching completions; wait, then retry. Nothing stopped.") + + +def make(args) -> dict: + code, data, root = args.code.resolve(), args.data_root.resolve(), args.root.resolve() + source = base.source_info(code) + if Path(__file__).resolve() != (code / "baseline/nanogpt_one_head/src/rg_nanogpt_one_head/tpu_high_dose_sweep.py").resolve(): + raise ValueError("this launcher is not running from --code") + previous = args.previous_root.resolve() + if root == previous or root in previous.parents or previous in root.parents: + raise ValueError("new output root must be separate from the previous campaign") + if (data == code or code in data.parents or root == data + or root in data.parents): + raise ValueError("data/output/source paths overlap unsafely") + require_previous_complete(previous) + cfg = load_study(code) + from .data import validate_prepared_data metadata = validate_prepared_data(data, cfg) base.check_root(root, code, False) - tasks = base.build_tasks(SEEDS, CHIPS, ["highdose"], ["muon_clip"]) - source = base.source_info(code) - plan = dict( - schema_version=1, executor="high-dose-raw-alpha-v1", - root=str(root), code=str(code), data_root=str(data), source=source, - hardware_block=args.hardware_block, chips=CHIPS, seeds=SEEDS, - loads=["highdose"], optimizers=["muon_clip"], canary_batch_size=48, - allow_ephemeral=False, corpus=metadata, - configs={"highdose": deepcopy(cfg)}, - tasks=[base.asdict(t) for t in tasks], - scientific_hypothesis={ - "primary_spectral_variable":"alpha_raw", - "threshold":2.0, - "independent_variable":"tracked_canary_exposure_dose", - "doses":[0,64,128,256,512,1024], - "acquisition_fraction":0.5, - "note":"alpha_clip_xmax is retained only as a secondary diagnostic", - }, - ) - return plan - -def main(argv=None): - p=argparse.ArgumentParser() - sub=p.add_subparsers(dest="cmd",required=True) - for cmd in ("plan","run"): - q=sub.add_parser(cmd) - q.add_argument("--code",type=Path,required=True) - q.add_argument("--data-root",type=Path,required=True) - q.add_argument("--root",type=Path,required=True) - q.add_argument("--hardware-block",required=True) - q.add_argument("--retries",type=int,default=2) - q=sub.add_parser("status"); q.add_argument("--root",type=Path,required=True) - args=p.parse_args(argv) - if args.cmd=="status": - return base.show_status(args.root.resolve()) - plan=make(args) - if args.cmd=="plan": - print(json.dumps({k:v for k,v in plan.items() if k!="configs"},indent=2)) - print("HIGH-DOSE STUDY: 5 runs x 39063 steps; doses 0,64,128,256,512,1024; RAW alpha primary; no training started") - return 0 - root=Path(plan["root"]); root.mkdir(parents=True,exist_ok=True) - with base.exclusive_lock(root/".sweep.lock"): - pf=root/"tpu_sweep_plan.json" - base.freeze_plan(pf,plan) - return base.run_sweep(pf,args.retries) - -if __name__=="__main__": + return dict(schema_version=1, executor=VERSION, root=str(root), code=str(code), + data_root=str(data), source=source, hardware_block=args.hardware_block, + chips=CHIPS, seeds=SEEDS, loads=["highdose"], optimizers=["muon_clip"], + canary_batch_size=48, allow_ephemeral=False, corpus=metadata, + configs={"highdose": cfg}, tasks=[asdict(t) for t in tasks()], + extension_sha256=base.sha256(Path(__file__)), + previous_root=str(previous), + scientific_hypothesis={"primary_spectral_variable": "alpha_raw", "threshold": 2.0, + "primary_summary": "minimum valid raw alpha across six hidden matrices; report argmin matrix", + "doses": DOSES, "canaries_per_dose": 8, "total_tracked_canaries": 48, + "acquisition_fraction": 0.5, "extra_random_bank_fraction": 0.0, + "clipped_alpha": "secondary diagnostic only; never substitute for raw", + "replication_unit": "training seed; doses share the same model at each step", + "interpretation": "within-run dose-response and stronger-recall pilot, not a between-model dose intervention"}) + + +def validate_plan_source(plan: dict) -> None: + code = Path(plan["code"]) + if base.source_info(code) != plan["source"]: + raise ValueError("source changed; use the original checkout") + path = code / "baseline/nanogpt_one_head/src/rg_nanogpt_one_head/tpu_high_dose_sweep.py" + if base.sha256(path) != plan["extension_sha256"]: + raise ValueError("high-dose launcher changed") + + +def run(plan: dict, retries: int) -> int: + root = Path(plan["root"]) + root.mkdir(parents=True, exist_ok=True) + with base.exclusive_lock(root / ".sweep.lock"): + validate_plan_source(plan) + pf = root / "tpu_sweep_plan.json" + base.freeze_plan(pf, plan) + done = threading.Event() + + def heartbeat(): + while not done.wait(60): + print("\n[high-dose] " + datetime.now(timezone.utc).isoformat(), flush=True) + try: + base.show_status(root) + except (OSError, ValueError, KeyError) as exc: + print(f"[high-dose] status unavailable on this poll: {exc}", flush=True) + sys.stdout.flush() + + thread = threading.Thread(target=heartbeat, daemon=True) + thread.start() + rc = 1 + try: + print("[high-dose] START: 5 seeds; chips 0,1,2,3; 39063 steps/run; RAW alpha primary", flush=True) + # Qualified worker restores MuonClip and the batched evaluator from + # the frozen plan; it does NOT call the historical load validator. + rc = base.run_sweep(pf, retries) + return rc + finally: + done.set() + thread.join(timeout=2) + base.atomic_json(root / "highdose_exit.json", {"exit_code": rc, "at": base.utc()}) + print(f"[high-dose] EXIT CODE: {rc}; results={root}", flush=True) + + +def main(argv=None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest="cmd", required=True) + for cmd in ("plan", "run"): + p = sub.add_parser(cmd) + for name in ("code", "data-root", "root"): + p.add_argument("--" + name, type=Path, required=True) + p.add_argument("--hardware-block", required=True) + p.add_argument("--previous-root", type=Path, default=DEFAULT_PREVIOUS) + p.add_argument("--retries", type=int, default=2) + p = sub.add_parser("status") + p.add_argument("--root", type=Path, required=True) + args = parser.parse_args(argv) + try: + if args.cmd == "status": + return base.show_status(args.root.resolve()) + if args.retries < 0: + raise ValueError("retries must be nonnegative") + plan = make(args) + if args.cmd == "plan": + print(json.dumps({k: v for k, v in plan.items() if k != "configs"}, indent=2)) + print("HIGH-DOSE PLAN OK: 5 runs x 39063 steps; doses 0,64,128,256,512,1024; RAW alpha primary; no training started") + return 0 + return run(plan, args.retries) + except (ValueError, RuntimeError, OSError, KeyError) as exc: + print(f"[high-dose] ERROR: {exc}", file=sys.stderr, flush=True) + return 2 + + +if __name__ == "__main__": raise SystemExit(main()) From a7cad4680d9afea7d259f584840fb7b471ddd1bf Mon Sep 17 00:00:00 2001 From: Charles Martin Date: Tue, 29 Sep 2026 22:37:36 -0700 Subject: [PATCH 5/8] Add detached tmux high-dose launch with persistent logs and no environment reinstall --- baseline/nanogpt_one_head/tpu_high_dose.sh | 70 ++++++++++++++++++++++ 1 file changed, 70 insertions(+) create mode 100644 baseline/nanogpt_one_head/tpu_high_dose.sh diff --git a/baseline/nanogpt_one_head/tpu_high_dose.sh b/baseline/nanogpt_one_head/tpu_high_dose.sh new file mode 100644 index 0000000..db4a4c1 --- /dev/null +++ b/baseline/nanogpt_one_head/tpu_high_dose.sh @@ -0,0 +1,70 @@ +#!/usr/bin/env bash +# Separate high-dose launcher. No cloud resources or historical results modified. +set -Eeuo pipefail +SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" +CODE="$(git -C "$SCRIPT_DIR" rev-parse --show-toplevel)" +ROOT="${RG_HIGH_DOSE_ROOT:-/mnt/disks/rg-data/fineweb-highdose-rawalpha-v1}" +DATA="${RG_HIGH_DOSE_DATA:-/mnt/disks/rg-data/rg-nanogpt-one-head/data}" +PREVIOUS="${RG_HIGH_DOSE_PREVIOUS:-/mnt/disks/rg-data/fineweb-memorization-tpu-v1}" +BLOCK="${RG_HIGH_DOSE_BLOCK:-v5e4-highdose-rawalpha-fp32}" +SESSION="ww_highdose" +USER_ROOT="$(getent passwd "$(id -u)" | cut -d: -f6)" +ENV_FILE="$USER_ROOT/.config/rg_optimizers/tpu_env.sh" +[[ -f "$ENV_FILE" ]] || { echo "STOP: activate the existing TPU environment first: $ENV_FILE"; exit 1; } +source "$ENV_FILE" +export PYTHONPATH="$SCRIPT_DIR/src" +export PYTHONDONTWRITEBYTECODE=1 +MODULE="rg_nanogpt_one_head.tpu_high_dose_sweep" +ARGS=(--code "$CODE" --root "$ROOT" --data-root "$DATA" --hardware-block "$BLOCK" --previous-root "$PREVIOUS") +CONTROL="${ROOT}.control" +case "${1:-start}" in + plan|run) + exec python3 -u -m "$MODULE" "$1" "${ARGS[@]}" + ;; + status) + exec python3 -u -m "$MODULE" status --root "$ROOT" + ;; + tail) + [[ -f "$CONTROL/latest_log" ]] || { echo "No launch log yet."; exit 1; } + exec tail -n 60 -f "$(cat "$CONTROL/latest_log")" + ;; + start) + command -v tmux >/dev/null || { echo "STOP: tmux is missing."; exit 1; } + if tmux has-session -t "=$SESSION" 2>/dev/null; then + echo "Existing $SESSION session preserved. Inspect with: bash '$0' status" + exit 0 + fi + # Check the mount before writing any logs or pretending boot storage is durable. + mountpoint -q /mnt/disks/rg-data || { echo "STOP: persistent disk is not mounted."; exit 1; } + mkdir -p "$CONTROL" + STAMP="$(date -u +%Y%m%dT%H%M%S%N)" + PLANLOG="$CONTROL/plan-$STAMP.log" + if python3 -u -m "$MODULE" plan "${ARGS[@]}" > "$PLANLOG" 2>&1; then + tail -n 1 "$PLANLOG" + else + rc=$? + tail -n 30 "$PLANLOG" + echo "STOP: preflight failed; training was NOT started. Log: $PLANLOG" + exit "$rc" + fi + LOG="$CONTROL/run-$STAMP.log" + printf '%s\n' "$LOG" > "$CONTROL/latest_log" + printf -v RUN '%q ' python3 -u -m "$MODULE" run "${ARGS[@]}" + printf -v QUOTED_LOG '%q' "$LOG" + PIPE="$RUN 2>&1 | tee -a $QUOTED_LOG" + printf -v COMMAND 'exec bash -o pipefail -c %q' "$PIPE" + tmux new-session -d -s "$SESSION" "$COMMAND" + echo "Started detached tmux session: $SESSION" + echo "Supervisor log: $LOG" + echo "Worker progress is printed every 60 seconds. Individual logs are under $ROOT/logs." + echo "Status: bash '$0' status" + echo "Live log: bash '$0' tail" + echo "No TPU lifetime was changed; this program stops after five full runs." + sleep 3 + tail -n 12 "$LOG" 2>/dev/null || true + ;; + *) + echo "Usage: bash $0 {plan|start|run|status|tail}" + exit 2 + ;; +esac From b726665c801a22942193cdc69943e556a17f74d5 Mon Sep 17 00:00:00 2001 From: Charles Martin Date: Tue, 29 Sep 2026 22:38:59 -0700 Subject: [PATCH 6/8] Test high-dose protocol, exact per-seed exposures, four-chip dispatch, and 48-canary scorer parity --- .../tests/test_tpu_high_dose_sweep.py | 172 ++++++++++++++++++ 1 file changed, 172 insertions(+) create mode 100644 baseline/nanogpt_one_head/tests/test_tpu_high_dose_sweep.py diff --git a/baseline/nanogpt_one_head/tests/test_tpu_high_dose_sweep.py b/baseline/nanogpt_one_head/tests/test_tpu_high_dose_sweep.py new file mode 100644 index 0000000..ed06992 --- /dev/null +++ b/baseline/nanogpt_one_head/tests/test_tpu_high_dose_sweep.py @@ -0,0 +1,172 @@ +from collections import Counter +from contextlib import nullcontext +from copy import deepcopy +import json +from pathlib import Path +from types import SimpleNamespace + +import pytest +import torch + +from rg_nanogpt_one_head import tpu_high_dose_sweep as high + +CODE = Path(__file__).resolve().parents[3] + + +def config(): + return high.load_study(CODE) + + +def reference(): + high.load_study(CODE) # registers the existing MuonClip extension + from rg_nanogpt_one_head.config import load_config + return load_config(CODE / high.CONFIG_DIR / "harmful_memorization_0pct.yaml") + + +def test_full_study_keeps_model_steps_and_raw_primary(): + cfg = config() + from rg_nanogpt_one_head.config import max_steps, epoch_step_map, tokens_per_step + assert cfg["model"]["vocab_size"] == 50257 + assert tokens_per_step(cfg) == 8192 + assert max_steps(cfg, 80000000) == 39063 + assert len(epoch_step_map(cfg, 80000000)) == 33 + assert cfg["weightwatcher"]["require_raw_alpha"] is True + assert cfg["weightwatcher"]["fix_fingers"] == "clip_xmax" # raw also retained + + +def test_five_jobs_and_four_independent_chips(): + jobs = high.tasks() + assert len(jobs) == 5 + assert Counter(t.chip for t in jobs) == {0: 2, 1: 1, 2: 1, 3: 1} + assert [t.seed for t in jobs] == high.SEEDS + assert {t.load for t in jobs} == {"highdose"} + assert {t.optimizer for t in jobs} == {"muon_clip"} + + +@pytest.mark.parametrize("section,key,value", [ + ("model", "n_embd", 256), ("training", "batch_size", 8), + ("training", "target_epochs", 2), ("training", "epoch_interval", 0.25), + ("memorization", "doses", [0, 64]), + ("memorization", "acquisition_fraction", 1.0), + ("memorization", "harmful_load_fraction", 0.1), + ("weightwatcher", "require_raw_alpha", False), + ("runtime", "deterministic_algorithms", False), +]) +def test_unregistered_changes_are_rejected(section, key, value): + cfg = config() + cfg[section][key] = value + with pytest.raises(ValueError, match="unregistered"): + high.validate_study(cfg, reference()) + + +@pytest.mark.parametrize("seed", high.SEEDS) +def test_exact_doses_and_no_unexposed_injection(tmp_path, seed): + from rg_nanogpt_one_head.random_canaries import RandomCanaryExperiment + experiment = RandomCanaryExperiment(config(), seed=seed, total_steps=39063, run_dir=tmp_path) + counts = Counter(item["id"] for item in experiment.schedule.values()) + assert len(experiment.canaries) == 48 + assert len(experiment.schedule) == 15872 + for canary in experiment.canaries: + assert counts[canary["id"]] == canary["dose"] + assert all(0 <= step < 19532 and 0 <= micro < 8 and 0 <= row < 4 + for step, micro, row in experiment.schedule) + manifest = json.loads((tmp_path / "random_canary_manifest.json").read_text()) + assert manifest["harmful_bank_size"] == 0 + assert manifest["harmful_presentations"] == 0 + + +def previous(root, n=25): + tasks = [high.base.asdict(t) for t in high.base.build_tasks()] + (root / "receipts").mkdir(parents=True) + (root / "tpu_sweep_plan.json").write_text(json.dumps({"tasks": tasks})) + for t in tasks[:n]: + done = {"completed": True, "optimizer_steps": 39063, "seed": t["seed"], + "optimizer": t["optimizer"], "fingerprint": f"fp-{t['index']}"} + p = root / f"load_{t['load']}" / "results" / t["optimizer"] / f"seed_{t['seed']}" + p.mkdir(parents=True) + (p / "run_complete.json").write_text(json.dumps(done)) + (root / "receipts" / f"task_{t['index']:03d}.json").write_text( + json.dumps({"task": t, "completion": done})) + return tasks + + +def test_previous_campaign_guard_is_read_only(tmp_path): + previous(tmp_path) + before = {str(p): p.read_bytes() for p in tmp_path.rglob("*") if p.is_file()} + high.require_previous_complete(tmp_path) + after = {str(p): p.read_bytes() for p in tmp_path.rglob("*") if p.is_file()} + assert before == after + + +def test_pending_old_job_blocks_launch(tmp_path): + previous(tmp_path, n=24) + with pytest.raises(RuntimeError, match="24/25"): + high.require_previous_complete(tmp_path) + + +def test_receipt_mismatch_blocks_launch(tmp_path): + previous(tmp_path) + p = tmp_path / "receipts/task_000.json" + receipt = json.loads(p.read_text()) + receipt["completion"]["fingerprint"] = "not-the-checkpoint" + p.write_text(json.dumps(receipt)) + with pytest.raises(RuntimeError, match="24/25"): + high.require_previous_complete(tmp_path) + + +def test_plan_can_be_consumed_by_qualified_worker(tmp_path, monkeypatch): + prior = tmp_path / "old" + previous(prior) + from rg_nanogpt_one_head import data + monkeypatch.setattr(data, "validate_prepared_data", lambda *args: {"splits": {"train": 80000000}}) + monkeypatch.setattr(high.base, "source_info", lambda path: {"commit": "unit-test"}) + monkeypatch.setattr(high.base, "check_root", lambda *args: None) + args = SimpleNamespace(code=CODE, root=tmp_path / "new", data_root=tmp_path / "data", + previous_root=prior, hardware_block="test-block") + plan = high.make(args) + assert plan["canary_batch_size"] == 48 + assert plan["scientific_hypothesis"]["primary_spectral_variable"] == "alpha_raw" + for task in high.tasks(): + cfg = high.base.task_config(plan, task) + assert cfg["memorization"]["doses"] == high.DOSES + assert cfg["runtime"]["tpu_memorization_execution"]["canary_batch_size"] == 48 + assert cfg["training"]["target_epochs"] == 4.0 + + +def test_original_executor_dispatches_highdose_plan_on_registered_chips(tmp_path, monkeypatch): + calls = [] + class Process: + def __init__(self, command, **kwargs): + calls.append((command, kwargs)) + self.pid = 1000 + len(calls) + def wait(self): + return 0 + monkeypatch.setattr(high.base.subprocess, "Popen", Process) + monkeypatch.setattr(high.base, "exclusive_lock", lambda path: nullcontext()) + plan = {"root": str(tmp_path), "code": str(CODE), "chips": high.CHIPS, + "hardware_block": "cpu-contract-test", "tasks": [high.base.asdict(t) for t in high.tasks()]} + p = tmp_path / "tpu_sweep_plan.json" + p.write_text(json.dumps(plan)) + assert high.base.run_sweep(p, 0) == 0 + assert len(calls) == 5 + assert Counter(v[1]["env"]["TPU_VISIBLE_CHIPS"] for v in calls) == {"0": 2, "1": 1, "2": 1, "3": 1} + assert all("rg_nanogpt_one_head.tpu_memorization_sweep" in command for command, _ in calls) + + +def test_batched_48_scorer_against_scalar_reference(tmp_path): + from rg_nanogpt_one_head.model import GPT, GPTConfig + from rg_nanogpt_one_head.random_canaries import RandomCanaryExperiment + from rg_nanogpt_one_head.canary_eval_batched import score_batch + torch.set_num_threads(1) + torch.manual_seed(1337) + cfg = config() + # Small CPU model tests padding/evaluator semantics, not TPU throughput. + cfg["model"].update(vocab_size=97, n_embd=16) + model = GPT(GPTConfig(**cfg["model"])).eval() + experiment = RandomCanaryExperiment(cfg, seed=1337, total_steps=39063, run_dir=tmp_path) + tokens = torch.stack([item["tokens"] for item in experiment.canaries]) + values = score_batch(model, tokens, prefix=64, suffix=32, device=torch.device("cpu")) + assert values.shape == (48, 4) + for i in (0, 7, 40, 47): + scalar = experiment._score_one(model, tokens[i], torch.device("cpu")) + assert values[i].tolist() == pytest.approx(scalar, abs=1e-5) From 9a7ee98f18fd3486e05e5593d0938a10bd031766 Mon Sep 17 00:00:00 2001 From: Charles Martin Date: Tue, 29 Sep 2026 22:40:18 -0700 Subject: [PATCH 7/8] Add CPU CI for high-dose protocol and inherited TPU executor contracts --- .github/workflows/tpu-high-dose-tests.yml | 25 +++++++++++++++++++++++ 1 file changed, 25 insertions(+) create mode 100644 .github/workflows/tpu-high-dose-tests.yml diff --git a/.github/workflows/tpu-high-dose-tests.yml b/.github/workflows/tpu-high-dose-tests.yml new file mode 100644 index 0000000..792e8b7 --- /dev/null +++ b/.github/workflows/tpu-high-dose-tests.yml @@ -0,0 +1,25 @@ +name: TPU high-dose CPU tests +on: + push: + branches: ['chatgpt/tpu-high-dose-memorization-20260930'] + pull_request: + paths: + - 'baseline/nanogpt_one_head/**' + - 'baseline/experiments/nanogpt_one_head_2026_08_21_baseline/configs/high_dose_memorization_raw_alpha.yaml' + - '.github/workflows/tpu-high-dose-tests.yml' +permissions: + contents: read +jobs: + high-dose-contracts: + runs-on: ubuntu-latest + timeout-minutes: 20 + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: '3.10' + - run: python -m pip install --upgrade pip + - run: python -m pip install 'torch==2.6.0' --index-url https://download.pytorch.org/whl/cpu + - run: python -m pip install -e './baseline/nanogpt_one_head[dev]' + - run: python -m pytest -q baseline/nanogpt_one_head/tests/test_tpu_high_dose_sweep.py baseline/nanogpt_one_head/tests/test_tpu_memorization_sweep.py baseline/nanogpt_one_head/tests/test_tpu_qualify_full.py + - run: bash -n baseline/nanogpt_one_head/tpu_high_dose.sh From d6c87ea57502afeed24fbd1008c71cdf4525a60d Mon Sep 17 00:00:00 2001 From: Charles Martin Date: Tue, 29 Sep 2026 22:41:41 -0700 Subject: [PATCH 8/8] Document five-seed high-dose raw-alpha protocol, detached launch, and analysis limitations --- baseline/nanogpt_one_head/TPU_HIGH_DOSE.md | 98 ++++++++++++++++++++++ 1 file changed, 98 insertions(+) create mode 100644 baseline/nanogpt_one_head/TPU_HIGH_DOSE.md diff --git a/baseline/nanogpt_one_head/TPU_HIGH_DOSE.md b/baseline/nanogpt_one_head/TPU_HIGH_DOSE.md new file mode 100644 index 0000000..548b3e9 --- /dev/null +++ b/baseline/nanogpt_one_head/TPU_HIGH_DOSE.md @@ -0,0 +1,98 @@ +# High-dose canaries: raw-alpha / stronger-recall pilot + +This is a NEW five-seed experiment, not a continuation of the 25-run additional-bank load sweep. The existing model, optimizers, training engine, corpus, and historical configurations are unchanged. The new entry point reuses the qualified four-independent-chip executor, including its exact run identity checks, atomic finite checkpoints, pre-resume archives, and bounded retries. + +## Registered protocol + +- Same pinned FineWeb-Edu corpus: 80M train, 1M validation, 1M test tokens; GPT-2 vocabulary 50,257. +- Same one-block, one-head, width-128 model; context 256; FP32; MuonClip plus auxiliary AdamW; unchanged learning rates, clipping, and weight decay. +- Five seeds: 1337, 2027, 4099, 31415, 271828. Four simultaneous independent single-chip runs, then the fifth on chip 0 (queue lengths 2/1/1/1). +- Every run contains doses **0, 64, 128, 256, 512, 1024**, with eight independent random sequences per dose: **48 tracked canaries**. +- A dose is the exact number of presentations of each canary over the acquisition window, NOT percent load and NOT presentations per epoch. All 48 sequences are scored in one fixed-shape batch. +- 15,872 total tracked presentations; 625,024 acquisition sequence slots; about 2.54% replacement during acquisition. No extra untracked random bank. Dose-0 sequences are never injected. +- Injections use the existing scheduler for the first 19,532 of 39,063 updates (approximately epochs 0-2); epochs 2-4 measure retention. Every seed still sees 8,192 token positions per update. +- Spectral states every 0.125 nominal epoch: 33 states including initialization. All six hidden matrices retain **alpha_raw**, alpha_clip_xmax, fit diagnostics, ERG and randomized-null diagnostics. **alpha_raw is the primary variable; clipped alpha is secondary only.** Existing console output prints both. + +The target is to see whether increasing exact exposure produces substantial teacher-forced and free-running recall and whether that co-occurs with changes in raw alpha. Literal recall and a raw-alpha crossing are hypotheses, not guaranteed outcomes. + +## Interpretation and analysis + +Primary behavioral measures are mean suffix NLL, teacher-forced token accuracy, free-running token accuracy and exact 32-token continuation recall, relative to never-injected dose-0 canaries in the SAME model. Preserve absolute NLL/recall as well as contrasts. Primary spectral summaries are each matrix's raw alpha and the minimum VALID raw alpha across the six matrices, retaining the identity of the minimizing matrix. Do not treat failed-fit sentinels as small alpha. Exclude initialization from association plots but preserve it for baseline changes. + +Match spectral and behavioral measurements by exact optimizer step. Separate acquisition from retention, report seed-level effects, and control for common training-time trends. Checkpoint and layer rows are not independent replications. + +All dose groups coexist in each model: there is one spectrum per matrix/checkpoint, NOT a separate spectrum for each dose. This design tests within-model dose response and whether strong recall can occur without raw alpha below 2. It does not by itself identify a causal between-model dose -> spectrum effect. A separate randomized between-run dose intervention would be needed for that claim. Do not pool these results with the prior load sweep as the same protocol; the changed dose inventory also changes which random sequences are assigned to the dose labels. + +## Before starting on the existing TPU + +Use the existing installed TPU environment; do not reinstall PyTorch or update the actively running checkout. Clone this branch into a SEPARATE directory on the persistent disk. The default launcher requires all 25 old jobs to have matching saved completion receipts. If it says 24/25, wait and rerun start later; it does not stop that last old job. + +The original corpus is reused read-only from: + +`/mnt/disks/rg-data/rg-nanogpt-one-head/data` + +The old run root is read only for the completion gate: + +`/mnt/disks/rg-data/fineweb-memorization-tpu-v1` + +New results and logs: + +```text +/mnt/disks/rg-data/fineweb-highdose-rawalpha-v1/ + tpu_sweep_plan.json + logs/task_*.log + receipts/task_*.json + status/task_*.json + load_highdose/results/muon_clip/seed_*/ + metrics.csv + random_canary_manifest.json + random_canary_metrics.csv + spectral/layers.csv + checkpoint_initial.pt + checkpoint_latest.pt + checkpoint_best.pt + checkpoint_final.pt + run_complete.json + highdose_exit.json +/mnt/disks/rg-data/fineweb-highdose-rawalpha-v1.control/ + plan-*.log + run-*.log + latest_log +``` + +## Commands inside the TPU VM (not Cloud Shell) + +Use a new, clean checkout pinned to the tested high-dose commit; the following path is separate from rg_optimizers_full: + +```bash +CODE=/mnt/disks/rg-data/rg_optimizers_highdose +# After cloning/checking out the exact high-dose commit: +bash "$CODE/baseline/nanogpt_one_head/tpu_high_dose.sh" start +``` + +`start` validates the old completion receipts, current source, exact protocol, verified corpus and persistent placement. It saves the verbose plan to disk, prints a short summary, and launches detached tmux session **ww_highdose** with persistent output logging from the first process message. It does not create a nested interactive tmux client. A session with the same name is preserved, never killed. Worker/seed progress is printed every 60 seconds. + +```bash +bash "$CODE/baseline/nanogpt_one_head/tpu_high_dose.sh" status +bash "$CODE/baseline/nanogpt_one_head/tpu_high_dose.sh" tail +``` + +After confirming four workers are advancing, logging out of SSH is safe. Ctrl-C when following `tail` stops only that log viewer; do not send Ctrl-C to the training pane. To inspect the pane, `tmux attach -d -t ww_highdose`; detach with Ctrl-b then d. + +A failed/interrupted high-dose run is resumed by the same start command and the same frozen source/options after its prior session exits. The inherited executor checks identity and archives partial state before rollback. There is no overwrite option. Changing machines or Python/package versions can correctly fail identity validation; do not bypass it. + +Defaults may be overridden BEFORE first launch using RG_HIGH_DOSE_ROOT, RG_HIGH_DOSE_DATA, RG_HIGH_DOSE_PREVIOUS, and RG_HIGH_DOSE_BLOCK. The wrapper expects the persistent volume mounted at /mnt/disks/rg-data. Do not change options in an already registered root. + +## Timing and lifetime + +At previous measured throughput, two seed-runtime waves suggest roughly 2.5-3 hours; allow about 3-4 hours initially for the doubled spectral cadence and 48-canary evaluations. This is a planning estimate, not a measured high-dose benchmark. + +The code DOES NOT extend, restart, create, delete or shut down the TPU. Its pre-existing Flex-Start termination time still applies; verify enough time remains before launching. tmux survives an SSH disconnect, not VM termination. Persistent data survive independently of the TPU node as long as the data disk is retained. + +## Validation + +The high-dose test suite covers configuration invariants, five-seed/four-chip task construction, all five exact exposure schedules, no injected dose-0 controls, read-only old-completion checks, inherited worker dispatch, and 48-canary batched/scalar metric parity on a small CPU model. The shared executor has already been exercised on the user's four-chip TPU by the prior experiment. The NEW high-dose scientific runs have not been executed by these CPU tests. + +```bash +PYTHONPATH=baseline/nanogpt_one_head/src python3 -m pytest -q baseline/nanogpt_one_head/tests/test_tpu_high_dose_sweep.py +```