""" SELF-IMPROVING FROZEN LLM via an EXTERNAL verifier (the deployable payoff, for real) -- the self-improvement map says: with an EXTERNAL, independent verifier a model self-improves without weight updates or labels. Here it is with a REAL frozen LLM (ollama qwen2.5vl:7b) on a task it often gets WRONG (long multiplication), verified by EXECUTION (compute a*b -- exact, external, uncorrelated with the model's errors). Loop over a stream of problems: retrieve the k nearest VERIFIED worked-solutions from a never-forget memory as few-shot context -> the frozen LLM solves the new problem -> EXECUTION verifies -> if correct, store its worked solution. As verified exemplars accumulate, in-context accuracy COMPOUNDS -- no gradients, no labels, only the execution signal. Baseline: same stream, NO memory (0-shot each) -> flat. Measures rolling accuracy over the stream. Run: python self_improve_llm_verifier.py [--n 120 --digits 3 --k 4 --model qwen2.5vl:7b] """ import argparse, json, os, re, time, urllib.request import numpy as np OLLAMA = "http://127.0.0.1:11434/api/generate" def gen(prompt, model, npred=220): req = urllib.request.Request(OLLAMA, data=json.dumps({"model": model, "prompt": prompt, "stream": False, "options": {"temperature": 0.0, "num_predict": npred}}).encode(), headers={"Content-Type": "application/json"}) return json.loads(urllib.request.urlopen(req, timeout=180).read())["response"] def parse_ans(txt): if "####" in txt: txt = txt.split("####")[-1] m = re.findall(r"-?\d[\d,]*", txt) return int(m[-1].replace(",", "")) if m else None PROMPT = ("You multiply two integers. Work step by step briefly, then output the final answer on a new line as: #### .\n\n") def main(): ap = argparse.ArgumentParser(); ap.add_argument("--n", type=int, default=120); ap.add_argument("--digits", type=int, default=3) ap.add_argument("--k", type=int, default=4); ap.add_argument("--model", default="qwen2.5vl:7b"); ap.add_argument("--seed", type=int, default=0) ap.add_argument("--control", action="store_true") # also run the UNVERIFIED-exemplar control (store all attempts, ignore execution) to isolate verification a = ap.parse_args(); rng = np.random.default_rng(a.seed); t0 = time.time() lo, hi = 10 ** (a.digits - 1), 10 ** a.digits - 1 probs = [(int(rng.integers(lo, hi)), int(rng.integers(lo, hi))) for _ in range(a.n)] print(f"SELF-IMPROVING LLM [{a.model}] {a.digits}x{a.digits}-digit multiplication, EXECUTION verifier, never-forget verified-exemplar memory; {a.n} problems, k={a.k}", flush=True) def run(mode): # mode: 'none' (0-shot) | 'verified' (store execution-verified only) | 'all' (store every attempt = unverified control) mem = []; correct = [] for i, (x, y) in enumerate(probs): ex = "" if mode != "none" and mem: for (mx, my, sol) in mem[-a.k:]: ex += f"Question: {mx} * {my} =\n{sol}\n\n" try: out = gen(PROMPT + ex + f"Question: {x} * {y} =\n", a.model) except Exception: out = "" pred = parse_ans(out); ok = (pred == x * y); correct.append(ok) if mode == "verified" and ok: # store ONLY execution-verified solutions sol = out.strip(); sol += "" if "####" in sol else f"\n#### {x*y}"; mem.append((x, y, sol)) elif mode == "all" and pred is not None: # store EVERY attempt (unverified control -- includes wrong ones) mem.append((x, y, out.strip())) if (i + 1) % 20 == 0: print(f" {mode:8s}[{i+1}/{a.n}] rolling-acc(last20) {np.mean(correct[-20:]):.2f} mem {len(mem)} ({time.time()-t0:.0f}s)", flush=True) return np.array(correct), len(mem) print(" --- WITH never-forget verified-exemplar memory ---", flush=True); mem_c, nmem = run("verified") print(" --- BASELINE: no memory (0-shot each) ---", flush=True); base_c, _ = run("none") ctrl_c = None if a.control: print(" --- CONTROL: UNVERIFIED exemplars (store all attempts, incl. wrong) ---", flush=True); ctrl_c, _ = run("all") def half(c): return np.mean(c[:len(c) // 2]), np.mean(c[len(c) // 2:]) m1, m2 = half(mem_c); b1, b2 = half(base_c) print(f"\n[self-improve-llm-verifier] {a.model}, {a.digits}x{a.digits} mult, {a.n} problems:", flush=True) print(f" memory : 1st-half acc {m1:.2f} -> 2nd-half {m2:.2f} (verified exemplars accumulated: {nmem})", flush=True) print(f" no-mem : 1st-half acc {b1:.2f} -> 2nd-half {b2:.2f}", flush=True) if ctrl_c is not None: print(f" UNVERIFIED-ctrl: overall {np.mean(ctrl_c):.2f} (few-shot with the model's OWN possibly-wrong solutions)", flush=True) print(f" >> ISOLATE VERIFICATION: verified {np.mean(mem_c):.2f} vs UNVERIFIED {np.mean(ctrl_c):.2f} vs no-mem {np.mean(base_c):.2f} -- {'VERIFICATION is the driver (verified >> unverified)' if np.mean(mem_c) > np.mean(ctrl_c) + 0.05 else 'gain is just few-shot, not verification'}", flush=True) print(f" >> self-improvement via execution-verification: overall memory {np.mean(mem_c):.2f} vs base {np.mean(base_c):.2f} (+{np.mean(mem_c)-np.mean(base_c):.2f})", flush=True) out = os.path.join(os.path.dirname(os.path.abspath(__file__)), "closed_form_neat_outputs", "metrics_self_improve_llm_verifier.json") json.dump({"mem": mem_c.tolist(), "base": base_c.tolist(), "ctrl": (ctrl_c.tolist() if ctrl_c is not None else None), "nmem": nmem, "digits": a.digits, "model": a.model}, open(out, "w"), indent=2) print(f"Wrote {out} ({time.time()-t0:.0f}s)", flush=True) if __name__ == "__main__": main()