Archived implementation
This is the complete Python source code, displayed verbatim and not executed by this page.
Source code
Complete archived implementation, displayed verbatim. It is not executed by this page.
"""
SELF-IMPROVING FROZEN LLM via an EXTERNAL verifier (the deployable payoff, for real) -- the self-improvement map says: with an EXTERNAL,
independent verifier a model self-improves without weight updates or labels. Here it is with a REAL frozen LLM (ollama qwen2.5vl:7b) on a task
it often gets WRONG (long multiplication), verified by EXECUTION (compute a*b -- exact, external, uncorrelated with the model's errors). Loop over
a stream of problems: retrieve the k nearest VERIFIED worked-solutions from a never-forget memory as few-shot context -> the frozen LLM solves the
new problem -> EXECUTION verifies -> if correct, store its worked solution. As verified exemplars accumulate, in-context accuracy COMPOUNDS -- no
gradients, no labels, only the execution signal. Baseline: same stream, NO memory (0-shot each) -> flat. Measures rolling accuracy over the stream.
Run: python self_improve_llm_verifier.py [--n 120 --digits 3 --k 4 --model qwen2.5vl:7b]
"""
import argparse, json, os, re, time, urllib.request
import numpy as np
OLLAMA = "http://127.0.0.1:11434/api/generate"
def gen(prompt, model, npred=220):
req = urllib.request.Request(OLLAMA, data=json.dumps({"model": model, "prompt": prompt, "stream": False,
"options": {"temperature": 0.0, "num_predict": npred}}).encode(), headers={"Content-Type": "application/json"})
return json.loads(urllib.request.urlopen(req, timeout=180).read())["response"]
def parse_ans(txt):
if "####" in txt: txt = txt.split("####")[-1]
m = re.findall(r"-?\d[\d,]*", txt)
return int(m[-1].replace(",", "")) if m else None
PROMPT = ("You multiply two integers. Work step by step briefly, then output the final answer on a new line as: #### <integer>.\n\n")
def main():
ap = argparse.ArgumentParser(); ap.add_argument("--n", type=int, default=120); ap.add_argument("--digits", type=int, default=3)
ap.add_argument("--k", type=int, default=4); ap.add_argument("--model", default="qwen2.5vl:7b"); ap.add_argument("--seed", type=int, default=0)
ap.add_argument("--control", action="store_true") # also run the UNVERIFIED-exemplar control (store all attempts, ignore execution) to isolate verification
a = ap.parse_args(); rng = np.random.default_rng(a.seed); t0 = time.time()
lo, hi = 10 ** (a.digits - 1), 10 ** a.digits - 1
probs = [(int(rng.integers(lo, hi)), int(rng.integers(lo, hi))) for _ in range(a.n)]
print(f"SELF-IMPROVING LLM [{a.model}] {a.digits}x{a.digits}-digit multiplication, EXECUTION verifier, never-forget verified-exemplar memory; {a.n} problems, k={a.k}", flush=True)
def run(mode): # mode: 'none' (0-shot) | 'verified' (store execution-verified only) | 'all' (store every attempt = unverified control)
mem = []; correct = []
for i, (x, y) in enumerate(probs):
ex = ""
if mode != "none" and mem:
for (mx, my, sol) in mem[-a.k:]: ex += f"Question: {mx} * {my} =\n{sol}\n\n"
try: out = gen(PROMPT + ex + f"Question: {x} * {y} =\n", a.model)
except Exception: out = ""
pred = parse_ans(out); ok = (pred == x * y); correct.append(ok)
if mode == "verified" and ok: # store ONLY execution-verified solutions
sol = out.strip(); sol += "" if "####" in sol else f"\n#### {x*y}"; mem.append((x, y, sol))
elif mode == "all" and pred is not None: # store EVERY attempt (unverified control -- includes wrong ones)
mem.append((x, y, out.strip()))
if (i + 1) % 20 == 0:
print(f" {mode:8s}[{i+1}/{a.n}] rolling-acc(last20) {np.mean(correct[-20:]):.2f} mem {len(mem)} ({time.time()-t0:.0f}s)", flush=True)
return np.array(correct), len(mem)
print(" --- WITH never-forget verified-exemplar memory ---", flush=True); mem_c, nmem = run("verified")
print(" --- BASELINE: no memory (0-shot each) ---", flush=True); base_c, _ = run("none")
ctrl_c = None
if a.control:
print(" --- CONTROL: UNVERIFIED exemplars (store all attempts, incl. wrong) ---", flush=True); ctrl_c, _ = run("all")
def half(c): return np.mean(c[:len(c) // 2]), np.mean(c[len(c) // 2:])
m1, m2 = half(mem_c); b1, b2 = half(base_c)
print(f"\n[self-improve-llm-verifier] {a.model}, {a.digits}x{a.digits} mult, {a.n} problems:", flush=True)
print(f" memory : 1st-half acc {m1:.2f} -> 2nd-half {m2:.2f} (verified exemplars accumulated: {nmem})", flush=True)
print(f" no-mem : 1st-half acc {b1:.2f} -> 2nd-half {b2:.2f}", flush=True)
if ctrl_c is not None:
print(f" UNVERIFIED-ctrl: overall {np.mean(ctrl_c):.2f} (few-shot with the model's OWN possibly-wrong solutions)", flush=True)
print(f" >> ISOLATE VERIFICATION: verified {np.mean(mem_c):.2f} vs UNVERIFIED {np.mean(ctrl_c):.2f} vs no-mem {np.mean(base_c):.2f} -- {'VERIFICATION is the driver (verified >> unverified)' if np.mean(mem_c) > np.mean(ctrl_c) + 0.05 else 'gain is just few-shot, not verification'}", flush=True)
print(f" >> self-improvement via execution-verification: overall memory {np.mean(mem_c):.2f} vs base {np.mean(base_c):.2f} (+{np.mean(mem_c)-np.mean(base_c):.2f})", flush=True)
out = os.path.join(os.path.dirname(os.path.abspath(__file__)), "closed_form_neat_outputs", "metrics_self_improve_llm_verifier.json")
json.dump({"mem": mem_c.tolist(), "base": base_c.tolist(), "ctrl": (ctrl_c.tolist() if ctrl_c is not None else None), "nmem": nmem, "digits": a.digits, "model": a.model}, open(out, "w"), indent=2)
print(f"Wrote {out} ({time.time()-t0:.0f}s)", flush=True)
if __name__ == "__main__":
main()