เจาะลึก RRSI: คู่มือ AI Agent ปรับปรุงตัวเองได้จาก Google Research

บทช่วยสอนนี้จะพาไปใช้งาน RRSI (Regularized Recursive Self-Improvement) ซึ่งเป็นวิธีการที่ช่วยให้ LLM agent สามารถสร้าง harness, prompt, tool, memory และ sub-agent ของตัวเองขึ้นมาใหม่รอบ frozen model ได้โดยไม่เกิด overfitting ต่อภารกิจเดิม
กระบวนการของ RRSI แบบเต็มรูปแบบจะใช้ Claude Opus บน Vertex AI ร่างการแก้ไขและทดสอบใน Docker benchmark ซึ่งต้องใช้ทรัพยากรสูง แต่ส่วนสำคัญที่สุดคือการใช้ Python ปกติกำหนดกฎการเลือกเก็บผลการแก้ไข เราจะทดลองติดตั้ง package จาก repository ทางการเพื่อศึกษา estimator, calibrated noise band, อัลกอริทึมการเลือก, edit budget, และระบบป้องกัน Social Engineering ผ่านสภาพแวดล้อมจำลอง เพื่อเปรียบเทียบการตัดสินใจของ RRSI กับการค้นหาแบบทั่วไปที่ไม่มีการควบคุม (unregularized)
import os
import sys
import json
import math
import copy
import random
import tempfile
import textwrap
import traceback
import subprocess
import statistics as st
from pathlib import Path
RESULTS = {}
def banner(title):
print("\n" + "=" * 78)
print(title)
print("=" * 78)
def section(name):
def wrap(fn):
def run(*a, **kw):
banner(name)
try:
out = fn(*a, **kw)
RESULTS[name] = out if isinstance(out, str) else "ok"
return out
except Exception as e:
RESULTS[name] = f"SKIPPED / FAILED -> {type(e).__name__}: {e}"
print(f"\n[!] {name} did not complete: {type(e).__name__}: {e}")
traceback.print_exc(limit=3)
return None
return run
return wrap
banner("1. Install RRSI and map the paper onto the code")
subprocess.run([sys.executable, "-m", "pip", "install", "-q", "git+https://github.com/google-research/rrsi.git@be50316e1db05914068a973f322770ef08ed7ba1"], check=True)
from importlib.metadata import version
from rrsi.config import RRSIConfig
from rrsi.evaluate import TaskResult, EvalResult, aggregate, evaluate
from rrsi.calibrate import calibrate
from rrsi.selection import Candidate, cost_rule, judge, select_round
from rrsi.schedule import edit_budget, budget_table
from rrsi.history import History, stall_flag, exploration
from rrsi.components import K, K_STR, normalize, novelty
from rrsi.critic import precheck, review
from rrsi.domain import Domain
print(f" rrsi {version('rrsi')} | anthropic {version('anthropic')} | Python {sys.version.split()[0]}")
print("\n RRSI evolves an agent's HARNESS (prompts, tools, memory, control flow, sub-agents) around a")
print(" frozen model. The full loop drafts edits with Claude Opus on Vertex AI and scores them in Docker")
print(" benchmarks. The part that decides which edits to KEEP is plain Python, and that is what we drive:")
for symbol, where in [
("S_hat, C_hat Eq. (estimate)", "rrsi.evaluate.aggregate"),
("delta noise band", "rrsi.calibrate.calibrate"),
("Algorithm 2 floor + cost rule", "rrsi.selection.judge / select_round"),
("b_t Eq. (anneal)", "rrsi.schedule.edit_budget"),
("Critic leakage screen", "rrsi.critic.precheck / review"),
("L_t, g_t, B_t history, yield, prune", "rrsi.history.History"),
("sigma_t, U_t stall + exploration", "rrsi.history.stall_flag / exploration"),
("nu structural novelty", "rrsi.components.novelty"),
]:
print(f" {symbol:40s} -> {where}")
CFG = RRSIConfig()
print(f"\n paper defaults: T={CFG.T} rounds, k={CFG.k} trials/task, m={CFG.m} candidates/round,"
f" b in [{CFG.b_min},{CFG.b_max}]")
print(f" beta0={CFG.beta0} beta1={CFG.beta1} w_s={CFG.w_s} w_c={CFG.w_c} w_n={CFG.w_n}"
f" delta_z={CFG.delta_z}")
print("\n Nothing below needs an API key, a GPU or a dataset download.")การติดตั้ง RRSI ทำได้โดยระบุ commit จาก google-research repository โดยตรงเนื่องจากยังไม่มีบน PyPI ในส่วนของ RRSIConfig จะทำหน้าที่เก็บค่าไฮเปอร์พารามิเตอร์ตามงานวิจัยดั้งเดิม เพื่อนำไปใช้ในฟังก์ชันต่างๆ เช่น การประมาณต้นทุนใน evaluate, การวัด noise band ใน calibrate และการตรวจสอบความปลอดภัยใน critic ซึ่งทั้งหมดนี้สามารถรันได้โดยไม่ต้องใช้ API key หรือ GPU
@section("2. Evaluate(H): a score and a cost, and why a crash counts as zero")
def estimator():
base = {
"task_000": TaskResult(rewards=[1, 1], tokens=[11_800, 12_400]),
"task_001": TaskResult(rewards=[1, 0], tokens=[15_100, 14_600]),
"task_002": TaskResult(rewards=[0, 0], tokens=[21_000, 19_500]),
}
ev = aggregate("H0", 2, base)
print(f" three tasks x k=2 trials -> S_hat = {ev.S:.3f} C_hat = {ev.C:,.0f} tokens/trial"
f" ({ev.n_expected} trials expected, {ev.missing} missing)")
crashy = dict(base)
crashy["task_002"] = TaskResult(rewards=[0.0, 0.0], tokens=[None, None], missing=2)
ev_crash = aggregate("crashy", 2, crashy)
dropped = {t: r for t, r in crashy.items() if not r.missing}
naive = sum(sum(r.rewards) for r in dropped.values()) / sum(len(r.rewards) for r in dropped.values())
print("\n A candidate crashes on the hardest task instead of failing it:")
print(f" an estimator that drops missing trials reports {naive:.3f} <- looks like a gain")
print(f" RRSI's aggregate (missing = 0, full denominator) {ev_crash.S:.3f} <- no reward for crashing")
print(f" ...and C_hat uses only recorded token counts: {ev_crash.C:,.0f}")
rubric = {"memo": TaskResult(rewards=[0.5, 1.0], weights=[10, 10]),
"brief": TaskResult(rewards=[0.0, 0.0], weights=[90, 90])}
ev_w = aggregate("rubric", 2, rubric)
print("\n Weighted rewards (Harvey LAB style: weight = number of rubric criteria):")
print(f" mean of per-task means = {st.mean(r.mean for r in rubric.values()):.3f}"
f" vs RRSI's S_hat = {ev_w.S:.3f} (fraction of all criteria passed)")
return f"crash scored {ev_crash.S:.3f} under RRSI vs {naive:.3f} if dropped"
estimator()RRSI วัดประสิทธิภาพจากสองค่าหลักคือ S (คะแนนเฉลี่ย) และ C (ต้นทุน token เฉลี่ย) โดยมีกลไกป้องกันที่น่าสนใจคือ การจัดการกับการทดสอบที่ล้มเหลว (crash) ซึ่ง RRSI จะนับคะแนนเป็นศูนย์ทันที แทนที่จะตัดออกจากการคำนวณ เพื่อป้องกันไม่ให้ candidate ที่ crash ในภารกิจยากๆ ดูเหมือนมีประสิทธิภาพสูงขึ้นเกินจริง นอกจากนี้ยังรองรับการให้คะแนนแบบถ่วงน้ำหนักตามเกณฑ์ rubric ที่ซับซ้อนได้อีกด้วย
FAMILIES = ["parse", "search", "edit", "test"]
H0 = {"skill": {f: 0.0 for f in FAMILIES}, "memo": [], "cost": 1.0}
def make_world(seed, n_evolve, n_heldout=80):
"""Tasks with a family and a difficulty. Evolve ids are task_NNN, held-out ids held_NNN."""
r = random.Random(seed)
world = {f"task_{i:03d}": (FAMILIES[i % 4], r.gauss(0, 1)) for i in range(n_evolve)}
world.update({f"held_{i:03d}": (FAMILIES[i % 4], r.gauss(0, 1)) for i in range(n_heldout)})
return world
def p_success(h, world, task):
"""The frozen policy: a logistic in harness skill minus task difficulty, or 0.97 if memorised."""
if task in h["memo"]:
return 0.97
family, difficulty = world[task]
return 1 / (1 + math.exp(-(0.3 + h["skill"][family] - difficulty)))
def run_trials(h, world, ids, k, rng):
return {t: TaskResult(rewards=[float(rng.random() < p_success(h, world, t)) for _ in range(k)],
tokens=[round(12_000 * h["cost"] * math.exp(rng.gauss(0, 0.08))) for _ in range(k)])
for t in ids}
def true_score(h, world, ids):
return sum(p_success(h, world, t) for t in ids) / len(ids)
def calibrated_delta(world, k, seed, repeats=3):
"""delta from `repeats` independent evaluations of the SAME harness H0."""
ids = [t for t in world if t.startswith("task_")]
rng = random.Random(10_000 + seed)
evals = [aggregate(f"base{j}", k, run_trials(H0, world, ids, k, rng)) for j in range(repeats)]
return calibrate(evals, z=CFG.delta_z, reps=200), evals
@section("3. The noise band: how far one harness's score moves between two evaluations")
def noise_band():
world = make_world(0, 40)
ids = [t for t in world if t.startswith("task_")]
rng = random.Random(1)
scores = [aggregate(f"H0#{j}", 2, run_trials(H0, world, ids, 2, rng)).S for j in range(6)]
print(f" H0 evaluated 6 times on 40 tasks x k=2: {[round(s, 3) for s in scores]}")
print(f" true expected score {true_score(H0, world, ids):.3f}; spread of the estimates"
f" {max(scores) - min(scores):.3f}. Nothing about the harness changed.")
print(f"\n calibrate() turns repeated evaluations into delta = z * sd(null dS), z = {CFG.delta_z}:")
print(f" {'evaluator':>12s} {'trials':>7s} {'delta':>8s} method")
deltas = {}
for n, k in [(40, 2), (100, 4), (400, 8)]:
cal, _ = calibrated_delta(make_world(0, n), k, seed=0)
deltas[f"{n}x{k}"] = cal["delta"]
print(f" {f'{n} x k={k}':>12s} {n * k:>7d} {cal['delta']:>8.4f} {cal['method']}")
print("\n The paper's calibrated bands are 0.017 (coding), 0.004 (workspace) and 0.020 (engineering).")
print(" A gain smaller than delta is indistinguishable from re-running the same harness, and")
print(" Algorithm 2 treats it that way. Remember the first row: it becomes the lesson of step 11.")
return "delta " + ", ".join(f"{k}={v:.3f}" for k, v in deltas.items())
noise_band()หัวใจสำคัญก่อนเริ่มปรับปรุงคือการวัด Noise Band เพื่อให้รู้ว่าคะแนนที่แกว่งไปมาเกิดจากโชคหรือการพัฒนาจริง โดยฟังก์ชัน calibrate จะคำนวณค่า delta จากการรันซ้ำหลายครั้ง หากคะแนนที่เพิ่มขึ้นน้อยกว่าค่า delta ระบบจะถือว่าไม่มีความแตกต่างจากการรันค่าเดิมซ้ำ ซึ่งช่วยลดความผิดพลาดในการเลือกโมเดลที่โชคดีเพียงชั่วคราว
def ev_at(S, C, job, n=200):
"""An EvalResult with exactly score S (in steps of 1/n) and C tokens per trial."""
hits = round(S * n)
return aggregate(job, 1, {f"task_{i:03d}": TaskResult(rewards=[1.0 if i < hits else 0.0], tokens=[C])
for i in range(n)})
INCUMBENT = ev_at(0.630, 10_000, "incumbent")
S_STAR, DELTA = 0.640, 0.020
@section("4. Algorithm 2, branch by branch: the floor, the cost rule and the noise band")
def algorithm_2():
print(f" incumbent S = {INCUMBENT.S:.3f}, C = {INCUMBENT.C:,.0f} best ever S* = {S_STAR} delta = {DELTA}")
print(f" floor = S* - delta = {S_STAR - DELTA:.3f}\n")
cases = [
("A slipped below the floor", 0.600, 10_000, ["prompt"], None),
("B real gain, pays for its tokens", 0.690, 12_000, ["prompt"], None),
("C real gain, far too expensive", 0.660, 25_000, ["subagent"], None),
("D slightly WORSE but cheaper", 0.625, 8_000, ["context_mgmt"], None),
("E in the band but costlier", 0.640, 11_500, ["prompt"], None),
("F in the band, new sub-agent", 0.630, 10_000, ["subagent"], None),
("F' in the band, prompt tweak", 0.630, 10_000, ["prompt"], None),
("G like B, breaks a domain guard", 0.690, 12_000, ["prompt"], ["valid-output rate fell"]),
]
rows = []
for label, S, C, comps, guards in cases:
cand = Candidate(label[:2].strip(), [{"id": "C1", "component": c} for c in comps], ev=ev_at(S, C, label))
dec = judge(cand, INCUMBENT, S_STAR, DELTA, CFG, incumbent_counts={}, guards=guards)
rows.append((label, dec))
print(f" {label:36s} S={S:.3f} C={C:>6,} {'ADMIT ' if dec.admissible else 'reject'}")
print(textwrap.indent(textwrap.fill(dec.reason, 88), " " * 6))
print("\n Three rules, in the order RRSI applies them:")
print(" 1. never fall below the best score ever seen, minus the noise band (A)")
print(" 2. a gain bigger than delta must pay for any extra tokens: dC <= 0.10 + 40*dS (B, C)")
print(" 3. inside the band scores are a tie, so prefer the cheaper harness, and let a")
print(" never-tried structural component break the tie (D, E, F, F')")
print(" D is the surprising one: RRSI admits a harness that scored LOWER than the incumbent.")
return f"{sum(d.admissible for _, d in rows)}/{len(rows)} admissible; D admitted at dS={rows[3][1].delta_S:+.3f}"
algorithm_2()Algorithm 2 ใช้กฎสามข้อในการตัดสินใจเลือกเก็บการแก้ไข: 1. คะแนนต้องไม่ต่ำกว่าเกณฑ์พื้นฐาน (floor) 2. หากคะแนนเพิ่มขึ้นต้องแลกมาด้วยต้นทุน token ที่สมเหตุสมผล 3. หากคะแนนอยู่ในช่วงเกณฑ์ Noise Band จะเลือกตัวที่ราคาถูกกว่าหรือมีความแปลกใหม่ทางโครงสร้าง กลไกนี้ทำให้ RRSI ยอมรับ candidate ที่ได้คะแนนน้อยกว่าเดิมแต่ประหยัดทรัพยากรมากขึ้น ซึ่งเป็นกลยุทธ์ระยะยาวที่มีประสิทธิภาพ
@section("5. One round of selection: the highest score does not always win")
def one_round():
c_hi = Candidate("C", [{"id": "C1", "component": "subagent"}], ev=ev_at(0.660, 25_000, "C"))
d_lo = Candidate("D", [{"id": "C1", "component": "context_mgmt"}], ev=ev_at(0.625, 8_000, "D"))
leak = Candidate("X", [{"id": "C1", "component": "memory"}], gate_failure="critic_reject")
winner, decisions = select_round([c_hi, d_lo, leak], INCUMBENT, S_STAR, DELTA, CFG, incumbent_counts={})
for d in decisions:
print(f" {d.variant}: {'admissible' if d.admissible else 'rejected '} {d.reason[:92]}")
new_star = max(S_STAR, winner.ev.S)
print(f"\n winner: {winner.variant} (S {winner.ev.S:.3f}, C {winner.ev.C:,.0f})")
print(f" the new incumbent scores {winner.ev.S - INCUMBENT.S:+.3f} vs the old one and costs"
f" {(winner.ev.C - INCUMBENT.C) / INCUMBENT.C:+.0%} tokens; S* stays {new_star:.3f}")
print("\n C scored highest and still lost: its +3pp does not pay for +150% tokens. D moved the")
print(" incumbent DOWN inside the noise band because it is 20% cheaper. Because S* only ever")
print(" rises, the floor never follows the incumbent down, so a chain of 'cheaper but slightly")
print(" worse' swaps cannot walk the score away. X never reached evaluation at all.")
return f"winner {winner.variant} at dS={winner.ev.S - INCUMBENT.S:+.3f}, dC={(winner.ev.C - INCUMBENT.C) / INCUMBENT.C:+.0%}"
one_round()ในการคัดเลือกรายรอบ Candidate ที่ได้คะแนนสูงสุดอาจไม่ใช่ผู้ชนะเสมอไป เช่น ในกรณีที่การเพิ่มคะแนนเพียงเล็กน้อยแต่แลกด้วยต้นทุนที่พุ่งสูงถึง 150% จะถูกปัดตกไป ระบบจะยึด S* (คะแนนดีที่สุดที่เคยทำได้) เป็นหลักเสมอ เพื่อป้องกันไม่ให้การเลือกของถูกที่คุณภาพต่ำลงเรื่อยๆ ทำให้ประสิทธิภาพโดยรวมถดถอย
@section("6. The proposal side: an edit budget that anneals from 4 toward 1")
def edit_budget_schedule():
table = budget_table(CFG.T, CFG.b_min, CFG.b_max)
print(" b_t = ceil(b_min + (b_max - b_min) * (1 + cos(pi t / T)) / 2)")
print(f" T={CFG.T}, b in [{CFG.b_min}, {CFG.b_max}]:")
print(" t : " + " ".join(f"{t:>2d}" for t in range(CFG.T)))
print(" b_t : " + " ".join(f"{b:>2d}" for b in table))
print(f" round {CFG.T} (after the run) -> {edit_budget(CFG.T, CFG.T, CFG.b_min, CFG.b_max)}")
print("\n Early candidates may bundle up to 4 coordinated edits. Note the ceil(): the cosine term is")
print(f" only exactly zero at t = T, so inside a {CFG.T}-round run the budget bottoms out at"
f" {min(table)}, not {CFG.b_min}.")
print(" Every edit in a bundle inherits the bundle's single measurement, so the shrinking budget is")
print(" what makes late history attributable to fewer components. It caps how MANY edits ride")
print(" together, never WHICH mechanisms the harness may eventually contain.")
return "budget " + "".join(str(b) for b in table)
edit_budget_schedule()ฝั่งการเสนอแก้ไขจะใช้ระบบ Annealed L0 Budget ซึ่งจะอนุญาตให้มีการแก้ไขพร้อมกันได้มากถึง 4 รายการในช่วงแรก และค่อยๆ ลดลงเหลือ 1 รายการในช่วงท้าย การลดงบประมาณนี้ช่วยให้ระบบสามารถระบุผลลัพธ์ย้อนกลับไปยังส่วนประกอบที่แก้ไขได้ชัดเจนยิ่งขึ้นในช่วงท้ายของการรัน
CRITIC_PATTERNS = [
(r"\btask_\d{3}\b", "hard-codes an evolve-set task id"),
(r"expected_output|grader|rubric\[", "reads the grader or the expected answer"),
]
COMPONENT_SIGNALS = [
("control_flow", [r"\bretry\(", r"max_attempts"]),
("context_mgmt", [r"compress_context", r"keep_last"]),
("config", [r"CONFIG\["]),
]
class ScreenOnly(Domain):
name = "screen"
critic_patterns = CRITIC_PATTERNS
briefs = {"critic": "A coding agent harness."}
@section("7. The critic's deterministic layer, and how edits are tagged")
def critic_and_tags():
dom = ScreenOnly()
diffs = {
"memorise answers": "+ memory = Memory('answers')\n+ memory.remember('task_007', cached_patch)",
"peek at the grader": "+ if os.path.exists('/grader/expected_output.txt'): return read_it()",
"leaked credential": "+ api_key = 'sk-live-0123456789abcdefghijkl'",
"empty diff": " ",
"general retry rule": "+ for attempt in range(max_attempts): result = retry(step)",
}
for label, diff in diffs.items():
try:
verdict = review(dom, diff, summary=label, targets_mode="evolve")
print(f" {label:20s} -> {verdict['verdict']:6s} {verdict['reasons']}")
except ZeroDivisionError as e:
print(f" {label:20s} -> passed the deterministic layer; the LLM layer raised"
f" ZeroDivisionError: {e}")
print("\n Gotcha: with RRSI_VERTEX_PROJECTS unset, rrsi.llm.generate computes `x % len(projects)`")
print(" outside its try block, so the helpful 'set RRSI_VERTEX_PROJECTS' error is never reached.")
print(" A clean diff is supposed to go to Claude for an intent review; here there is no Claude.")
print("\n Every edit is tagged with the component it touches, and a tag needs evidence in the diff:")
tag_cases = [
("skill", "+ \"Remember to run the tests before finishing.\""),
("control_flow", "+ for attempt in range(max_attempts): result = retry(step)"),
(None, "+ review = subcall('reviewer', transcript)"),
("memory", "+ context = compress_context(context, keep_last=8)"),
]
for declared, diff in tag_cases:
tag = normalize(declared, diff, COMPONENT_SIGNALS)
print(f" declared {str(declared):13s} -> tagged {tag:13s} {diff[2:60]!r}")
print(" A proposer cannot label a prompt tweak as a new 'skill' to look novel: without evidence")
print(" the tag falls back to what the diff actually is.")
counts = {"prompt": 3, "subagent": 1}
print(f"\n novelty(nu) counts STRUCTURAL components {K_STR} the incumbent has never accepted.")
print(f" against an incumbent with accepted edits {counts}:")
for comps in (["memory"], ["subagent"], ["prompt", "client_tool", "memory"]):
print(f" {str(comps):38s} nu = {novelty(comps, counts)}")
return "precheck rejected 4/5 diffs without an LLM call"
critic_and_tags()ระบบ critic ทำหน้าที่เป็นเกราะป้องกันข้อมูลรั่วไหล โดยใช้รูปแบบ deterministic ตรวจสอบโค้ด เช่น การแอบจดจำ task id หรือการอ่านคำตอบจาก grader ระบบจะปัดตกทันทีโดยไม่ต้องเรียกใช้ LLM ช่วยประหยัดค่าใช้จ่าย นอกจากนี้ยังมีฟังก์ชัน normalize ที่คอยตรวจสอบว่าการแก้ไขที่ส่งมามีหลักฐานสอดคล้องกับป้ายกำกับ (tag) หรือไม่ เพื่อป้องกันการแอบอ้างความแปลกใหม่ทางโครงสร้างเพื่อหวังโบนัสคะแนน
@section("8. The edit history: what was tried, what paid off, what to prune")
def edit_history():
with tempfile.TemporaryDirectory() as tmp:
h = History(Path(tmp) / "history.jsonl")
log = [
(0, "A", [("prompt", "tell the agent to read the failing test first")], "ACCEPTED", 0.030, 0.02, True, 0.66),
(1, "A", [("prompt", "ask for a plan before editing")], "REJECTED", -0.010, 0.05, False, 0.65),
(1, "B", [("subagent", "add a reviewer sub-agent"), ("memory", "persist lint rules")],
"REJECTED", -0.020, 0.40, False, 0.64),
(2, "A", [("config", "raise the step limit")], "ACCEPTED", 0.005, -0.03, True, 0.665),
(3, "A", [("prompt", "shorter system prompt")], "LOST", -0.001, -0.10, False, 0.664),
(4, "B", [("memory", "cache task_014 solution")], "critic_reject", None, None, False, None),
(5, "A", [("prompt", "stricter output format")], "REJECTED", -0.004, 0.00, False, 0.661),
]
for t, v, edits, outcome, dS, dC, acc, S in log:
h.append_candidate(t, v, [{"id": f"C{i + 1}", "component": c, "hypothesis": hyp}
for i, (c, hyp) in enumerate(edits)],
outcome, dS, dC, acc, S, 12_000 if S else None, diff=None)
t_now = 6
print(f" {len(h.records())} per-edit records from {len(log)} candidates"
f" (the two-edit bundle in round 1 wrote two records with ONE measurement)")
print(f" T_t, tried components : {sorted(h.tried())} (the critic-rejected edit is not 'tried')")
g_t = h.yield_g(t_now, CFG.n_prune)
print(" g_t, best gain in the last n_prune rounds: "
+ ", ".join(f"{c} {'none measured' if g == -math.inf else f'{g:+.3f}'}" for c, g in sorted(g_t.items())))
prune = h.prune_set(t_now, CFG.n_prune)
print(f" B_t, prune set : {[p['component'] for p in prune]}")
for p in prune:
if p["accepted_edits_in_incumbent"]:
print(f" {p['component']}: still in the incumbent but no recent gain ->"
f" {[e['hypothesis'] for e in p['accepted_edits_in_incumbent']]}")
trajectory = [0.630, 0.660, 0.660, 0.665, 0.665, 0.665, 0.665]
sigma = stall_flag(trajectory, t_now, CFG.w, DELTA)
ex = exploration(t_now, sigma, h.tried(), CFG.m_draft)
print(f"\n S over rounds {trajectory}: moved {trajectory[t_now] - trajectory[t_now - CFG.w]:+.3f}"
f" in the last w={CFG.w} rounds -> sigma_t = {sigma}")
print(" what the proposer is told next round:")
print(textwrap.indent(textwrap.fill(ex["text"], 84), " "))
print("\n The proposer sees this history, so a falsified hypothesis ('ask for a plan first', -1pp)")
print(" is not redrawn, and a stalled run is pushed toward components it has never touched.")
return f"prune set {[p['component'] for p in prune]}, stall flag {sigma}"
edit_history()การเก็บประวัติ (History) ช่วยให้ระบบเรียนรู้จากความผิดพลาด สมมติฐานที่เคยถูกปฏิเสธจะไม่ถูกนำกลับมาใช้ซ้ำ และหากคะแนนหยุดชะงัก (stall) ระบบจะกระตุ้นให้ไปสำรวจส่วนประกอบใหม่ๆ ที่ยังไม่เคยแตะต้อง นอกจากนี้ยังมีการคัดส่วนประกอบที่ไม่มีการพัฒนาในระยะหลัง (prune) ออกจากกระบวนการเพื่อเพิ่มประสิทธิภาพ
def write_harness(root, h):
root = Path(root)
root.mkdir(parents=True, exist_ok=True)
(root / "harness.json").write_text(json.dumps(h))
return root
class SimulatedAgentDomain(Domain):
"""A Domain adapter over the simulated agent: the same contract RRSI's coding,
workspace and engineering instances implement."""
name = "simulated"
critic_patterns = CRITIC_PATTERNS
component_signals = COMPONENT_SIGNALS
briefs = {"critic": "A coding agent harness evaluated on parse/search/edit/test tasks."}
def __init__(self, world, seed):
self.world, self.rng = world, random.Random(seed)
def evolve_ids(self):
return [t for t in self.world if t.startswith("task_")]
def heldout_ids(self):
return [t for t in self.world if t.startswith("held_")]
def smoke_ids(self, incumbent_per_task=None):
return self.evolve_ids()[:3]
def run(self, root, runs_dir, job, ids, k, log_prefix=""):
out = Path(runs_dir) / "jobs" / job
if (out / "trials.json").exists():
return # resume-safe, as the contract requires
h = json.loads((Path(root) / "harness.json").read_text())
per = run_trials(h, self.world, ids, k, self.rng)
out.mkdir(parents=True, exist_ok=True)
(out / "trials.json").write_text(json.dumps(
{t: {"rewards": r.rewards, "tokens": r.tokens} for t, r in per.items()}))
def score(self, runs_dir, job, ids, k):
d = json.loads((Path(runs_dir) / "jobs" / job / "trials.json").read_text())
return {t: TaskResult(rewards=d[t]["rewards"], tokens=d[t]["tokens"]) for t in ids}, {}
@section("9. A Domain adapter: plugging an environment into RRSI's own evaluate()")
def domain_adapter():
dom = SimulatedAgentDomain(make_world(0, 40), seed=3)
with tempfile.TemporaryDirectory() as tmp:
root, runs = write_harness(Path(tmp) / "wt_H0", H0), Path(tmp) / "runs"
ev = evaluate(dom, root, runs, "H0", dom.evolve_ids(), k=2)
files = sorted(str(p.relative_to(tmp)) for p in Path(tmp).rglob("*") if p.is_file())
print(f" evaluate(domain, worktree, runs_dir, 'H0', 40 ids, k=2) -> S={ev.S:.3f} C={ev.C:,.0f}")
print(f" files: {files}")
again = evaluate(dom, root, runs, "H0", dom.evolve_ids(), k=2)
print(f" evaluate() again on the same job -> S={again.S:.3f} (read back, not re-run)")
print("\n The harness lives in files under a worktree root, exactly as RRSI's real instances keep")
print(" one git worktree per candidate. RRSI never runs an agent or grades anything itself; the")
print(" adapter's run() and score() do, and everything in steps 2-8 consumes what they return.")
return f"adapter evaluated H0 at S={ev.S:.3f} through rrsi.evaluate.evaluate"
domain_adapter()RRSI แยกส่วนการทำงานออกจากสภาพแวดล้อมจริงผ่าน Domain adapter ทำให้สามารถรองรับได้ทั้งกรณีศึกษาการเขียนโค้ด หรืองานวิศวกรรมอื่นๆ ระบบเน้นความปลอดภัยในการรันต่อ (resume-safety) โดยจะอ่านข้อมูลจากการทดลองเดิมแทนการรันใหม่หากพบไฟล์ผลลัพธ์เดิม
def propose_edit(r, evolve_ids):
"""The scripted proposer. Each draw is one edit whose TRUE effect we know."""
u = r.random()
if u < 0.35:
fam = r.choice(FAMILIES)
comp = r.choice(["prompt", "control_flow", "context_mgmt"])
diff = {"prompt": f"+ \"On {fam} tasks, check the edge cases before finishing.\"",
"control_flow": f"+ for attempt in range(max_attempts): result = retry({fam}_step)",
"context_mgmt": f"+ context = compress_context(context, keep_last=12) # {fam}"}[comp]
return {"kind": "general", "component": comp, "family": fam, "effect": r.gauss(0.20, 0.30),
"dcost": 0.02, "diff": diff}
if u < 0.55:
ids = r.sample(evolve_ids[:40], 3)
return {"kind": "leaky", "component": "memory", "ids": ids, "dcost": 0.03,
"diff": "+ memory = Memory('solutions')\n" + "\n".join(
f"+ memory.remember('{i}', cached_patch)" for i in ids)}
if u < 0.70:
return {"kind": "inert", "component": "config", "dcost": r.uniform(-0.02, 0.05),
"diff": "+ CONFIG['log_level'] = 'debug'"}
if u < 0.85:
return {"kind": "expensive", "component": "subagent", "effect": 0.15, "mult": 1.5,
"diff": "+ review = subcall('reviewer', transcript)"}
return {"kind": "compress", "component": "context_mgmt", "effect": -0.03, "mult": 0.85,
"diff": "+ context = compress_context(context, keep_last=6)"}
def apply_edit(h, e):
h = copy.deepcopy(h)
if e["kind"] == "general":
h["skill"][e["family"]] += e["effect"]
h["cost"] *= 1 + e["dcost"]
elif e["kind"] == "leaky":
h["memo"] = sorted(set(h["memo"]) | set(e["ids"]))
h["cost"] *= 1 + e["dcost"]
elif e["kind"] == "inert":
h["cost"] *= 1 + e["dcost"]
else: # expensive / compress: every family
for f in FAMILIES:
h["skill"][f] += e["effect"]
h["cost"] *= e["mult"]
return h
def search(world, mode, *, seed, k, delta, T=CFG.T, m=CFG.m):
"""One run of harness search. mode: 'greedy' | 'critic only' | 'rrsi'. Same proposer stream in all."""
dom = SimulatedAgentDomain(world, seed=seed * 7 + 1)
proposer = random.Random(seed * 13 + 5)
ids = dom.evolve_ids()
with tempfile.TemporaryDirectory() as tmp:
tmp, h = Path(tmp), copy.deepcopy(H0)
runs, hist = tmp / "runs", History(tmp / "history.jsonl")
inc = evaluate(dom, write_harness(tmp / "H0", h), runs, "H0", ids, k)
S_star, audit, n_evals = inc.S, [], 1
for t in range(T):
b_t = edit_budget(t, T, CFG.b_min, CFG.b_max)
cands, drafts = [], []
for v in "AB"[:m]:
edits = [propose_edit(proposer, ids) for _ in range(proposer.randint(1, b_t))]
hc = h
for e in edits:
hc = apply_edit(hc, e)
diff = "\n".join(e["diff"] for e in edits)
tags = [{"id": f"C{i + 1}", "component": normalize(e["component"], e["diff"], COMPONENT_SIGNALS),
"hypothesis": e["kind"]} for i, e in enumerate(edits)]
c = Candidate(f"r{t}{v}", tags)
if mode != "greedy" and precheck(diff, dom.critic_patterns):
c.gate_failure = "critic_reject" # screened BEFORE any evaluation is spent
else:
c.ev = evaluate(dom, write_harness(tmp / c.variant, hc), runs, c.variant, ids, k)
n_evals += 1
cands.append(c)
drafts.append((hc, edits))
if mode == "rrsi":
winner, decisions = select_round(cands, inc, S_star, delta, CFG,
hist.incumbent_component_counts())
for c, d in zip(cands, decisions): # the records loop.round() writes
outcome = (c.gate_failure if c.ev is None else
"ACCEPTED" if c is winner else "LOST" if d.admissible else "REJECTED")
hist.append_candidate(t, c.variant, c.edits, outcome, d.delta_S, d.delta_C,
c is winner, d.S, d.C, None, d.reason)
else: # unregularized: keep the best score if it rose
live = [c for c in cands if c.ev is not None and c.ev.S > inc.S]
winner = max(live, key=lambda c: c.ev.S) if live else None
for c, (hc, edits) in zip(cands, drafts):
audit += [(e["kind"], c is winner) for e in edits]
if winner is not None:
h, inc = drafts[cands.index(winner)][0], winner.ev
S_star = max(S_star, inc.S)
return {"evolve_measured": inc.S, "evolve_true": true_score(h, world, ids),
"heldout_true": true_score(h, world, dom.heldout_ids()), "tokens": h["cost"],
"memorised": len(h["memo"]), "evals": n_evals, "audit": audit}
MODES = ["greedy", "critic only", "rrsi"]
def compare(n_evolve, k, seeds):
out = {mode: [] for mode in MODES}
deltas = []
for s in seeds:
world = make_world(s, n_evolve)
cal, _ = calibrated_delta(world, k, seed=s)
deltas.append(cal["delta"])
for mode in MODES:
out[mode].append(search(world, mode, seed=s, k=k, delta=cal["delta"]))
return out, st.mean(deltas)
def fmt(xs, d=3):
return f"{st.mean(xs):.{d}f}±{st.pstdev(xs):.{d}f}"
@section("10. Greedy vs RRSI in a world where we know the truth")
def miniature():
seeds = range(8)
out, delta = compare(40, 2, seeds)
globals()["MINIATURE"] = (out, delta) # step 11 reuses this row
h0_held = st.mean(true_score(H0, make_world(s, 40), [f"held_{i:03d}" for i in range(80)]) for s in seeds)
print(f" 40 evolve tasks x k=2, 80 held-out tasks, T={CFG.T}, m={CFG.m}, {len(seeds)} seeds,"
f" mean calibrated delta {delta:.3f}")
print(f" H0 held-out (true) = {h0_held:.3f}\n")
print(f" {'mode':12s} {'evolve meas':>12s} {'evolve TRUE':>12s} {'held-out TRUE':>14s}"
f" {'tokens':>11s} {'memorised':>10s} {'evals':>6s}")
for mode in MODES:
rs = out[mode]
print(f" {mode:12s} {fmt([r['evolve_measured'] for r in rs]):>12s} {fmt([r['evolve_true'] for r in rs]):>12s}"
f" {fmt([r['heldout_true'] for r in rs]):>14s} {fmt([r['tokens'] for r in rs], 2) + 'x':>11s}"
f" {st.mean(r['memorised'] for r in rs):>10.1f} {st.mean(r['evals'] for r in rs):>6.1f}")
print("\n Ground-truth audit: share of proposed edits of each kind that ended up in the incumbent")
kinds = ["general", "expensive", "compress", "inert", "leaky"]
print(f" {'mode':12s}" + "".join(f"{k:>11s}" for k in kinds))
for mode in MODES:
tally = {k: [0, 0] for k in kinds}
for r in out[mode]:
for kind, accepted in r["audit"]:
tally[kind][0] += 1
tally[kind][1] += accepted
print(f" {mode:12s}" + "".join(f"{tally[k][1]:>5d}/{tally[k][0]:<5d}" for k in kinds))
print("\n Where the evolve-set score stops transferring, split in two:")
print(f" {'mode':12s} {'winner curse':>14s} {'memorisation':>14s} (evolve meas - evolve true | evolve true - held-out true)")
for mode in MODES:
curse = st.mean(r["evolve_measured"] - r["evolve_true" ] for r in out[mode])
memo = st.mean(r["evolve_true"] - r["heldout_true"] for r in out[mode])
print(f" {mode:12s} {curse:>+14.3f} {memo:>+14.3f}")
print(" Selecting the best of noisy scores inflates every mode about equally; no rule here removes")
print(" that - only re-measuring on tasks the search never saw does. The critic removes almost all")
print(" of the memorisation, and it does so before evaluation, which is why its runs cost fewer evals.")
g, c, r_ = (st.mean(x["heldout_true"] for x in out[m_]) for m_ in MODES)
tg, tc, tr = (st.mean(x["tokens"] for x in out[m_]) for m_ in MODES)
return f"held-out {g:.3f} / {c:.3f} / {r_.3f}, tokens x{tg:.2f} / x{tc:.2f} / x{tr:.2f} (greedy / critic / rrsi)"
miniature()การเปรียบเทียบระหว่างโหมด Greedy (เลือกตามคะแนนสูงสุด), Critic only (เพิ่มการคัดกรอง) และ RRSI (ใช้ Algorithm 2) แสดงให้เห็นว่า Greedy ทำคะแนนสูงสุดใน evolve set แต่กลับมีต้นทุน token พุ่งสูงเกือบ 3 เท่า และมีการจดจำคำตอบ (leakage) เป็นจำนวนมาก ขณะที่ RRSI สามารถควบคุมต้นทุนให้อยู่ในระดับต่ำกว่า และกำจัดการจดจำคำตอบได้อย่างสิ้นเชิง
@section("11. Turn the evaluator up: RRSI's caution is calibrated, not configured")
def noise_sweep():
rows = [("40x2", MINIATURE[1], MINIATURE[0], 8)] # from step 10
for n, k in [(100, 4), (400, 8)]:
out, delta = compare(n, k, range(5))
rows.append((f"{n}x{k}", delta, out, 5))
print(f" {'evaluator':>10s} {'seeds':>5s} {'delta':>7s} "
+ "".join(f"{m_ + ' held / tokens':>26s}" for m_ in MODES))
for label, delta, out, n_seeds in rows:
cells = "".join(f"{st.mean(r['heldout_true'] for r in out[m_]):>13.3f} /"
f" x{st.mean(r['tokens'] for r in out[m_]):<9.2f}" for m_ in MODES)
print(f" {label:>10s} {n_seeds:>5d} {delta:>7.3f} {cells}")
first, last = rows[0], rows[-1]
rr_first = st.mean(r["heldout_true"] for r in first[2]["rrsi"])
rr_last = st.mean(r["heldout_true"] for r in last[2]["rrsi"])
tok_ratio = (st.mean(r["tokens"] for r in last[2]["greedy"]) /
st.mean(r["tokens"] for r in last[2]["rrsi"]))
print(f"\n As the evaluator sharpens, delta falls from {first[1]:.3f} to {last[1]:.3f} (the paper: 0.004-0.020),"
f" and RRSI's held-out score rises from {rr_first:.3f} to {rr_last:.3f}.")
print(f" At the sharpest setting the unregularized search is spending {tok_ratio:.1f}x RRSI's tokens.")
print("\n Read the table honestly: this world has no diminishing returns, so every sub-agent the")
print(" greedy search stacks keeps buying accuracy. That is the most favourable world possible for")
print(" spending, and the greedy search does score higher. RRSI trades some of that score for a")
print(" bounded token bill, zero memorised answers and fewer wasted evaluations - and the size of")
print(" the trade is set by delta, which it measures from your evaluator rather than taking from you.")
return (f"delta {first[1]:.3f} -> {last[1]:.3f}; RRSI held-out {rr_first:.3f} -> {rr_last:.3f};"
f" greedy uses {tok_ratio:.1f}x the tokens")
noise_sweep()เมื่อเพิ่มความแม่นยำของตัวประเมิน ค่า delta จะลดลง ส่งผลให้ RRSI ทำคะแนนได้ดีขึ้นเรื่อยๆ แม้ Greedy จะยังทำคะแนนดิบได้สูงกว่า แต่ต้องแลกด้วยต้นทุน token ที่มากกว่า RRSI ถึง 6.5 เท่าในสภาวะที่แม่นยำที่สุด RRSI จึงพิสูจน์ให้เห็นถึงความคุ้มค่าในการรักษาสมดุลระหว่างประสิทธิภาพและทรัพยากรที่ใช้
banner("SUMMARY")
for name, res in RESULTS.items():
print(f" {name:<76s} {res}")
print("""
What the miniature does not model
- Edits that help the evolve suite but hurt a different suite. The paper's LLM critic and its
held-out and out-of-distribution splits exist for those; a regex screen cannot catch them.
- A proposer that reads the history. Ours is scripted, so it redraws falsified ideas freely.
Where to go next
- Run a real instance: `python3 rrsi.py --domain coding smoke` after configuring Claude on Vertex AI
(RRSI_VERTEX_PROJECTS) and harbor; see domains/coding/README.md.
- Add a domain: implement rrsi.domain.Domain in domains/<name>/adapter.py, as step 9 did in the notebook.
- Re-adjudicate a stored round under a different delta or cost rule without re-running anything:
`python3 rrsi.py --domain <name> readjudicate --t <t>`.
- Paper: arxiv.org/abs/2609.24972 Code: github.com/google-research/rrsi
""")สรุปได้ว่า RRSI นำเสนอแนวทางที่เข้มงวดในการปรับปรุง AI Agent โดยใช้กฎการตัดสินใจที่อ้างอิงจากข้อมูลจริงและการวัดผลที่แม่นยำ แทนที่จะเชื่อเพียงคะแนนที่เพิ่มขึ้นซึ่งอาจเป็นเพียง noise การตรวจสอบผ่าน critic แบบ deterministic และการคัดกรองการรั่วไหลของข้อมูลช่วยให้การวิวัฒนาการของ agent เป็นไปอย่างยั่งยืนและประหยัดค่าใช้จ่าย แม้ในโลกจำลองที่การจ่าย token มากกว่าจะได้คะแนนดีกว่า แต่ในความเป็นจริงคุณค่าของ RRSI จะเด่นชัดขึ้นเมื่อทรัพยากรมีจำกัดและต้องการความถูกต้องของข้อมูลสูงสุด
ความคิดเห็น (0)
เข้าสู่ระบบเพื่อร่วมแสดงความเห็น
สมัครสมาชิกมาเป็นคนแรกที่แสดงความเห็นกันเลยโบร
