Google Research发布的RRSI框架详细解析了如何通过噪声带(noise bands)、成本规则(cost rules)和泄漏屏蔽(leakage screens)实现安全、可控的Agent自改进循环,适合作为Agent架构设计参考。
在本教程中,我们实现 RRSI(正则化递归自我改进),这是一种让 LLM 智能体围绕冻结模型重写自身测试平台(harness)、提示词、工具、记忆、控制流和子智能体的方法,同时避免测试平台过度拟合其演化的任务。完整的 RRSI 循环使用 Vertex AI 上的 Claude Opus 起草编辑,并在 Docker 基准测试中评分——这不是免费笔记本能运行的。RRSI 中真正承载论文思想的部分——决定保留哪些候选编辑的规则——是纯 Python 代码,这正是我们直接驱动的部分。我们从官方仓库安装该包,走过其估计器、标定噪声带、选择算法的两个分支、退火编辑预算、确定性泄漏筛查以及编辑历史,然后将模拟智能体接入 RRSI 自身的 Domain 接口。由于我们自己构建了模拟环境,我们知道每个编辑的真实效果,这使我们可以根据真实标签审计 RRSI 的决策,并与未正则化的搜索(简单地保留得分最高的方案)进行比较。
import os
import sys
import json
import math
import copy
import random
import tempfile
import textwrap
import traceback
import subprocess
import statistics as st
from pathlib import Path
RESULTS = {}
def banner(title):
print("\n" + "=" * 78)
print(title)
print("=" * 78)
def section(name):
def wrap(fn):
def run(*a, **kw):
banner(name)
try:
out = fn(*a, **kw)
RESULTS[name] = out if isinstance(out, str) else "ok"
return out
except Exception as e:
RESULTS[name] = f"SKIPPED / FAILED -> {type(e).__name__}: {e}"
print(f"\n[!] {name} did not complete: {type(e).__name__}: {e}")
traceback.print_exc(limit=3)
return None
return run
return wrap
banner("1. Install RRSI and map the paper onto the code")
subprocess.run([sys.executable, "-m", "pip", "install", "-q", "git+https://github.com/google-research/rrsi.git@be50316e1db05914068a973f322770ef08ed7ba1"], check=True)
from importlib.metadata import version
from rrsi.config import RRSIConfig
from rrsi.evaluate import TaskResult, EvalResult, aggregate, evaluate
from rrsi.calibrate import calibrate
from rrsi.selection import Candidate, cost_rule, judge, select_round
from rrsi.schedule import edit_budget, budget_table
from rrsi.history import History, stall_flag, exploration
from rrsi.components import K, K_STR, normalize, novelty
from rrsi.critic import precheck, review
from rrsi.domain import Domain
print(f" rrsi {version('rrsi')} | anthropic {version('anthropic')} | Python {sys.version.split()[0]}")
print("\n RRSI evolves an agent's HARNESS (prompts, tools, memory, control flow, sub-agents) around a")
print(" frozen model. The full loop drafts edits with Claude Opus on Vertex AI and scores them in Docker")
print(" benchmarks. The part that decides which edits to KEEP is plain Python, and that is what we drive:")
for symbol, where in [
("S_hat, C_hat Eq. (estimate)", "rrsi.evaluate.aggregate"),
("delta noise band", "rrsi.calibrate.calibrate"),
("Algorithm 2 floor + cost rule", "rrsi.selection.judge / select_round"),
("b_t Eq. (anneal)", "rrsi.schedule.edit_budget"),
("Critic leakage screen", "rrsi.critic.precheck / review"),
("L_t, g_t, B_t history, yield, prune", "rrsi.history.History"),
("sigma_t, U_t stall + exploration", "rrsi.history.stall_flag / exploration"),
("nu structural novelty", "rrsi.components.novelty"),
]:
print(f" {symbol:40s} -> {where}")
CFG = RRSIConfig()
print(f"\n paper defaults: T={CFG.T} rounds, k={CFG.k} trials/task, m={CFG.m} candidates/round,"
f" b in [{CFG.b_min},{CFG.b_max}]")
print(f" beta0={CFG.beta0} beta1={CFG.beta1} w_s={CFG.w_s} w_c={CFG.w_c} w_n={CFG.w_n}"
f" delta_z={CFG.delta_z}")
print("\n Nothing below needs an API key, a GPU or a dataset download.")
我们从 google-research 仓库安装 RRSI,锁定到编写此笔记本时所对应的 commit,因为该包不在 PyPI 上。它唯一的依赖是 Anthropic 客户端,搜索角色使用它来调用 Claude,而我们不会调用它。然后我们打印仓库自身文档中论文符号与实现它们的函数之间的映射:evaluate 中的经验分数和成本估计、calibrate 中的噪声带、selection 中的 Algorithm 2、schedule 中的退火编辑预算、critic 中的泄漏筛查,以及 history 中包含 yield、prune、stall 和 exploration 摘要的编辑历史。RRSIConfig 保存论文的超参数,下面每个函数接收它时与真实循环完全一致。
@section("2. Evaluate(H): a score and a cost, and why a crash counts as zero")
def estimator():
base = {
"task_000": TaskResult(rewards=[1, 1], tokens=[11_800, 12_400]),
"task_001": TaskResult(rewards=[1, 0], tokens=[15_100, 14_600]),
"task_002": TaskResult(rewards=[0, 0], tokens=[21_000, 19_500]),
}
ev = aggregate("H0", 2, base)
print(f" three tasks x k=2 trials -> S_hat = {ev.S:.3f} C_hat = {ev.C:,.0f} tokens/trial"
f" ({ev.n_expected} trials expected, {ev.missing} missing)")
crashy = dict(base)
crashy["task_002"] = TaskResult(rewards=[0.0, 0.0], tokens=[None, None], missing=2)
ev_crash = aggregate("crashy", 2, crashy)
dropped = {t: r for t, r in crashy.items() if not r.missing}
naive = sum(sum(r.rewards) for r in dropped.values()) / sum(len(r.rewards) for r in dropped.values())
print("\n A candidate crashes on the hardest task instead of failing it:")
print(f" an estimator that drops missing trials reports {naive:.3f} <- looks like a gain")
print(f" RRSI's aggregate (missing = 0, full denominator) {ev_crash.S:.3f} <- no reward for crashing")
print(f" ...and C_hat uses only recorded token counts: {ev_crash.C:,.0f}")
rubric = {"memo": TaskResult(rewards=[0.5, 1.0], weights=[10, 10]),
"brief": TaskResult(rewards=[0.0, 0.0], weights=[90, 90])}
ev_w = aggregate("rubric", 2, rubric)
print("\n Weighted rewards (Harvey LAB style: weight = number of rubric criteria):")
print(f" mean of per-task means = {st.mean(r.mean for r in rubric.values()):.3f}"
f" vs RRSI's S_hat = {ev_w.S:.3f} (fraction of all criteria passed)")
return f"crash scored {ev_crash.S:.3f} under RRSI vs {naive:.3f} if dropped"
estimator()
RRSI 测量每个测试平台的两个数值:S,即每个任务每次试验的平均奖励;C,即每次试验的平均策略 token 数。TaskResult 记录一个任务的试验并对其进行聚合。值得在任何智能体评估中借鉴的细节是如何处理缺失试验。当候选方案在最困难的任务上崩溃而不是失败时,丢弃缺失试验的估计器报告 0.750,使崩溃看起来像一种改进。同时,RRSI 将每个缺失试验计为零奖励,同时保持完整分母,报告的数值与之前相同为 0.500,因此候选方案不会因为毁掉它觉得困难的任务而看起来更好。加权奖励适用于采用评分rubric的测试套件(如 Harvey LAB),其中 S 成为通过的所有标准比例,而不是每个任务均值的均值。
在任何规则能够将真正的收益与运气区分开之前,它需要知道同一个 harness 的分数会自己移动多远。我们构建了一个小型模拟智能体,其在每个任务上的成功率是 harness 技能减去任务难度的 logistic 函数,并对未更改的起始 harness 在四十个任务上各进行两次试验,评估六次:尽管没有任何变化,分数散度达到了 0.113。calibrate 将对同一 harness 的重复评估转化为 delta,即两次运行之间差异的标准差的两倍。在 80 次试验时 delta 约为 0.108;在 3,200 次试验时降至约 0.013,处于论文为其各实例报告的范围内(0.004 到 0.020)。就选择目的而言,任何小于 delta 的收益与重新运行同一个 harness 无法区分。
def ev_at(S, C, job, n=200):
"""An EvalResult with exactly score S (in steps of 1/n) and C tokens per trial."""
hits = round(S * n)
return aggregate(job, 1, {f"task_{i:03d}": TaskResult(rewards=[1.0 if i < hits else 0.0], tokens=[C])
for i in range(n)})
INCUMBENT = ev_at(0.630, 10_000, "incumbent")
S_STAR, DELTA = 0.640, 0.020
@section("4. Algorithm 2, branch by branch: the floor, the cost rule and the noise band")
def algorithm_2():
print(f" incumbent S = {INCUMBENT.S:.3f}, C = {INCUMBENT.C:,.0f} best ever S* = {S_STAR} delta = {DELTA}")
print(f" floor = S* - delta = {S_STAR - DELTA:.3f}\n")
cases = [
("A slipped below the floor", 0.600, 10_000, ["prompt"], None),
("B real gain, pays for its tokens", 0.690, 12_000, ["prompt"], None),
("C real gain, far too expensive", 0.660, 25_000, ["subagent"], None),
("D slightly WORSE but cheaper", 0.625, 8_000, ["context_mgmt"], None),
("E in the band but costlier", 0.640, 11_500, ["prompt"], None),
("F in the band, new sub-agent", 0.630, 10_000, ["subagent"], None),
("F' in the band, prompt tweak", 0.630, 10_000, ["prompt"], None),
("G like B, breaks a domain guard", 0.690, 12_000, ["prompt"], ["valid-output rate fell"]),
]
rows = []
for label, S, C, comps, guards in cases:
cand = Candidate(label[:2].strip(), [{"id": "C1", "component": c} for c in comps], ev=ev_at(S, C, label))
dec = judge(cand, INCUMBENT, S_STAR, DELTA, CFG, incumbent_counts={}, guards=guards)
rows.append((label, dec))
print(f" {label:36s} S={S:.3f} C={C:>6,} {'ADMIT ' if dec.admissible else 'reject'}")
print(textwrap.indent(textwrap.fill(dec.reason, 88), " " * 6))
print("\n Three rules, in the order RRSI applies them:")
print(" 1. never fall below the best score ever seen, minus the noise band (A)")
print(" 2. a gain bigger than delta must pay for any extra tokens: dC <= 0.10 + 40*dS (B, C)")
print(" 3. inside the band scores are a tie, so prefer the cheaper harness, and let a")
print(" never-tried structural component break the tie (D, E, F, F')")
print(" D is the surprising one: RRSI admits a harness that scored LOWER than the incumbent.")
return f"{sum(d.admissible for _, d in rows)}/{len(rows)} admissible; D admitted at dS={rows[3][1].delta_S:+.3f}"
algorithm_2()
Algorithm 2 以纯函数实现,因此我们可以向其输入候选者并原样读取其理由。我们将 incumbent 固定在 S 0.630 和 10,000 tokens,最佳历史分数为 0.640,delta 为 0.020,然后让八个候选者通过 judge。低于 floor(即历史最佳分数减去 delta)的候选者直接被拒绝。大于 delta 的收益必须根据规则为其额外 tokens 买单:相对成本变化保持在 0.10 + 40 倍收益以下;+6 分增益对应 +20% tokens 被接受,而 +3 分增益对应 +150% tokens 则不被接受。在噪声带内,分数被视为平局,由 100 倍增益减去 15 倍成本变化的整形分数决定,外加对从未被接受的 structural component 的小额奖励——这就是 RRSI 接受候选者 D 的方式,D 的分数低于 incumbent 但成本低 20%;同样得分的情况下,新的 sub-agent 击败了 prompt 调整而获胜。领域护栏会无论分数高低一律否决。
@section("5. One round of selection: the highest score does not always win")
def one_round():
c_hi = Candidate("C", [{"id": "C1", "component": "subagent"}], ev=ev_at(0.660, 25_000, "C"))
d_lo = Candidate("D", [{"id": "C1", "component": "context_mgmt"}], ev=ev_at(0.625, 8_000, "D"))
leak = Candidate("X", [{"id": "C1", "component": "memory"}], gate_failure="critic_reject")
winner, decisions = select_round([c_hi, d_lo, leak], INCUMBENT, S_STAR, DELTA, CFG, incumbent_counts={})
for d in decisions:
print(f" {d.variant}: {'admissible' if d.admissible else 'rejected '} {d.reason[:92]}")
new_star = max(S_STAR, winner.ev.S)
print(f"\n winner: {winner.variant} (S {winner.ev.S:.3f}, C {winner.ev.C:,.0f})")
print(f" the new incumbent scores {winner.ev.S - INCUMBENT.S:+.3f} vs the old one and costs"
f" {(winner.ev.C - INCUMBENT.C) / INCUMBENT.C:+.0%} tokens; S* stays {new_star:.3f}")
print("\n C scored highest and still lost: its +3pp does not pay for +150% tokens. D moved the")
print(" incumbent DOWN inside the noise band because it is 20% cheaper. Because S* only ever")
print(" rises, the floor never follows the incumbent down, so a chain of 'cheaper but slightly")
print(" worse' swaps cannot walk the score away. X never reached evaluation at all.")
return f"winner {winner.variant} at dS={winner.ev.S - INCUMBENT.S:+.3f}, dC={(winner.ev.C - INCUMBENT.C) / INCUMBENT.C:+.0%}"
one_round()
一轮选择:最高分并不总是胜出。C 得分最高却仍然落选:其 +3 个百分点不足以支付 +150% tokens。D 因便宜 20% 将 incumbent 移到了噪声带内更低的位置。由于 S* 只升不降,floor 永远不会随 incumbent 下降,因此一连串"更便宜但略差"的交换不会导致分数走偏。X 根本没能进入评估阶段。
select_round 对每轮中的每个候选应用评判器,保留得分最高且可接受的候选。我们向它提供昂贵的候选、更便宜但略差的候选,以及批评者已拒绝的候选。最高分者落选,因为其三分收益不足以抵消 150% 的额外 token 消耗;被批评者拒绝的候选永远不会进入评估阶段;而胜者将 incumbent 压低半分,同时将 token 成本削减五分之一。保障安全的关键在于 S*,它只会上升:底线锚定在历史最佳分数而非 incumbent,因此一连串更便宜但略差的交换无法在多轮中将分数逐渐拉低。
@section("6. The proposal side: an edit budget that anneals from 4 toward 1")
def edit_budget_schedule():
table = budget_table(CFG.T, CFG.b_min, CFG.b_max)
print(" b_t = ceil(b_min + (b_max - b_min) * (1 + cos(pi t / T)) / 2)")
print(f" T={CFG.T}, b in [{CFG.b_min}, {CFG.b_max}]:")
print(" t : " + " ".join(f"{t:>2d}" for t in range(CFG.T)))
print(" b_t : " + " ".join(f"{b:>2d}" for b in table))
print(f" round {CFG.T} (after the run) -> {edit_budget(CFG.T, CFG.T, CFG.b_min, CFG.b_max)}")
print("\n Early candidates may bundle up to 4 coordinated edits. Note the ceil(): the cosine term is")
print(f" only exactly zero at t = T, so inside a {CFG.T}-round run the budget bottoms out at"
f" {min(table)}, not {CFG.b_min}.")
print(" Every edit in a bundle inherits the bundle's single measurement, so the shrinking budget is")
print(" what makes late history attributable to fewer components. It caps how MANY edits ride")
print(" together, never WHICH mechanisms the harness may eventually contain.")
return "budget " + "".join(str(b) for b in table)
edit_budget_schedule()
提案端规范了编辑的起草方式,而非决定保留哪些编辑。edit_budget 实现了论文中的退火 L0 预算:从 b_max 到 b_min 的余弦调度,默认情况下,前八轮每个候选项最多允许四个协调编辑,接下来五轮允许三个,最后七轮允许两个。公式中的向上取整意味着预算仅在 t = T 时(运行结束后一步)才达到最小值 1,阅读公式时容易忽略这一点。因为一个 bundle 中的每个编辑都继承该 bundle 的单一测量结果,所以不断缩减的预算使得后期历史可归因于更少的组件。预算限制了同时进行的编辑数量,从不限制 harness 最终可能包含哪些机制。
CRITIC_PATTERNS = [
(r"\btask_\d{3}\b", "hard-codes an evolve-set task id"),
(r"expected_output|grader|rubric\[", "reads the grader or the expected answer"),
]
COMPONENT_SIGNALS = [
("control_flow", [r"\bretry\(", r"max_attempts"]),
("context_mgmt", [r"compress_context", r"keep_last"]),
("config", [r"CONFIG\["]),
]
class ScreenOnly(Domain):
name = "screen"
critic_patterns = CRITIC_PATTERNS
briefs = {"critic": "A coding agent harness."}
@section("7. The critic's deterministic layer, and how edits are tagged")
def critic_and_tags():
dom = ScreenOnly()
diffs = {
"memorise answers": "+ memory = Memory('answers')\n+ memory.remember('task_007', cached_patch)",
"peek at the grader": "+ if os.path.exists('/grader/expected_output.txt'): return read_it()",
"leaked credential": "+ api_key = 'sk-live-0123456789abcdefghijkl'",
"empty diff": " ",
"general retry rule": "+ for attempt in range(max_attempts): result = retry(step)",
}
for label, diff in diffs.items():
try:
verdict = review(dom, diff, summary=label, targets_mode="evolve")
print(f" {label:20s} -> {verdict['verdict']:6s} {verdict['reasons']}")
except ZeroDivisionError as e:
print(f" {label:20s} -> passed the deterministic layer; the LLM layer raised"
f" ZeroDivisionError: {e}")
print("\n Gotcha: with RRSI_VERTEX_PROJECTS unset, rrsi.llm.generate computes `x % len(projects)`")
print(" outside its try block, so the helpful 'set RRSI_VERTEX_PROJECTS' error is never reached.")
print(" A clean diff is supposed to go to Claude for an intent review; here there is no Claude.")
print("\n Every edit is tagged with the component it touches, and a tag needs evidence in the diff:")
tag_cases = [
("skill", "+ \"Remember to run the tests before finishing.\""),
("control_flow", "+ for attempt in range(max_attempts): result = retry(step)"),
(None, "+ review = subcall('reviewer', transcript)"),
("memory", "+ context = compress_context(context, keep_last=8)"),
]
for declared, diff in tag_cases:
tag = normalize(declared, diff, COMPONENT_SIGNALS)
print(f" declared {str(declared):13s} -> tagged {tag:13s} {diff[2:60]!r}")
print(" A proposer cannot label a prompt tweak as a new 'skill' to look novel: without evidence")
print(" the tag falls back to what the diff actually is.")
counts = {"prompt": 3, "subagent": 1}
print(f"\n novelty(nu) counts STRUCTURAL components {K_STR} the incumbent has never accepted.")
print(f" against an incumbent with accepted edits {counts}:")
for comps in (["memory"], ["subagent"], ["prompt", "client_tool", "memory"]):
print(f" {str(comps):38s} nu = {novelty(comps, counts)}")
return "precheck rejected 4/5 diffs without an LLM call"
critic_and_tags()
批评者在任何评估消耗之前以两层方式筛选每个候选 diff。第一层是对通用凭证模式及领域的拒绝列表进行确定性预检;通过检测 evolve 集任务 ID 和评分器路径的模式,它拒绝了记忆 task_007 答案的 diff、读取预期输出的 diff、泄露 API key 的 diff,以及空 diff,全程不调用模型。干净的 diff 落入第二层,由 Claude 进行意图审查,而 notebook 在此暴露了一个真实的坑点:当 RRSI_VERTEX_PROJECTS 未设置时,rrsi.llm.generate 在其 try 块外计算项目数的模索引并抛出 ZeroDivisionError,因此代码中有用的配置错误提示永远无法到达。我们还看了编辑如何被打标签:normalize 仅在 diff 携带证据时才保留声明的组件,因此提议者不能将提示调整伪装成新技能来赢得新颖性奖励,错误标记的上下文管理变更会被标记为其实际内容。
@section("8. The edit history: what was tried, what paid off, what to prune")
def edit_history():
with tempfile.TemporaryDirectory() as tmp:
h = History(Path(tmp) / "history.jsonl")
log = [
(0, "A", [("prompt", "tell the agent to read the failing test first")], "ACCEPTED", 0.030, 0.02, True, 0.66),
(1, "A", [("prompt", "ask for a plan before editing")], "REJECTED", -0.010, 0.05, False, 0.65),
(1, "B", [("subagent", "add a reviewer sub-agent"), ("memory", "persist lint rules")],
"REJECTED", -0.020, 0.40, False, 0.64),
(2, "A", [("config", "raise the step limit")], "ACCEPTED", 0.005, -0.03, True, 0.665),
(3, "A", [("prompt", "shorter system prompt")], "LOST", -0.001, -0.10, False, 0.664),
(4, "B", [("memory", "cache task_014 solution")], "critic_reject", None, None, False, None),
(5, "A", [("prompt", "stricter output format")], "REJECTED", -0.004, 0.00, False, 0.661),
]
for t, v, edits, outcome, dS, dC, acc, S in log:
h.append_candidate(t, v, [{"id": f"C{i + 1}", "c