回应 Fasthei 的 Request changes 两个阻塞项:
1) 安全 / fail-closed 沙箱隔离(原仅靠 ENABLE_QUALITY_EVAL + 运维约定):
- 新增第二道显式确认 HEICODE_SANDBOX_ISOLATED(断言运行在隔离 Pod 内)。
- sandbox.run_tests() 与 quality.evaluate_run_quality() 执行任何代码前调用
assert_isolated(),未确认即抛 SandboxIsolationError——不写文件、不起子进程。
- 启动期 assert_quality_eval_safe():ENABLE_QUALITY_EVAL 开但隔离未确认 → 拒绝启动
(平台级硬失败,非运维口头约定)。
- 文档(security-boundary §8.1/§9、CLAUDE.md)与测试同步:test-sandbox/test-quality
先断言未确认时硬失败,再显式确认后继续。
2) #10 DecisionTrace 可回放(原仅存被选中任务的标量):
- Decision 现记录完整重放上下文:整个候选集(每候选 tau/eta/weight/p_norm/dependents)、
alpha/beta/epsilon、seed、free_slots、total_weight、explore_draw、select_pick、
select_index、explored 分支。
- 新增 DecisionEngine.replay_decision(trace):仅凭一条 trace(无 RNG/活体状态)复现被选任务;
test-decision-engine 断言「重放==实选」跨 50 次决策(探索+利用)成立。
- decision-engine.md §3.3 更新为可回放 DecisionTrace。
附:新增 docs/TESTING.md(reviewer 速查:依赖安装 + 每套测试命令,复审者此前因缺 fakeredis
未能跑到断言)。本地 11 项 gate 全绿。
影响范围:Swarm(orchestrator + 测试 + 文档)。不改 Manager↔Swarm 契约;新增开关
HEICODE_SANDBOX_ISOLATED(默认未设=拒绝执行)。仍非验收:gain/Benchmark_Agent 仍 NaN。
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
213 lines
10 KiB
Python
213 lines
10 KiB
Python
"""Tests for the ACO decision engine (benchmark Group A, Option A: score-at-pull).
|
|
|
|
Hermetic (REDIS_FAKE) and deterministic (seeded RNG over a probabilistic mechanism).
|
|
Run from agent_swarm_v6 (install deps first — needs fakeredis):
|
|
pip install -r orchestrator/requirements.txt -r agent/requirements.txt
|
|
python scripts/test-decision-engine.py
|
|
"""
|
|
import asyncio
|
|
import math
|
|
import os
|
|
import random
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
os.environ["REDIS_FAKE"] = "1"
|
|
os.environ["ENABLE_ACO_DISPATCH"] = "1"
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
|
|
|
from orchestrator.redis_client import redis_client
|
|
from orchestrator import swarm_runtime as sr_mod
|
|
from orchestrator.swarm_runtime import swarm_runtime
|
|
from orchestrator.task_queue import task_queue
|
|
from orchestrator.decision_engine import (
|
|
DecisionEngine, TAU_INITIAL, TAU_MAX, TAU_MIN, aco_dispatch_enabled,
|
|
)
|
|
from benchmark.collectors.run_collector import SwarmRunMetricsCollector
|
|
|
|
failures = []
|
|
|
|
|
|
def check(name, cond):
|
|
print(("PASS" if cond else "FAIL"), "-", name)
|
|
if not cond:
|
|
failures.append(name)
|
|
|
|
|
|
async def _noop(self, *a, **k):
|
|
return None
|
|
|
|
|
|
async def main():
|
|
await redis_client.connect()
|
|
sr_mod.SwarmRuntime._post_callback = _noop
|
|
engine = DecisionEngine(rng=random.Random(42))
|
|
|
|
check("flag readable", aco_dispatch_enabled() is True)
|
|
|
|
# --- pheromone trail (τ) ---
|
|
check("cold start tau = 0.5", await engine.get_tau("testing", "agent-X") == TAU_INITIAL)
|
|
|
|
up = await engine.deposit(agent_role="testing", agent_id="agent-X", success=True)
|
|
check("success deposit raises tau", up > TAU_INITIAL)
|
|
down_start = await engine.get_tau("testing", "agent-Y")
|
|
down = await engine.deposit(agent_role="testing", agent_id="agent-Y", success=False)
|
|
check("failure deposit lowers tau", down < down_start)
|
|
|
|
# repeated success saturates at the cap; repeated failure floors (clamps hold)
|
|
for _ in range(20):
|
|
hi = await engine.deposit(agent_role="testing", agent_id="agent-X", success=True)
|
|
lo = await engine.deposit(agent_role="testing", agent_id="agent-Y", success=False)
|
|
check("tau capped at TAU_MAX", hi <= TAU_MAX)
|
|
check("tau floored at TAU_MIN", lo >= TAU_MIN)
|
|
|
|
# trails are per-(role, agent): agent-X's testing trail does not bleed into other roles
|
|
check("trail keyed by role", await engine.get_tau("documentation", "agent-X") == TAU_INITIAL)
|
|
|
|
# cost eats into the deposit: same outcome, higher cost_ratio → lower tau
|
|
a = await engine.deposit(agent_role="r2", agent_id="cheap", success=True, cost_ratio=0.0)
|
|
b = await engine.deposit(agent_role="r2", agent_id="pricey", success=True, cost_ratio=1.0)
|
|
check("higher cost ratio -> lower deposit", b < a)
|
|
|
|
# --- heuristic (η) ---
|
|
t_old = await task_queue.create_task(task_id="d-implementation", description="x",
|
|
agent_role="implementation",
|
|
required_capabilities=["python"], enqueue=False)
|
|
t_old.created_at = time.time() - 600 # old → urgency saturated
|
|
t_new = await task_queue.create_task(task_id="d2-implementation", description="y",
|
|
agent_role="implementation",
|
|
required_capabilities=["python"], enqueue=False)
|
|
eta_old = DecisionEngine.compute_eta(t_old, ["python"], free_slots=2, dependents_count=3)
|
|
eta_new = DecisionEngine.compute_eta(t_new, ["python"], free_slots=2, dependents_count=0)
|
|
check("older/critical task scores higher eta", eta_old > eta_new)
|
|
# focused specialist (exact caps) beats generalist with many unrelated caps (Jaccard match)
|
|
eta_spec = DecisionEngine.compute_eta(t_new, ["python"], free_slots=1, dependents_count=0)
|
|
eta_gen = DecisionEngine.compute_eta(t_new, ["python", "a", "b", "c", "d"], free_slots=1, dependents_count=0)
|
|
check("specialist match > generalist match", eta_spec > eta_gen)
|
|
check("eta floored positive", eta_new > 0)
|
|
|
|
# --- probabilistic selection ---
|
|
# agent Z earned a strong testing trail; tasks compete for Z's pull.
|
|
for _ in range(10):
|
|
await engine.deposit(agent_role="strong", agent_id="agent-Z", success=True)
|
|
for _ in range(10):
|
|
await engine.deposit(agent_role="weak", agent_id="agent-Z", success=False)
|
|
strong_t = await task_queue.create_task(task_id="s-strong", description="s", agent_role="strong",
|
|
required_capabilities=["python"], enqueue=False)
|
|
weak_t = await task_queue.create_task(task_id="w-weak", description="w", agent_role="weak",
|
|
required_capabilities=["python"], enqueue=False)
|
|
picks = {"s-strong": 0, "w-weak": 0}
|
|
sel_engine = DecisionEngine(rng=random.Random(7))
|
|
for _ in range(200):
|
|
d = await sel_engine.select("agent-Z", ["python"], [strong_t, weak_t],
|
|
free_slots=1, dependents_counts={})
|
|
picks[d.task_id] += 1
|
|
check("high-tau role wins most pulls (seeded)", picks["s-strong"] > picks["w-weak"])
|
|
check("epsilon keeps exploring the weak trail", picks["w-weak"] > 0)
|
|
|
|
# decision payload sanity: p_norm normalized, p_score = standard formula
|
|
d = await sel_engine.select("agent-Z", ["python"], [strong_t, weak_t],
|
|
free_slots=1, dependents_counts={"s-strong": 2})
|
|
check("p_norm in (0,1]", 0 < d.p_norm <= 1.0)
|
|
check("p_score = tau^a*eta^b*100 > 0", d.p_score > 0)
|
|
|
|
# --- DecisionTrace is REPLAYABLE (#10 DoD): full candidate set + draws + hyperparams ---
|
|
trace = d.telemetry()
|
|
check("trace records full candidate set with tau/eta/weight/p_norm",
|
|
len(trace["candidates"]) == 2
|
|
and all({"task_id", "agent_role", "tau", "eta", "weight", "p_norm", "dependents"} <= set(c)
|
|
for c in trace["candidates"]))
|
|
check("trace records hyperparams + seed + draws + branch",
|
|
all(k in trace for k in ("alpha", "beta", "epsilon", "seed", "free_slots",
|
|
"total_weight", "explore_draw", "select_pick", "select_index", "explored")))
|
|
check("trace records per-candidate context (dependents/free_slots)",
|
|
trace["free_slots"] == 1 and any(c["dependents"] == 2 for c in trace["candidates"]))
|
|
# replay from the trace ALONE (no RNG, no live state) reproduces the chosen task
|
|
replayed = DecisionEngine.replay_decision(trace)
|
|
check("replay_decision(trace) reproduces the chosen task", replayed == d.task_id)
|
|
# replay holds across many decisions (both explore and exploit branches)
|
|
replay_ok = True
|
|
for _ in range(50):
|
|
dd = await sel_engine.select("agent-Z", ["python"], [strong_t, weak_t],
|
|
free_slots=1, dependents_counts={})
|
|
if DecisionEngine.replay_decision(dd.telemetry()) != dd.task_id:
|
|
replay_ok = False
|
|
break
|
|
check("replay matches the live choice over 50 decisions (explore+exploit)", replay_ok)
|
|
|
|
# same seed → identical pick sequence (CI determinism)
|
|
seq1 = [
|
|
(await DecisionEngine(rng=random.Random(99)).select(
|
|
"agent-Z", ["python"], [strong_t, weak_t], free_slots=1, dependents_counts={})).task_id
|
|
for _ in range(5)
|
|
]
|
|
e2 = DecisionEngine(rng=random.Random(99))
|
|
seq2 = [
|
|
(await e2.select("agent-Z", ["python"], [strong_t, weak_t],
|
|
free_slots=1, dependents_counts={})).task_id
|
|
for _ in range(1)
|
|
]
|
|
check("seeded selection deterministic", seq1[0] == seq2[0])
|
|
|
|
# --- telemetry -> collector (tau/eta/p_decision become real) ---
|
|
body = {
|
|
"mode": "swarm",
|
|
"requirement": {"objective": "decision test"},
|
|
"callback": {"url": "http://127.0.0.1:9/cb", "subscribed_events": []},
|
|
"metadata": {"manager_deployment_id": "m-aco"},
|
|
}
|
|
run, _ = await swarm_runtime.get_or_create_run(body=body, idempotency_key=None, correlation_id="ca")
|
|
await swarm_runtime.record_decision(run, d.telemetry())
|
|
t = await task_queue.create_task(task_id="t-implementation", description="t",
|
|
agent_role="implementation", enqueue=False)
|
|
from orchestrator.task_queue import TaskStatus
|
|
t.status = TaskStatus.COMPLETED
|
|
t.assigned_agent_id = "agent-Z"
|
|
await task_queue._save_task(t)
|
|
await swarm_runtime.attach_task(run, t.task_id)
|
|
|
|
collector = SwarmRunMetricsCollector(run.swarm_id)
|
|
m = await collector.collect()
|
|
check("tau real in collector", not math.isnan(m.tau) and collector.coverage["tau"])
|
|
check("eta real in collector", not math.isnan(m.eta) and collector.coverage["eta"])
|
|
check("p_decision real in collector", not math.isnan(m.p_decision) and collector.coverage["p_decision"])
|
|
# Group A must NOT fake the still-blocked aggregates
|
|
check("gain/benchmark still NaN", math.isnan(m.s_gain) and math.isnan(m.benchmark))
|
|
|
|
# run with no decisions -> NaN + coverage False (flag-off / non-ACO runs are honest)
|
|
run2, _ = await swarm_runtime.get_or_create_run(
|
|
body={**body, "metadata": {"manager_deployment_id": "m-aco2"}},
|
|
idempotency_key=None, correlation_id="cb")
|
|
t2 = await task_queue.create_task(task_id="t2-implementation", description="t2",
|
|
agent_role="implementation", enqueue=False)
|
|
t2.status = TaskStatus.COMPLETED
|
|
t2.assigned_agent_id = "A"
|
|
await task_queue._save_task(t2)
|
|
await swarm_runtime.attach_task(run2, t2.task_id)
|
|
c2 = SwarmRunMetricsCollector(run2.swarm_id)
|
|
m2 = await c2.collect()
|
|
check("no decisions -> tau/eta/p_decision NaN",
|
|
math.isnan(m2.tau) and math.isnan(m2.eta) and math.isnan(m2.p_decision)
|
|
and c2.coverage["tau"] is False)
|
|
|
|
# --- queue helper: enumeration does not dequeue ---
|
|
q_task = await task_queue.create_task(task_id="q-implementation", description="q",
|
|
agent_role="implementation",
|
|
required_capabilities=["python"])
|
|
before = await task_queue.get_pending_count()
|
|
cands = await task_queue.get_ready_pending_tasks(["python"])
|
|
after = await task_queue.get_pending_count()
|
|
check("get_ready_pending_tasks enumerates without dequeue",
|
|
any(c.task_id == "q-implementation" for c in cands) and before == after)
|
|
|
|
print()
|
|
if failures:
|
|
print(f"{len(failures)} decision-engine check(s) FAILED: {failures}")
|
|
sys.exit(1)
|
|
print("all ACO decision-engine checks passed")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
asyncio.run(main())
|