aps-agent/tests/golden/test_fallback_p4_quality.py

1563 lines
66 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# ============================================================
# P4 质量 golden 设施(Agent-B,全部确定性;零真实 LLM/网络)。
# 手法与既有 golden 一致:FakeStore + fake runner 注入
# propose_reply(runner=...)/execute_confirmed + monkeypatch
# fallback_lane.build_pi_runner;S4/S6/S7 走既有白名单/沙盒/审批语义。
#
# Q-01..Q-40 场景映射(本条文件内可复算场景以 `# 场景: Q-xx` 标注;
# 对既有专项等价覆盖的场景见文件底部 _Q_REFERENCE_MAP 与 notes)。
# ============================================================
from __future__ import annotations
import contextlib
import copy
import hashlib
import json
import subprocess
import uuid
from pathlib import Path
from typing import ClassVar
import pytest
from server.agent_core import fallback_highrisk, fallback_lane, harness
from server.agent_core.feature_flags import load_feature_flags
from server.agent_core.providers import reset_provider
from server.aps_domain.workflow import execute_confirmed
from server.auth.context import IdentityContext, bind_identity, reset_identity
from server.contracts import IntentResult
from server.integrations.pi_bridge import PiBridge
from server.state.checkpoints import CheckpointStore
from server.state.seed import seed_world
DEMO_PRODUCT = "CTRL-A"
# Q-01..Q-40 文件内实现清单;被引用为既有等价的场景不在清单里再复制。
_Q_IMPLEMENTED = [
"Q-01", "Q-02", "Q-03", "Q-04", "Q-05", "Q-06", "Q-07", "Q-08",
"Q-09", "Q-10", "Q-11", "Q-12", "Q-13", "Q-14", "Q-15", "Q-16",
"Q-17", "Q-18", "Q-19", "Q-20", "Q-21", "Q-22", "Q-23", "Q-24",
"Q-25", "Q-26", "Q-27", "Q-28", "Q-29", "Q-30", "Q-31", "Q-32",
"Q-33", "Q-34", "Q-35", "Q-36", "Q-37", "Q-38", "Q-39", "Q-40",
]
@pytest.fixture(autouse=True)
def _isolate(tmp_path, monkeypatch):
monkeypatch.setenv("APS_FALLBACK_DIR", str(tmp_path / "fb"))
monkeypatch.setenv("APS_FEATURES_PATH", str(tmp_path / "features.json"))
monkeypatch.setenv("APS_FALLBACK_HIGHRISK_PATH", str(tmp_path / "fallback-highrisk.json"))
monkeypatch.delenv("LLM_API_KEY", raising=False)
monkeypatch.delenv("LLM_BASE_URL", raising=False)
monkeypatch.delenv("LLM_MODEL", raising=False)
monkeypatch.delenv("LLM_PROVIDER", raising=False)
reset_provider()
from server.state import checkpoints as _cp_mod
from server.state import store as _store_mod
kept_scoped = dict(_store_mod._stores)
kept_cps = dict(_cp_mod._checkpoints)
yield
_store_mod._stores.clear()
_store_mod._stores.update(kept_scoped)
_cp_mod._checkpoints.clear()
_cp_mod._checkpoints.update(kept_cps)
reset_provider()
class FakeStore:
"""P2/P3 同款 FakeStore:.checkpoints 注入点 + restore 深拷贝整体替换。"""
_KIND_TABLE: ClassVar[dict[str, str]] = {
"salesOrder": "salesOrders", "material": "materials",
"audit": "auditEvents", "importBatch": "importBatches",
}
def __init__(self, tmp_path: Path):
self.data = seed_world()
self._counters: dict[str, int] = {}
self.tenant_uuid = "platform"
self.world_key = "default"
self.path = str(tmp_path / "world.json")
self.checkpoints = CheckpointStore(str(tmp_path / "checkpoints.json"))
def next_id(self, kind: str) -> int:
if kind not in self._counters:
table = self._KIND_TABLE.get(kind)
self._counters[kind] = max(
(x.get("id", 0) for x in self.data.get(table, [])
if isinstance(x.get("id"), int)), default=0) if table else 0
self._counters[kind] += 1
return self._counters[kind]
def save(self) -> None:
pass
def restore(self, world: dict) -> None:
self.data = copy.deepcopy(world)
self._counters.clear()
def _cfg(tmp_path: Path, **kw) -> fallback_lane.FallbackConfig:
return fallback_lane.FallbackConfig(pi_home=str(tmp_path / "pi-home"), **kw)
def _write_features(tmp_path: Path, features: dict | None = None) -> None:
doc = {"version": 1, "features": {"fallback": True} if features is None else features}
(tmp_path / "features.json").write_text(json.dumps(doc, ensure_ascii=False), encoding="utf-8")
def _write_whitelist(tmp_path: Path, doc) -> None:
path = tmp_path / "fallback-highrisk.json"
if isinstance(doc, str):
path.write_text(doc, encoding="utf-8")
else:
path.write_text(json.dumps(doc, ensure_ascii=False), encoding="utf-8")
def _wl_doc(*, s4=True, s6=True, s7=True, s6_intents=None, s6_max=None,
s7_roles=("ops", "admin")) -> dict:
scenarios = {
"S4": {"enabled": s4,
"intents": ["flex.simulate_due", "flex.compare",
"scenario.compare", "scenario.sensitivity"],
"roles": ["planner", "admin"]},
"S6": {"enabled": s6, "intents": list(s6_intents or ["mes.report"]),
"roles": ["planner", "admin"]},
"S7": {"enabled": s7,
"intents": ["agent.fallback.ops.config.apply",
"agent.fallback.policy.update"],
"roles": list(s7_roles)},
}
if s6_max is not None:
scenarios["S6"]["maxItemsPerRun"] = s6_max
return {"whitelistVersion": 1, "updatedAt": "2026-09-04T10:00:00",
"updatedBy": "p4-quality-test", "scenarios": scenarios}
def _intent(query: str, name: str = "unknown", **params) -> IntentResult:
"""Pi 的结构化信封:query 是原话,params 是 Pi 已选定的工具槽位。"""
return IntentResult(intent=name, params={"query": query, **params},
confidence=0.1, source="LLM")
def _fp(world: dict) -> str:
return harness.world_fingerprint(world)
@contextlib.contextmanager
def _as(user_id: int, *roles: str):
token = bind_identity(IdentityContext(
user_id, f"user-{user_id}", f"User {user_id}", "platform", roles=tuple(roles)))
try:
yield
finally:
reset_identity(token)
def _bind_personal_world(store: FakeStore, user_id: int) -> None:
from server.state import store as store_mod
store.world_key = f"personal-{user_id}"
store_mod._stores[("platform", store.world_key)] = store
def _run_dirs(tmp_path: Path) -> list[Path]:
root = tmp_path / "fb"
return sorted([p for p in root.iterdir() if p.is_dir() and p.name != "pi-home"]) \
if root.exists() else []
def _run_dir_of(tmp_path: Path) -> Path:
dirs = _run_dirs(tmp_path)
assert len(dirs) == 1
return dirs[0]
def _fb_audits(store: FakeStore) -> list[dict]:
return [e for e in store.data.get("auditEvents", [])
if e.get("action") == "agent.fallback.propose"]
def _propose_run_dir(store: FakeStore) -> Path:
audits = _fb_audits(store)
assert len(audits) == 1
run_dir = str(audits[0].get("rationale", {}).get("runDir") or "")
assert run_dir and Path(run_dir).is_dir()
return Path(run_dir)
def _exec_audits(store: FakeStore) -> list[dict]:
return [e for e in store.data.get("auditEvents", [])
if e.get("action") == "agent.fallback.execute"]
def _stage_audits(store: FakeStore) -> list[dict]:
return [e for e in store.data.get("auditEvents", [])
if e.get("action") == "agent.fallback.execute.stage"]
def _world_write_audits(store: FakeStore) -> list[dict]:
return [e for e in store.data.get("auditEvents", [])
if e.get("category") == "WORLD_WRITE"]
def _pending_record(confirm_id: str) -> dict:
return harness._approval_store.pending[confirm_id]
def _mutate_pending(confirm_id: str, mutate) -> None:
mutate(_pending_record(confirm_id))
harness._approval_store.save()
def _read_issued_call_ids(work_dir: Path) -> list[str]:
calls = work_dir.parent / "calls.jsonl"
if not calls.exists():
return []
return [json.loads(line)["call_id"]
for line in calls.read_text(encoding="utf-8").splitlines()
if line.strip() and json.loads(line).get("status") == "issued"]
def make_success_runner(report_template: str):
"""Q-01/Q-03 共用:1 次 read + stop;callId 取自 calls.jsonl 惰性凭证。"""
def runner(task: str, work_dir: Path):
yield {"type": "tool_execution_start", "toolName": "read",
"toolCallId": "t1"}
yield {"type": "tool_execution_end", "toolName": "read",
"toolCallId": "t1", "result": "snapshot ok"}
ids = _read_issued_call_ids(work_dir)
report = report_template.format(call_id=ids[0] if ids else "call-missing")
yield {"type": "message_update", "delta": {"text": report}}
yield {"type": "message_end", "message": {"role": "assistant",
"stopReason": "stop",
"content": [{"type": "text", "text": report}]}}
yield {"type": "agent_end", "messages": [
{"role": "assistant", "stopReason": "stop",
"content": [{"type": "text", "text": report}]}]}
return runner
def make_plan_runner(plan_builder, report: str = "status: success\n\n已生成执行计划。"):
def runner(task: str, work_dir: Path):
run_dir = work_dir.parent
plan = plan_builder(run_dir)
if plan is not None:
(run_dir / "outbox").mkdir(parents=True, exist_ok=True)
(run_dir / "outbox" / "plan.json").write_text(
json.dumps(plan, ensure_ascii=False), encoding="utf-8")
yield {"type": "message_end", "message": {"role": "assistant",
"stopReason": "stop",
"content": [{"type": "text", "text": report}]}}
yield {"type": "agent_end", "messages": []}
return runner
def make_exec_runner(requests: list[dict]):
"""执行段 fake runner:先写动作请求邮箱,再心跳等待至编排器收束。"""
def runner(task: str, work_dir: Path):
actions = work_dir.parent / "outbox" / "actions"
actions.mkdir(parents=True, exist_ok=True)
for req in requests:
name = f"{req['seq']}-{req['intent']}.json"
(actions / name).write_text(json.dumps(req, ensure_ascii=False),
encoding="utf-8")
for _ in range(50):
yield {"type": "harness_heartbeat"}
yield {"type": "message_end", "message": {"role": "assistant",
"stopReason": "stop",
"content": [{"type": "text", "text": "status: success"}]}}
yield {"type": "agent_end", "messages": []}
return runner
def _stop_runner(task: str, work_dir: Path):
yield {"type": "message_end", "message": {"role": "assistant",
"stopReason": "stop",
"content": [{"type": "text", "text": "status: success\n\n建议人工核对。"}]}}
yield {"type": "agent_end", "messages": []}
async def _stage(store: FakeStore, tmp_path: Path, runner, query: str):
reply = await fallback_lane.propose_reply(
store, "s1", _intent(query), runner=runner, config=_cfg(tmp_path))
confirm_ids = []
if reply is not None and getattr(reply, "blocks", None):
confirm_ids = [b.props["confirmId"] for b in reply.blocks
if getattr(b, "type", "") == "confirm-card"]
return reply, confirm_ids
def _approve(store: FakeStore, confirm_id: str, *, approve: bool = True,
actor: str = "tester") -> str:
return execute_confirmed(store, confirm_id, approve=approve, actor=actor)
def _first_order_no(store: FakeStore) -> str:
return store.data["salesOrders"][0]["orderNo"]
def _orders_rows(n: int = 3, prefix: str = "RY") -> list[dict]:
return [{"customerName": "锐扬精密", "productCode": DEMO_PRODUCT,
"quantity": 10 + i, "deliveryDate": "2026-09-20",
"orderNo": f"{prefix}-{9001 + i}"} for i in range(n)]
def _write_artifact(run_dir: Path, name: str, payload: dict) -> str:
path = run_dir / "outbox" / "artifacts" / name
path.parent.mkdir(parents=True, exist_ok=True)
blob = json.dumps(payload, ensure_ascii=False)
path.write_text(blob, encoding="utf-8")
return hashlib.sha256(blob.encode("utf-8")).hexdigest()
def _frozen_import_plan(run_dir: Path, rows: list[dict], *, digest_delta: str = "") -> dict:
artifact = {"batches": [{"kind": "orders", "sheet": "要货单-0903", "rows": rows}]}
sha = _write_artifact(run_dir, "step1-orders.json", artifact)
return {
"planVersion": 1, "scenario": "S3", "goal": "把客户文件的订单导入订单池",
"steps": [{
"seq": 1, "mode": "frozen", "intent": "import.commit",
"summary": f"导入订单批 {len(rows)} 行(kind=orders)",
"artifactRef": "outbox/artifacts/step1-orders.json",
"artifactSha256": sha + digest_delta,
"params": None,
"constraints": {"kinds": ["orders"], "maxRows": 500},
"expected": [{"table": "flexOrders", "added": len(rows)}],
}],
}
def _inline_import_plan(rows: list[dict]) -> dict:
return {
"planVersion": 1, "scenario": "S3", "goal": "导入订单",
"steps": [{
"seq": 1, "mode": "frozen", "intent": "import.commit",
"summary": "导入订单批 1 行",
"params": {"batches": [{"kind": "orders", "rows": rows}]},
"constraints": {"kinds": ["orders"], "maxRows": 500},
"expected": [{"table": "salesOrders", "added": len(rows)}],
}],
}
def _assisted_complete_plan(order_no: str) -> dict:
return {
"planVersion": 1, "scenario": "S2", "goal": "修复旧单状态",
"steps": [{
"seq": 1, "mode": "assisted", "intent": "order.complete",
"summary": f"把旧单 {order_no} 标记完成",
"params": None,
"constraints": {"allowedParamKeys": ["orderNo"], "orderNoPrefix": "SO"},
"expected": [{"table": "salesOrders", "modified": 1}],
}],
}
def _s4_plan(steps: list[dict] | None = None) -> dict:
return {"planVersion": 1, "scenario": "S4", "goal": "缺策略沙盒试排",
"steps": steps or [{
"seq": 1, "mode": "frozen", "intent": "flex.simulate_due",
"summary": "交期探测", "constraints": {},
"params": {"productCode": DEMO_PRODUCT, "quantity": 10},
"expected": [{"sandboxOutput": "乐观/预计/悲观完工时间"}],
}]}
def _seed_s6_world(store: FakeStore, items: list[tuple]) -> None:
store.data.setdefault("workOrders", [])
store.data.setdefault("mesLinks", [])
for wo_id, ext_id, status, pct, qty in items:
store.data["workOrders"].append({
"id": wo_id, "mesExternalId": ext_id, "status": status,
"progressPct": pct, "qtyDone": qty, "productCode": DEMO_PRODUCT})
store.data["mesLinks"].append({
"kind": "dispatch", "woId": wo_id, "externalWoId": ext_id,
"idemKey": f"idem-{wo_id}"})
def _seed_mes_mirror(tmp_path: Path, monkeypatch, count: int):
from server.integrations import mes_stub
from server.integrations.mes_stub import MockMesClient
client = MockMesClient(tmp_path / "mes_mirror.json")
monkeypatch.setattr(mes_stub, "_client", client)
for i in range(count):
client.create_work_order({"productCode": DEMO_PRODUCT}, idem_key=f"idem-m-{i}")
return client
def _s6_plan(items: list[dict]) -> dict:
return {"planVersion": 1, "scenario": "S6", "goal": "MES 断连补录",
"steps": [{"seq": i + 1, "mode": "frozen", "intent": "mes.report",
"summary": f"补录工单 {p['woId']}", "params": p,
"constraints": {}, "expected": []}
for i, p in enumerate(items)]}
def _report_links(store: FakeStore) -> list[dict]:
return [l for l in store.data.get("mesLinks") or [] if l.get("kind") == "report"]
def _mes_report_audits(store: FakeStore) -> list[dict]:
return [e for e in store.data.get("auditEvents") or []
if e.get("action") == "mes.report" and e.get("category") == "INTEGRATION"]
def _mirror_reports(client) -> list[dict]:
return client._load().get("reports") or []
def _jsonl_lines(path: Path) -> list[dict]:
if not path.exists():
return []
return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()
if line.strip()]
# ---------------------------------------------------------------------------
# P4 指标聚合(result.json/events/calls/审计/确认卡,全部在当前产品字段内)
# ---------------------------------------------------------------------------
def _metric_row(qid: str, store: FakeStore, reply, *,
run_dir: Path | None = None, expected_confirm_cards: int,
expected_approval_rounds: int, expected_ok: bool,
confirm_ids: list[str] | None = None) -> dict:
"""逐场景收集指标;不引入新产品字段,只锁现有语义。"""
result = json.loads((run_dir / "result.json").read_text(encoding="utf-8")) \
if run_dir is not None and (run_dir / "result.json").exists() else {}
event_count = len(_jsonl_lines(run_dir / "events.jsonl")) \
if run_dir is not None else 0
exec_event_count = len(_jsonl_lines(run_dir / "execution.events.jsonl")) \
if run_dir is not None else 0
calls = _jsonl_lines(run_dir / "calls.jsonl") if run_dir is not None else []
output_bytes = int(result.get("output_bytes") or 0)
audit_events = store.data.get("auditEvents") or []
confirm_cards = len([b for b in (getattr(reply, "blocks", None) or [])
if getattr(b, "type", "") == "confirm-card"])
history = getattr(getattr(harness, "_approval_store", None), "history", [])
card_ids = set(confirm_ids or [])
approval_rounds = len([
h for h in history
if h.get("confirmId") in card_ids and h.get("status") in {
"APPROVED", "REJECTED", "SOD_DENIED", "EXPIRED", "GRANT_EXPIRED"}])
row = {
"qid": qid,
"expected_ok": expected_ok,
"ok": bool(result.get("ok")),
"stop_reason": result.get("stop_reason", ""),
"confirm_cards": confirm_cards,
"expected_confirm_cards": expected_confirm_cards,
"approval_rounds": approval_rounds,
"expected_approval_rounds": expected_approval_rounds,
"result": result,
"events": event_count + exec_event_count,
"calls": len(calls),
"steps": int(result.get("steps") or 0),
"latency_sec": float(result.get("elapsed_sec") or 0),
"output_bytes": output_bytes,
"audit_count": len(audit_events),
"cost_points": round(
(event_count + exec_event_count) / 10 + len(calls)
+ int(result.get("steps") or 0) + output_bytes / 4096, 3),
}
row["confirm_cards_ok"] = confirm_cards == expected_confirm_cards
row["approval_rounds_ok"] = approval_rounds == expected_approval_rounds
return row
def _aggregate_metrics(rows: list[dict]) -> dict:
total = len(rows)
success_expected = [r for r in rows if r["expected_ok"]]
ok_runs = [r for r in success_expected
if r["ok"] and r["confirm_cards_ok"] and r["approval_rounds_ok"]]
violations = [
r for r in rows
if not r["confirm_cards_ok"] or not r["approval_rounds_ok"]
or (r["expected_ok"] and not r["ok"])
or any(a.get("result") == "FAILED" and not (
(a.get("rationale") or {}).get("rolledBack")
and (a.get("rationale") or {}).get("rollbackVerified"))
for a in []) # 该子句在 _false_write_rows 里做完整版本,占位避免双统计
]
return {
"total": total,
"success_expected": len(success_expected),
"ok": len(ok_runs),
"success_rate": (len(ok_runs) / len(success_expected)) if success_expected else None,
"violations": len(violations),
"false_write_rate": len(violations) / total if total else 0.0,
"avg_events": round(sum(r["events"] for r in rows) / total, 2) if total else 0,
"avg_calls": round(sum(r["calls"] for r in rows) / total, 2) if total else 0,
"avg_latency_sec": round(
sum(r["latency_sec"] for r in rows) / total, 3) if total else 0,
"avg_cost_points": round(
sum(r["cost_points"] for r in rows) / total, 3) if total else 0,
"rows": rows,
}
def _false_write_rows(store_rows: list[tuple[dict, FakeStore]]) -> list[dict]:
"""误写率守卫:失败/拒绝发生世界写、成功执行 verdict=MISMATCH、回滚未验证。"""
bad = []
for row, store in store_rows:
audits = _world_write_audits(store)
for audit in audits:
r = audit.get("rationale") or {}
if audit.get("result") in {"FAILED", "DENIED"}:
rolled_back = r.get("rolledBack")
verified = r.get("rollbackVerified")
if not (rolled_back and verified) and audit.get("action") in {
"agent.fallback.execute", "agent.fallback.policy.update",
"agent.fallback.ops.config.apply"}:
bad.append({"qid": row["qid"], "reason": "失败未回滚/未验证"})
if audit.get("result") == "SUCCESS" and row.get("expected_ok") is False:
bad.append({"qid": row["qid"], "reason": "失败场景出现成功 WORLD_WRITE"})
run_dir = row.get("result", {}).get("run_dir")
if run_dir:
report = Path(str(run_dir)) / "outbox" / "verify-report.md"
if report.exists() and "verdict: MISMATCH" in report.read_text(encoding="utf-8"):
bad.append({"qid": row["qid"], "reason": "执行 verify MISMATCH"})
return bad
def _format_p4_report(metrics: dict, false_writes: list[dict]) -> str:
rate = metrics["success_rate"]
success_pct = "100.0" if rate is None else f"{rate * 100:.1f}"
lines = [
"P4 确定性质量门禁报告(fake runner,零真实 LLM/网络)",
(f"成功率: {success_pct}%(期望成功 run: {metrics['ok']}/"
f"{metrics['success_expected']};门禁: >=85.0%)"),
(f"误写率: {metrics['false_write_rate'] * 100:.1f}%(门禁: 0%;"
f"误写 run: {len(false_writes)})"),
"确认轮次吻合: 全场景逐 row 校验(场景期望见 metrics rows)",
(f"平均事件数近似: {metrics['avg_events']};平均 calls: {metrics['avg_calls']};"
f"平均耗时近似: {metrics['avg_latency_sec']}s;"
f"平均 cost_points 近似: {metrics['avg_cost_points']}"),
]
if false_writes:
lines.append("误写详情: " + json.dumps(false_writes, ensure_ascii=False))
return "\n".join(lines)
@pytest.mark.parametrize(("qid", "with_tool", "report_template"), [
("Q-01", True, "status: success\n\n报告正文:订单快照已读 [callId: {call_id}]"),
("Q-02", False, "status: success\n\n纯推理结论:建议人工复核订单结构。"),
])
async def test_q01_q02_readonly_success(qid, with_tool, report_template, tmp_path):
# 场景: Q-01/Q-02
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
fp_before = _fp(store.data)
if with_tool:
runner = make_success_runner(report_template)
else:
def runner(task: str, work_dir: Path):
yield {"type": "message_end", "message": {"role": "assistant",
"stopReason": "stop",
"content": [{"type": "text", "text": report_template}]}}
yield {"type": "agent_end", "messages": []}
reply = await fallback_lane.propose_reply(
store, "s1", _intent("帮我分析下订单结构"),
runner=runner, config=_cfg(tmp_path))
assert reply is not None
assert not getattr(reply, "blocks", None) # 只读无确认卡
assert "[智能兜底 · 草稿]" in reply.text
run_dir = _run_dir_of(tmp_path)
result = json.loads((run_dir / "result.json").read_text(encoding="utf-8"))
assert result["ok"] is True and result["stop_reason"] == "stop"
assert result["steps"] == (1 if with_tool else 0)
assert (run_dir / "outbox" / "report.md").is_file()
assert _fp(store.data) == fp_before # 世界零写
audits = _fb_audits(store)
assert len(audits) == 1 and audits[0]["result"] == "SUCCESS"
# 审计/事件/calls 均可回读,作为指标聚合素材
assert (run_dir / "events.jsonl").is_file()
if with_tool:
assert len(_jsonl_lines(run_dir / "calls.jsonl")) >= 1
async def _forged_report_runner():
forged = "call-" + str(uuid.uuid4())
report = f"status: success\n\n编造的数据结论 [callId: {forged}]"
def runner(task: str, work_dir: Path):
yield {"type": "message_end", "message": {"role": "assistant",
"stopReason": "stop",
"content": [{"type": "text", "text": report}]}}
yield {"type": "agent_end", "messages": []}
return runner
async def _empty_report_runner():
def runner(task: str, work_dir: Path):
yield {"type": "message_end", "message": {"role": "assistant",
"stopReason": "stop", "content": []}}
yield {"type": "agent_end", "messages": []}
return runner
async def _boom_runner():
def runner(task: str, work_dir: Path):
raise RuntimeError("spawn exploded")
yield # pragma: no cover - 保持生成器形态
return runner
async def _heartbeat_runner():
def runner(task: str, work_dir: Path):
while True:
yield {"type": "harness_heartbeat"}
return runner
async def _busy_runner():
def runner(task: str, work_dir: Path):
for i in range(3):
yield {"type": "tool_execution_start", "toolName": "read",
"toolCallId": f"t{i}"}
yield {"type": "agent_end", "messages": []}
return runner
async def _huge_output_runner():
def runner(task: str, work_dir: Path):
chunk = "x" * 4096
for _ in range(4):
yield {"type": "message_update", "delta": {"text": chunk}}
yield {"type": "message_end", "message": {"role": "assistant",
"stopReason": "stop", "content": [{"type": "text", "text": chunk}]}}
yield {"type": "agent_end", "messages": []}
return runner
@pytest.mark.parametrize(("qid", "cfg_kwargs", "expected_stop", "runner_kind"), [
("Q-03", {}, "forged_citation", "forged"),
("Q-04", {}, "error:empty_report", "empty"),
("Q-05", {}, "harness_error", "boom"),
("Q-06", {"timeout_sec": 0.05}, "breaker:timeout", "heartbeat"),
("Q-07", {"max_steps": 2}, "breaker:max_steps", "busy"),
("Q-08", {"max_output_bytes": 1024}, "breaker:max_output", "huge"),
])
async def test_q03_to_q08_failure_matrix(qid, cfg_kwargs, expected_stop, runner_kind,
tmp_path):
# 场景: Q-03/Q-04/Q-05/Q-06/Q-07/Q-08
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
fp_before = _fp(store.data)
builders = {
"forged": _forged_report_runner,
"empty": _empty_report_runner,
"boom": _boom_runner,
"heartbeat": _heartbeat_runner,
"busy": _busy_runner,
"huge": _huge_output_runner,
}
runner = await builders[runner_kind]()
reply = await fallback_lane.propose_reply(
store, "s1", _intent("随便说说"), runner=runner,
config=_cfg(tmp_path, **cfg_kwargs))
assert reply is not None
assert reply.text == "本次处理未完成,未执行任何操作。请稍后重试。"
audits = _fb_audits(store)
assert len(audits) == 1 and audits[0]["result"] == "FAILED"
assert audits[0]["rationale"]["stopReason"].startswith(expected_stop)
assert _fp(store.data) == fp_before # 失败零世界写
assert not getattr(reply, "blocks", None)
async def test_q09_snapshot_export_failure_degrades_to_pure_reasoning(tmp_path, monkeypatch):
# 场景: Q-09
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
fp_before = _fp(store.data)
def snapshot_fails(self, world, dirs):
raise RuntimeError("injected snapshot failure")
monkeypatch.setattr(PiBridge, "export_snapshot", snapshot_fails)
report = "status: success\n\n纯推理:快照不可用仍能给出草稿结论。"
def pure_reasoning_runner(task: str, work_dir: Path):
yield {"type": "message_end", "message": {"role": "assistant",
"stopReason": "stop",
"content": [{"type": "text", "text": report}]}}
yield {"type": "agent_end", "messages": []}
reply = await fallback_lane.propose_reply(
store, "s1", _intent("分析订单"),
runner=pure_reasoning_runner, config=_cfg(tmp_path))
assert reply is not None
assert "纯推理:快照不可用" in reply.text
run_dir = _run_dir_of(tmp_path)
log = (run_dir / "orchestrator.log").read_text(encoding="utf-8")
assert "快照导出失败" in log
result = json.loads((run_dir / "result.json").read_text(encoding="utf-8"))
assert result["ok"] is True
assert _fp(store.data) == fp_before
@pytest.mark.parametrize("mode", ["missing", "corrupt", "nonbool"])
async def test_q10_feature_switch_fail_closed(mode, tmp_path):
# 场景: Q-10
path = tmp_path / "features.json"
if mode == "corrupt":
path.write_text("{not json", encoding="utf-8")
elif mode == "nonbool":
path.write_text(json.dumps({"version": 1, "features": {"fallback": "yes"}},
ensure_ascii=False), encoding="utf-8")
# missing:不写文件
store = FakeStore(tmp_path)
before_audits = len(store.data.get("auditEvents") or [])
reply = await fallback_lane.propose_reply(
store, "s1", _intent("随便说说"), runner=_stop_runner, config=_cfg(tmp_path))
assert reply is None # 开关关 → 原路径零副作用
assert len(store.data.get("auditEvents") or []) == before_audits
assert _run_dirs(tmp_path) == []
flags = load_feature_flags(str(path))
assert flags["features"]["fallback"]["enabled"] is False
async def test_q11_s3_legal_import_plan_stages_card(tmp_path):
# 场景: Q-11
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
rows = _orders_rows(3)
reply = await fallback_lane.propose_reply(
store, "s1", _intent("把这份客户表格导进来"),
runner=make_plan_runner(lambda rd: _frozen_import_plan(rd, rows)),
config=_cfg(tmp_path))
assert reply is not None and reply.blocks
block = reply.blocks[0]
assert block.type == "confirm-card"
assert block.props["action"] == "agent.fallback.execute"
assert block.props["power"] == "P2"
confirm_id = block.props["confirmId"]
pending = _pending_record(confirm_id)
assert pending["params"]["planFingerprint"] == \
fallback_lane.plan_fingerprint(pending["params"]["plan"])
assert "fallback-plan" in "|".join(pending.get("evidenceRefs") or [])
stage_audits = _stage_audits(store)
assert len(stage_audits) == 1 and stage_audits[0]["category"] == "GATE"
# 卡片内容由结构化字段生成,Pi 散文 goal 不进卡
assert "把客户文件的订单导入订单池" not in "\n".join(block.props["summary"])
async def test_q12_frozen_execute_success_and_verify(tmp_path):
# 场景: Q-12
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
rows = _orders_rows(3)
orders_before = len(store.data["flexOrders"])
_reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _frozen_import_plan(rd, rows)),
"把这份客户表格导进来")
assert len(confirm_ids) == 1
msg = _approve(store, confirm_ids[0])
assert "兜底计划已执行完成" in msg
assert len(store.data["flexOrders"]) == orders_before + 3
audits = _exec_audits(store)
assert audits[0]["result"] == "SUCCESS"
assert audits[0]["rationale"]["status"] == "success"
assert audits[0]["rationale"]["cpAfter"]
pairs = store.checkpoints.pairs
assert "auto:fallback.execute" in [p["reason"] for p in pairs]
assert "auto:fallback.execute.post" in [p["reason"] for p in pairs]
report = (audits[0]["rationale"].get("verifyReport") or "")
assert report and Path(report).is_file()
assert "verdict: PASS" in Path(report).read_text(encoding="utf-8")
async def test_q13_assisted_success(tmp_path, monkeypatch):
# 场景: Q-13
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
order_no = _first_order_no(store)
_reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _assisted_complete_plan(order_no)),
"把旧单完成状态修复一下")
assert len(confirm_ids) == 1
req = {"seq": 1, "intent": "order.complete", "params": {"orderNo": order_no}}
monkeypatch.setattr(fallback_lane, "build_pi_runner",
lambda config, *, mode="readonly": make_exec_runner([req]))
msg = _approve(store, confirm_ids[0])
assert "兜底计划已执行完成" in msg
assert store.data["salesOrders"][0]["status"] == "COMPLETED"
run_dir = _run_dir_of(tmp_path)
result_file = run_dir / "outbox" / "actions" / "1-order.complete.result.json"
assert json.loads(result_file.read_text(encoding="utf-8"))["ok"] is True
async def test_q14_plan_with_unregistered_intent_refused(tmp_path):
# 场景: Q-14
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
fp_before = _fp(store.data)
def bad_plan(run_dir: Path) -> dict:
return {"planVersion": 1, "scenario": "S3", "goal": "x",
"steps": [{"seq": 1, "mode": "frozen", "intent": "order.explode",
"params": {}, "constraints": {}}]}
reply, confirm_ids = await _stage(store, tmp_path, make_plan_runner(bad_plan),
"导入一下")
assert confirm_ids == []
assert "未通过校验" in reply.text
assert "未在兜底可执行白名单" in reply.text
assert _fp(store.data) == fp_before
async def test_q15_assisted_params_out_of_bounds_rollback(tmp_path, monkeypatch):
# 场景: Q-15
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
rows = _orders_rows(2)
order_before_count = len(store.data["salesOrders"])
def plan(run_dir: Path) -> dict:
return {"planVersion": 1, "scenario": "S3", "goal": "x",
"steps": [{"seq": 1, "mode": "assisted", "intent": "import.commit",
"summary": "导入", "params": None,
"constraints": {"maxRows": 1, "kinds": ["orders"],
"allowedParamKeys": ["batches"]},
"expected": []}]}
_reply, confirm_ids = await _stage(store, tmp_path, make_plan_runner(plan),
"把数据导进来")
assert len(confirm_ids) == 1
req = {"seq": 1, "intent": "import.commit",
"params": {"batches": [{"kind": "orders", "rows": rows}]}} # 2 行 > 1
monkeypatch.setattr(fallback_lane, "build_pi_runner",
lambda config, *, mode="readonly": make_exec_runner([req]))
msg = _approve(store, confirm_ids[0])
assert "已熔断并自动回滚" in msg
audit = _exec_audits(store)[0]
assert audit["result"] == "FAILED"
assert audit["rationale"]["deviation"].startswith("params:")
cp_before = store.checkpoints.get(audit["beforeSnapshot"])
assert _fp(store.data) == _fp(cp_before["world"])
assert len(store.data["salesOrders"]) == order_before_count
async def test_q16_assisted_extra_step_rollback(tmp_path, monkeypatch):
# 场景: Q-16
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
order_no = _first_order_no(store)
_reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _assisted_complete_plan(order_no)),
"修复旧单")
assert len(confirm_ids) == 1
second = store.data["salesOrders"][1]["orderNo"]
requests = [
{"seq": 1, "intent": "order.complete", "params": {"orderNo": order_no}},
{"seq": 2, "intent": "order.complete", "params": {"orderNo": second}},
]
monkeypatch.setattr(fallback_lane, "build_pi_runner",
lambda config, *, mode="readonly": make_exec_runner(requests))
msg = _approve(store, confirm_ids[0])
assert "已熔断并自动回滚" in msg
audit = _exec_audits(store)[0]
assert audit["rationale"]["deviation"].startswith("step_count:")
cp_before = store.checkpoints.get(audit["beforeSnapshot"])
assert _fp(store.data) == _fp(cp_before["world"])
assert store.data["salesOrders"][0]["status"] == "APPROVED"
async def test_q17_artifact_digest_mismatch_refused(tmp_path):
# 场景: Q-17
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
rows = _orders_rows(2)
reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _frozen_import_plan(rd, rows, digest_delta="00")),
"把数据导进来")
assert confirm_ids == []
assert "指纹虚报" in reply.text
async def test_q18_world_drift_between_stage_and_approve_denied(tmp_path, monkeypatch):
# 场景: Q-18
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
rows = _orders_rows(2)
monkeypatch.setattr(harness, "_capture_world_fingerprint",
lambda tenant_uuid, world_key: _fp(store.data))
_reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _frozen_import_plan(rd, rows)),
"把数据导进来")
assert len(confirm_ids) == 1
orders_before = len(store.data["flexOrders"])
store.data["salesOrders"][0]["priority"] = 99 # 审批窗口内世界漂移
msg = _approve(store, confirm_ids[0])
assert "世界指纹漂移" in msg
assert len(store.data["flexOrders"]) == orders_before
assert store.checkpoints.pairs == []
assert _exec_audits(store)[0]["result"] == "DENIED"
async def test_q19_expired_card_denied_zero_write(tmp_path, monkeypatch):
# 场景: Q-19
monkeypatch.setenv("APS_APPROVAL_TTL_SECONDS", "1")
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
rows = _orders_rows(2)
orders_before = len(store.data["salesOrders"])
_reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _frozen_import_plan(rd, rows)),
"把数据导进来")
assert len(confirm_ids) == 1
_mutate_pending(confirm_ids[0],
lambda rec: rec.__setitem__("expiresAtEpoch", 1_500_000_000))
msg = _approve(store, confirm_ids[0])
assert "已失效" in msg
assert len(store.data["salesOrders"]) == orders_before
assert store.checkpoints.pairs == []
assert _exec_audits(store) == []
async def test_q20_execution_exception_rolls_back(tmp_path):
# 场景: Q-20
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
order_no = _first_order_no(store)
def plan(run_dir: Path) -> dict:
return {"planVersion": 1, "scenario": "S2", "goal": "x",
"steps": [
{"seq": 1, "mode": "frozen", "intent": "order.complete",
"summary": "正常步", "params": {"orderNo": order_no},
"constraints": {}, "expected": []},
{"seq": 2, "mode": "frozen", "intent": "order.complete",
"summary": "坏步(目标不存在)",
"params": {"orderNo": "SO-NOT-EXIST"}, "constraints": {},
"expected": []},
]}
_reply, confirm_ids = await _stage(store, tmp_path, make_plan_runner(plan),
"修复旧单")
assert len(confirm_ids) == 1
msg = _approve(store, confirm_ids[0])
assert "兜底执行失败" in msg and "已自动回滚" in msg
audit = _exec_audits(store)[0]
assert audit["result"] == "FAILED"
assert audit["rationale"]["rolledBack"] is True
assert audit["rationale"]["rollbackVerified"] is True
cp_before = store.checkpoints.get(audit["beforeSnapshot"])
assert _fp(store.data) == _fp(cp_before["world"])
assert store.data["salesOrders"][0]["status"] == "APPROVED"
async def test_q21_approval_rejected_zero_execution(tmp_path):
# 场景: Q-21
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
rows = _orders_rows(2)
orders_before = len(store.data["salesOrders"])
_reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _frozen_import_plan(rd, rows)),
"把数据导进来")
assert len(confirm_ids) == 1
msg = _approve(store, confirm_ids[0], approve=False)
assert "已驳回" in msg
assert len(store.data["salesOrders"]) == orders_before
assert store.checkpoints.pairs == []
assert _exec_audits(store) == []
async def test_q22_p2_unaffected_by_missing_whitelist(tmp_path):
# 场景: Q-22
_write_features(tmp_path, {"fallback": True}) # 白名单文件刻意缺失
store = FakeStore(tmp_path)
loaded = fallback_highrisk.load_highrisk_whitelist()
assert loaded["ok"] is False
reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _inline_import_plan(_orders_rows(1))),
"导入一张客户订单")
assert len(confirm_ids) == 1
assert reply.blocks[0].props["power"] == "P2"
async def test_q23_s4_sandbox_success_no_main_write(tmp_path):
# 场景: Q-23
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
audits_before = len(store.data.get("auditEvents") or [])
_reply, confirm_ids = await _stage(store, tmp_path,
make_plan_runner(lambda rd: _s4_plan()),
"这个工艺没有现成策略,帮我试几个思路")
assert len(confirm_ids) == 1
fp_before = _fp(store.data)
msg = _approve(store, confirm_ids[0])
assert "沙盒试排已完成" in msg
run_dir = _run_dir_of(tmp_path)
step_out = run_dir / "outbox" / "sandbox-steps" / "step-1.json"
assert step_out.is_file()
assert json.loads(step_out.read_text(encoding="utf-8"))
assert (run_dir / "outbox" / "sandbox-report.md").is_file()
assert _fp(store.data) == fp_before
run_audits = [e for e in store.data.get("auditEvents", [])[audits_before:]
if e.get("category") == "WORLD_WRITE"]
assert len(run_audits) == 1 # 仅执行成功总账
assert run_audits[0]["result"] == "SUCCESS"
async def test_q24_s4_mixed_write_intent_refused(tmp_path):
# 场景: Q-24
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
def bad_plan(run_dir: Path) -> dict:
return _s4_plan([{
"seq": 1, "mode": "frozen", "intent": "import.commit",
"summary": "x", "constraints": {},
"params": {"batches": []}, "expected": []}])
reply, confirm_ids = await _stage(store, tmp_path,
make_plan_runner(bad_plan),
"帮我试几个思路")
assert confirm_ids == []
assert "未列入 S4 白名单放行集" in reply.text
async def test_q25_s4_expected_world_change_fields_refused(tmp_path):
# 场景: Q-25
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
def bad_plan(run_dir: Path) -> dict:
return _s4_plan([{
"seq": 1, "mode": "frozen", "intent": "flex.simulate_due",
"summary": "x", "constraints": {},
"params": {"productCode": DEMO_PRODUCT, "quantity": 10},
"expected": [{"table": "workOrders", "added": 1}]}])
reply, confirm_ids = await _stage(store, tmp_path,
make_plan_runner(bad_plan),
"帮我试几个思路")
assert confirm_ids == []
assert "sandboxOutput" in reply.text
async def test_q26_s6_single_legal_backlog(tmp_path, monkeypatch):
# 场景: Q-26
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
_seed_mes_mirror(tmp_path, monkeypatch, 1)
_seed_s6_world(store, [(9001, "MES-WO-0001", "RUNNING", 0, 0)])
_reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _s6_plan(
[{"woId": 9001, "progressPct": 80, "track": "fixed"}])),
"MES 连不上了,帮我补录")
assert len(confirm_ids) == 1
msg = _approve(store, confirm_ids[0])
assert "执行完成" in msg
assert store.data["workOrders"][0]["progressPct"] == 80
async def test_q27_s6_two_sibling_cards(tmp_path, monkeypatch):
# 场景: Q-27
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
_seed_mes_mirror(tmp_path, monkeypatch, 2)
_seed_s6_world(store, [
(9001, "MES-WO-0001", "RUNNING", 0, 0),
(9002, "MES-WO-0002", "RUNNING", 0, 0),
])
_reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _s6_plan([
{"woId": 9001, "progressPct": 80, "track": "fixed"},
{"woId": 9002, "progressPct": 60, "track": "fixed"}])),
"MES 连不上了,补录两笔")
assert len(confirm_ids) == 2
assert len(set(confirm_ids)) == 2
assert len({_pending_record(cid)["params"]["planFingerprint"]
for cid in confirm_ids}) == 2
msg1 = _approve(store, confirm_ids[0])
msg2 = _approve(store, confirm_ids[1])
assert "执行完成" in msg1 and "执行完成" in msg2
wos = {w["id"]: w for w in store.data["workOrders"]}
assert wos[9001]["progressPct"] == 80
assert wos[9002]["progressPct"] == 60
async def test_q28_s6_over_max_truncated(tmp_path, monkeypatch):
# 场景: Q-28
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc(s6_max=2))
store = FakeStore(tmp_path)
_seed_mes_mirror(tmp_path, monkeypatch, 3)
_seed_s6_world(store, [
(9001, "MES-WO-0001", "RUNNING", 0, 0),
(9002, "MES-WO-0002", "RUNNING", 0, 0),
(9003, "MES-WO-0003", "RUNNING", 0, 0),
])
items = [{"woId": w, "progressPct": 50, "track": "fixed"} for w in (9001, 9002, 9003)]
reply, confirm_ids = await _stage(
store, tmp_path, make_plan_runner(lambda rd: _s6_plan(items)),
"MES 连不上了帮我补录")
assert len(confirm_ids) == 2
assert "剩余 1 笔" in reply.text
assert "单轮上限 2" in reply.text
async def test_q29_s6_invalid_backlog_refused(tmp_path):
# 场景: Q-29
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
_seed_s6_world(store, [(9001, "MES-WO-0001", "RUNNING", 0, 0)])
reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _s6_plan(
[{"woId": 9999, "progressPct": 50, "track": "fixed"}])),
"MES 连不上了帮我补录")
assert confirm_ids == []
assert "不存在" in reply.text
reply2, ids2 = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _s6_plan(
[{"woId": 9001, "progressPct": 120, "track": "fixed"}])),
"MES 连不上了帮我补录")
assert ids2 == []
assert "[0,100]" in reply2.text
async def test_q30_s6_reconcile_match_drift_missing(tmp_path, monkeypatch):
# 场景: Q-30
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
_seed_s6_world(store, [
(9101, "MES-WO-0001", "RUNNING", 50, 5),
(9102, "MES-WO-0002", "RUNNING", 50, 5),
(9103, "MES-WO-0003", "RUNNING", 40, 4),
])
monkeypatch.setattr(fallback_lane, "_probe_mes_connectivity", lambda: "ok")
snapshots = {
"MES-WO-0001": {"id": "MES-WO-0001", "status": "RUNNING",
"progressPct": 50, "qtyDone": 5}, # MATCH
"MES-WO-0002": {"id": "MES-WO-0002", "status": "COMPLETED",
"progressPct": 100, "qtyDone": 10}, # DRIFT
"MES-WO-0003": None, # MISSING
}
monkeypatch.setattr(fallback_lane, "_fetch_external_wo",
lambda ext_id: snapshots.get(ext_id))
fp_before = _fp(store.data)
reply = await fallback_lane.propose_reply(
store, "s1", _intent("MES 恢复了,对一下账",
p3Scenario="S6", p3Action="reconcile"),
runner=_stop_runner, config=_cfg(tmp_path))
assert reply is not None
assert "一致 1" in reply.text and "漂移 1" in reply.text and "单边缺失 1" in reply.text
run_dir = _run_dir_of(tmp_path)
report = (run_dir / "outbox" / "reconcile-report.md").read_text(encoding="utf-8")
assert "| MES-WO-0001 | 9101 | MATCH |" in report
assert "| MES-WO-0002 | 9102 | DRIFT |" in report
assert "| MES-WO-0003 | 9103 | MISSING |" in report
recon = [e for e in store.data.get("auditEvents") or []
if e.get("action") == "agent.fallback.reconcile"]
assert recon[-1]["rationale"]["match"] == 1
assert recon[-1]["rationale"]["drift"] == 1
assert recon[-1]["rationale"]["missing"] == 1
assert _fp(store.data) == fp_before
async def test_q31_s6_reconcile_still_down_explicit(tmp_path, monkeypatch):
# 场景: Q-31
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
_seed_s6_world(store, [(9101, "MES-WO-0001", "RUNNING", 50, 5)])
monkeypatch.setattr(fallback_lane, "_probe_mes_connectivity", lambda: "failed")
fp_before = _fp(store.data)
reply = await fallback_lane.propose_reply(
store, "s1", _intent("MES 恢复了,对一下账",
p3Scenario="S6", p3Action="reconcile"),
runner=_stop_runner, config=_cfg(tmp_path))
assert reply is not None
assert "尚未恢复" in reply.text
assert _run_dirs(tmp_path) == []
assert _fp(store.data) == fp_before
async def test_q32_s6_offline_booking_success(tmp_path, monkeypatch):
# 场景: Q-32
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
client = _seed_mes_mirror(tmp_path, monkeypatch, 1)
_seed_s6_world(store, [(9001, "MES-WO-0001", "RUNNING", 0, 0)])
monkeypatch.setattr(fallback_lane, "_probe_mes_connectivity", lambda: "failed")
reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _s6_plan(
[{"woId": 9001, "progressPct": 80, "track": "fixed",
"offlineBooking": True}])),
"MES 连不上了,今天报工帮我补一下")
assert len(confirm_ids) == 1
assert "PENDING_SYNC" in json.dumps(
reply.blocks[0].props.get("summary"), ensure_ascii=False)
msg = _approve(store, confirm_ids[0])
assert "执行完成" in msg
links = _report_links(store)
assert len(links) == 1 and links[0].get("syncStatus") == "PENDING_SYNC"
assert _mirror_reports(client) == [] # 未触真实 MES post
audits = _mes_report_audits(store)
assert audits[0]["rationale"].get("offlineBooking") is True
assert audits[0]["rationale"].get("syncStatus") == "PENDING_SYNC"
async def test_q33_s6_offline_declared_while_online_stage_denied(tmp_path, monkeypatch):
# 场景: Q-33
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
_seed_mes_mirror(tmp_path, monkeypatch, 1)
_seed_s6_world(store, [(9001, "MES-WO-0001", "RUNNING", 0, 0)])
monkeypatch.setattr(fallback_lane, "_probe_mes_connectivity", lambda: "ok")
fp_before = _fp(store.data)
reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _s6_plan(
[{"woId": 9001, "progressPct": 80, "track": "fixed",
"offlineBooking": True}])),
"MES 连不上了,帮我补录")
assert confirm_ids == []
assert "离线落账只允许在断连事实下使用" in reply.text
assert _fp(store.data) == fp_before
assert _report_links(store) == []
async def test_q34_s6_mes_recovers_inside_approval_window(tmp_path, monkeypatch):
# 场景: Q-34
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
_seed_mes_mirror(tmp_path, monkeypatch, 1)
_seed_s6_world(store, [(9001, "MES-WO-0001", "RUNNING", 0, 0)])
probe = {"state": "failed"}
monkeypatch.setattr(fallback_lane, "_probe_mes_connectivity",
lambda: probe["state"])
_reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _s6_plan(
[{"woId": 9001, "progressPct": 80, "track": "fixed",
"offlineBooking": True}])),
"MES 连不上了,今天报工帮我补一下")
assert len(confirm_ids) == 1
probe["state"] = "ok" # 审批窗口内 MES 恢复
msg = _approve(store, confirm_ids[0])
assert "已熔断并自动回滚" in msg
assert store.data["workOrders"][0]["progressPct"] == 0
assert _report_links(store) == []
async def test_q35_s7_ops_diagnostics_redacted(tmp_path, monkeypatch):
# 场景: Q-35
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
home = tmp_path / "home"
(home / "logs").mkdir(parents=True)
(home / "logs" / "server.log").write_text(
"INFO startup ok\nERROR auth failed token=abc123secret\n", encoding="utf-8")
monkeypatch.setenv("APS_HOME", str(home))
fp_before = _fp(store.data)
with _as(7, "ops"):
reply = await fallback_lane.propose_reply(
store, "s1", _intent("帮我看下服务日志和配置有没有异常"),
runner=_stop_runner, config=_cfg(tmp_path))
assert reply is not None
run_dir = _run_dir_of(tmp_path)
ops_dir = run_dir / "inbox" / "ops"
for name in ("logs-tail.md", "config-snapshot.md", "health.md", "integrations.md"):
assert (ops_dir / name).is_file(), name
logs = (ops_dir / "logs-tail.md").read_text(encoding="utf-8")
assert "***REDACTED***" in logs and "abc123secret" not in logs
assert _fp(store.data) == fp_before
async def test_q36_s7_non_ops_physically_invisible(tmp_path, monkeypatch):
# 场景: Q-36
# Pi 驱动后不再由服务端关键词判定来意:非运维角色根本不注入 S7 读面,
# 服务日志/配置内容对本次运行物理不可见,回复里也不能出现它们。
_write_features(tmp_path, {"fallback": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
home = tmp_path / "home"
(home / "logs").mkdir(parents=True)
(home / "logs" / "server.log").write_text(
"INFO startup ok\nERROR auth failed token=abc123secret\n", encoding="utf-8")
monkeypatch.setenv("APS_HOME", str(home))
with _as(42, "planner"):
reply = await fallback_lane.propose_reply(
store, "s1", _intent("帮我看下服务日志有没有异常"),
runner=_stop_runner, config=_cfg(tmp_path))
assert reply is not None
assert "abc123secret" not in reply.text
assert "auth failed" not in reply.text
assert [p for p in tmp_path.rglob("logs-tail.md")] == []
assert [p for p in tmp_path.rglob("config-snapshot.md")] == []
assert [p for p in tmp_path.rglob("ops") if p.is_dir()] == []
async def test_q37_s7_policy_update_via_card(tmp_path):
# 场景: Q-37
_write_features(tmp_path, {"fallback": True})
initial = _wl_doc(s6=False, s6_intents=[])
_write_whitelist(tmp_path, initial)
store = FakeStore(tmp_path)
_bind_personal_world(store, 1)
def policy_plan(run_dir: Path) -> dict:
return {"planVersion": 1, "scenario": "S7", "goal": "开放 S6 补录",
"steps": [{"seq": 1, "mode": "frozen",
"intent": "agent.fallback.policy.update",
"summary": "S6 加 mes.report 并启用",
"params": {"scenario": "S6", "addIntents": ["mes.report"],
"enabled": True},
"constraints": {}}]}
with _as(1, "admin"):
_reply, confirm_ids = await _stage(
store, tmp_path, make_plan_runner(policy_plan),
"帮我把 mes.report 加进 S6 白名单")
assert len(confirm_ids) == 1
document = _pending_record(confirm_ids[0])["params"]["document"]
assert document["scenarios"]["S6"]["enabled"] is True
assert "mes.report" in document["scenarios"]["S6"]["intents"]
with _as(1, "admin"):
msg = _approve(store, confirm_ids[0])
assert "白名单已更新" in msg
loaded = fallback_highrisk.load_highrisk_whitelist()
assert loaded["ok"] is True
assert fallback_highrisk.intent_allowed(loaded, "S6", "mes.report") is True
assert list(tmp_path.glob("fallback-highrisk.json.bak-*"))
audits = [e for e in _world_write_audits(store)
if e.get("action") == "agent.fallback.policy.update"]
assert audits[-1]["result"] == "SUCCESS"
async def test_q38_s7_config_apply_dual_approval(tmp_path):
# 场景: Q-38
_write_features(tmp_path, {"fallback": True, "orders": True})
_write_whitelist(tmp_path, _wl_doc())
store = FakeStore(tmp_path)
_bind_personal_world(store, 1)
new_doc = {"version": 1, "features": {"fallback": True, "orders": False}}
def s7_plan(run_dir: Path) -> dict:
return {"planVersion": 1, "scenario": "S7", "goal": "关闭订单功能",
"steps": [{"seq": 1, "mode": "frozen",
"intent": "agent.fallback.ops.config.apply",
"summary": "替换 features.json",
"params": {"file": "features.json", "content": new_doc},
"constraints": {}}]}
with _as(1, "admin"):
reply, confirm_ids = await _stage(
store, tmp_path, make_plan_runner(s7_plan),
"帮我看下配置,把订单功能关掉")
assert len(confirm_ids) == 1
cid = confirm_ids[0]
assert reply.blocks[0].props["power"] == "P3"
with _as(1, "admin"):
msg1 = _approve(store, cid)
assert "第一重确认已记录" in msg1
with _as(1, "admin"):
msg2 = _approve(store, cid)
assert "另一名用户" in msg2
with _as(2, "admin"):
msg3 = _approve(store, cid)
assert "原子替换" in msg3
replaced = json.loads((tmp_path / "features.json").read_text(encoding="utf-8"))
assert replaced["features"]["orders"] is False
assert list(tmp_path.glob("features.json.bak-*"))
audits = [e for e in _world_write_audits(store)
if e.get("action") == "agent.fallback.ops.config.apply"]
assert audits[-1]["result"] == "SUCCESS"
assert len(audits[-1]["rationale"]["approvals"]) == 2
async def test_q39_whitelist_missing_corrupt_fail_closed(tmp_path):
# 场景: Q-39
_write_features(tmp_path, {"fallback": True})
store = FakeStore(tmp_path)
missing = fallback_highrisk.load_highrisk_whitelist()
assert missing["ok"] is False and "白名单文件缺失" in missing["error"]
# P2 不受白名单缺失影响(Q-22 等价断言,这里复用同一 P2 计划)
_reply, confirm_ids = await _stage(
store, tmp_path,
make_plan_runner(lambda rd: _inline_import_plan(_orders_rows(1))),
"导入一张客户订单")
assert len(confirm_ids) == 1
_write_whitelist(tmp_path, "{not a json")
corrupt = fallback_highrisk.load_highrisk_whitelist()
assert corrupt["ok"] is False and "JSON 解析失败" in corrupt["error"]
for plan in (_s4_plan(),
_s6_plan([{"woId": 9001, "progressPct": 50, "track": "fixed"}])):
with pytest.raises(fallback_lane.PlanError, match="JSON 解析失败"):
fallback_lane.validate_plan(plan, tmp_path, world=store.data)
async def test_q40_runtime_unavailable_fails_explicit(tmp_path, monkeypatch):
# 场景: Q-40
_write_features(tmp_path, {"fallback": True})
monkeypatch.setenv("APS_FALLBACK_PI_CLI", str(tmp_path / "no-such-cli.js"))
popen_calls = []
monkeypatch.setattr(subprocess, "Popen",
lambda *a, **kw: popen_calls.append((a, kw)))
store = FakeStore(tmp_path)
reply = await fallback_lane.propose_reply(store, "s1", _intent("随便说说"))
assert reply is not None
assert popen_calls == [] # 不触子进程
assert reply.text == "智能助手服务暂不可用,本次未执行任何操作。请稍后重试或联系管理员。"
audits = _fb_audits(store)
assert audits[0]["result"] == "FAILED"
assert audits[0]["rationale"]["stopReason"].startswith("unavailable:")
assert _run_dirs(tmp_path) == [] # 不可用不建 run 目录
# ---------------------------------------------------------------------------
# P4 指标聚合断言:成功场景 100%、误写 0、确认轮次吻合、门禁表达
# ---------------------------------------------------------------------------
async def test_p4_metrics_aggregate_success_and_confirm_rounds(tmp_path, monkeypatch):
_write_features(tmp_path, {"fallback": True})
rows: list[dict] = []
store_rows: list[tuple[dict, FakeStore]] = []
# Q-01:只读成功,0 卡 / 0 审批轮
store1 = FakeStore(tmp_path)
reply1 = await fallback_lane.propose_reply(
store1, "s1", _intent("分析订单"),
runner=make_success_runner("status: success\n\n只读结论 [callId: {call_id}]"),
config=_cfg(tmp_path))
row1 = _metric_row("Q-01", store1, reply1, run_dir=_propose_run_dir(store1),
expected_confirm_cards=0, expected_approval_rounds=0,
expected_ok=True, confirm_ids=[])
rows.append(row1)
store_rows.append((row1, store1))
# Q-11+Q-12:合法 import 出 1 卡并批准 1 轮,成功后 verify PASS
store2 = FakeStore(tmp_path)
reply2, cids2 = await _stage(
store2, tmp_path,
make_plan_runner(lambda rd: _frozen_import_plan(rd, _orders_rows(2))),
"把客户表格导进来")
assert len(cids2) == 1
msg2 = _approve(store2, cids2[0])
assert "执行完成" in msg2
row2 = _metric_row("Q-12", store2, reply2, run_dir=_propose_run_dir(store2),
expected_confirm_cards=1, expected_approval_rounds=1,
expected_ok=True, confirm_ids=cids2)
row2["verdict_pass"] = "verdict: PASS" in Path(
str(row2["result"].get("run_dir")) + "/outbox/verify-report.md"
).read_text(encoding="utf-8")
rows.append(row2)
store_rows.append((row2, store2))
# Q-21:驳回 1 轮、0 卡执行、零写
store3 = FakeStore(tmp_path)
reply3, cids3 = await _stage(
store3, tmp_path,
make_plan_runner(lambda rd: _frozen_import_plan(rd, _orders_rows(1))),
"把客户表格导进来")
assert len(cids3) == 1
_approve(store3, cids3[0], approve=False)
row3 = _metric_row("Q-21", store3, reply3, run_dir=_propose_run_dir(store3),
expected_confirm_cards=1, expected_approval_rounds=1,
expected_ok=True, confirm_ids=cids3)
rows.append(row3)
store_rows.append((row3, store3))
# Q-23:S4 沙盒 1 卡 1 轮批准,成功且零主干写
store4 = FakeStore(tmp_path)
_write_whitelist(tmp_path, _wl_doc())
reply4, cids4 = await _stage(
store4, tmp_path, make_plan_runner(lambda rd: _s4_plan()),
"这个工艺没有现成策略,帮我试几个思路")
assert len(cids4) == 1
_approve(store4, cids4[0])
row4 = _metric_row("Q-23", store4, reply4, run_dir=_propose_run_dir(store4),
expected_confirm_cards=1, expected_approval_rounds=1,
expected_ok=True, confirm_ids=cids4)
rows.append(row4)
store_rows.append((row4, store4))
metrics = _aggregate_metrics(rows)
false_writes = _false_write_rows(store_rows)
assert metrics["success_rate"] == 1.0
assert metrics["violations"] == 0
assert false_writes == []
assert metrics["false_write_rate"] == 0.0
for row in rows:
assert row["confirm_cards_ok"], row
assert row["approval_rounds_ok"], row
def test_p4_metric_report_gate_expressions(tmp_path):
fake_metrics = {
"success_rate": 1.0, "ok": 4, "success_expected": 4,
"false_write_rate": 0.0, "avg_events": 8.0, "avg_calls": 1.0,
"avg_latency_sec": 0.02,
"avg_cost_points": 2.0,
}
report = _format_p4_report(fake_metrics, [])
assert "成功率: 100.0%" in report
assert "门禁: >=85.0%" in report
assert "误写率: 0.0%" in report
assert "门禁: 0%" in report
# notes 会记录真实跑测值;本行锁报告层门禁表达不随机器漂移。
assert report.startswith("P4 确定性质量门禁报告")
_Q_REFERENCE_MAP = {
"Q-19": "expired card in this file;replay/tamper 等价见 test_fallback_p3_attack "
"test_a74_replay_executed_card_rejected / test_fallback_execute "
"test_forged_plan_fingerprint_refused",
"Q-30": "MATCH/DRIFT/MISSING in this file;单测等价 highrisk "
"test_s6_reconcile_report_and_evidence_frozen",
"Q-35": "ops diagnostics redaction in this file;等价 highrisk "
"test_s7_ops_readonly_diagnostics",
"Q-36": "non-ops invisible in this file;等价 highrisk test_s7_non_ops_role_invisible",
"Q-37": "policy update in this file;等价 highrisk test_policy_update_writes_whitelist_via_card",
"Q-38": "config apply dual approval in this file;等价 highrisk "
"test_s7_config_apply_p3_dual_approval_flow",
"Q-39": "whitelist missing/corrupt in this file;等价 highrisk H-1/H-2",
"Q-40": "runtime unavailable in this file;等价 lane "
"test_runtime_unavailable_falls_back_to_canned_reply",
}