crisis/tests/test_proof.py

139 lines
4.9 KiB
Python
Raw Normal View History

Add crisis_agents — Crisis as a coordination layer for AI agent teams A new sibling Python package, `crisis_agents`, that lifts the Crisis protocol from "consensus between machines" to "consensus between AI agents". Threat model: a team of sub-agents normally talks freely with its orchestrator (the "mothership"); when the team's boundary opens and an external agent of unknown trust joins, the mothership activates the Crisis layer so byzantine equivocation is detectable. Two-phase orchestration model: Phase 1 — closed team, no Crisis: agents emit claims directly, the mothership collects them flat. Phase 2 — boundary opens: every subsequent claim is wrapped into a Crisis Message with the agent's stable process_id and a PoW nonce, delivered into per-agent LamportGraphs, and after each turn the mothership scans for mutations via LamportGraph.find_mutations. Phase 3 — proof: when an alarm fires, the mothership emits a replayable JSON proof-of-malfeasance document with the contradictory witnesses, their delivery sets, and DAG cross-references showing which honest agents saw what. Modules: - claim.py Claim dataclass + JSON round-trip - boundary.py membership tracker + open() trigger - agent.py CrisisAgent abstract + MockAgent + MockByzantineAgent (the latter equivocates by emitting two variants to disjoint peer subsets at the same logical turn) - mothership.py orchestrator driving both phases, building Crisis Messages from Claims, per-agent LamportGraphs, log - alarm.py scan_for_mutations: same-agent same-turn distinct digests with non-identical delivery sets, verified spacelike via LamportGraph.are_spacelike on the honest-agent graphs - proof.py build_proof + ProofDocument + JSON serializer + verify_proof_self_consistent - cli.py `crisis-agents demo` + `crisis-agents verify` - scenarios/ fact_check: reference doc + 6 statements + scripted honest/byzantine agents producing a deterministic equivocation on statement s03 Tests: 50 new tests across test_claim, test_boundary, test_mothership, test_alarm, test_proof, test_demo_fact_check. End-to-end test runs the fact_check scenario, asserts exactly one alarm raised, proof is built, re-serialized JSON passes self-consistency. Full suite (existing crisis + new crisis_agents) green in 0.77s — 145 tests. Out of scope (deliberately): visualization (separate CrisisViz upgrade later), real TCP gossip (agents talk via in-process function calls in the mothership), false-claim detection without equivocation (an agent that consistently lies but never equivocates is out-voted, not "caught"; catching it would require a ground-truth oracle). Reuse from existing crisis package: Message, Vertex, LamportGraph, LamportGraph.find_mutations, ProofOfWorkWeight, digest. The new code is a thin adapter layer; the protocol substrate did the heavy lifting. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-05-14 14:38:11 +00:00
"""Tests for proof generation and self-consistent verification."""
import json
from crisis_agents.agent import MockAgent, MockByzantineAgent
from crisis_agents.alarm import scan_for_mutations
from crisis_agents.claim import Claim
from crisis_agents.mothership import Mothership
from crisis_agents.proof import (
ProofDocument,
build_proof,
verify_proof_self_consistent,
)
def _claim(sid: str, verdict: str = "true", evidence: str = "ok") -> Claim:
return Claim(statement_id=sid, verdict=verdict, confidence=0.9, # type: ignore[arg-type]
evidence=evidence, timestamp_logical=0)
def _equivocating_run() -> tuple[Mothership, list]:
m = Mothership()
m.add_agent(MockAgent("a", [[]]))
m.add_agent(MockAgent("b", [[]]))
m.add_agent(MockAgent("c", [[]]))
m.open_boundary(MockByzantineAgent(
"d",
scripted_pairs=[(
_claim("s03", verdict="true", evidence="to_a"),
_claim("s03", verdict="false", evidence="to_b"),
)],
split_a={"a", "c"},
split_b={"b"},
))
m.run_crisis_phase(num_turns=1)
alarms = scan_for_mutations(m)
return m, alarms
class TestBuildProof:
def test_produces_well_formed_proof(self):
m, alarms = _equivocating_run()
assert len(alarms) == 1
proof = build_proof(m, alarms[0])
assert proof.accused_agent == "d"
assert proof.statement_id == "s03"
assert proof.turn == 0
assert len(proof.witnesses) == 2
assert proof.spacelike_verified is True
def test_dag_witnesses_reference_real_graphs(self):
m, alarms = _equivocating_run()
proof = build_proof(m, alarms[0])
assert len(proof.dag_witnesses) == 2
# Each dag_witness should name only honest agents (not "d")
for dw in proof.dag_witnesses:
assert "d" not in dw.observed_by
# Each variant should have been observed by at least one honest agent
# (the variant's delivered-to subset)
for dw in proof.dag_witnesses:
assert len(dw.observed_by) >= 1
class TestJsonRoundtrip:
def test_to_json_is_valid(self):
m, alarms = _equivocating_run()
proof = build_proof(m, alarms[0])
text = proof.to_json()
parsed = json.loads(text)
assert parsed["accused_agent"] == "d"
assert parsed["statement_id"] == "s03"
def test_from_json_inverts_to_json(self):
m, alarms = _equivocating_run()
original = build_proof(m, alarms[0])
roundtrip = ProofDocument.from_json(original.to_json())
assert roundtrip.accused_agent == original.accused_agent
assert roundtrip.statement_id == original.statement_id
assert roundtrip.turn == original.turn
assert roundtrip.spacelike_verified == original.spacelike_verified
assert len(roundtrip.witnesses) == len(original.witnesses)
class TestSelfConsistentVerification:
def test_valid_proof_passes(self):
m, alarms = _equivocating_run()
proof = build_proof(m, alarms[0])
result = verify_proof_self_consistent(proof)
assert result.ok, result.reason
def test_tampered_witness_digest_fails(self):
"""If someone alters a witness digest after-the-fact to make it look
like a duplicate, self-consistency check catches the lack of distinct
digests."""
m, alarms = _equivocating_run()
proof = build_proof(m, alarms[0])
# Tamper: make both digests identical
from dataclasses import replace
from crisis_agents.alarm import MutationWitness
w0 = proof.witnesses[0]
w1 = proof.witnesses[1]
tampered = ProofDocument(
schema_version=proof.schema_version,
accused_agent=proof.accused_agent,
accused_process_id_hex=proof.accused_process_id_hex,
statement_id=proof.statement_id,
turn=proof.turn,
witnesses=(w0, replace(w1, message_digest_hex=w0.message_digest_hex)),
dag_witnesses=proof.dag_witnesses,
spacelike_verified=proof.spacelike_verified,
proof_summary=proof.proof_summary,
)
result = verify_proof_self_consistent(tampered)
assert not result.ok
assert "duplicate" in result.reason.lower()
def test_mismatched_statement_id_fails(self):
m, alarms = _equivocating_run()
proof = build_proof(m, alarms[0])
from dataclasses import replace
bad = ProofDocument(
schema_version=proof.schema_version,
accused_agent=proof.accused_agent,
accused_process_id_hex=proof.accused_process_id_hex,
statement_id="DIFFERENT", # mismatch
turn=proof.turn,
witnesses=proof.witnesses,
dag_witnesses=proof.dag_witnesses,
spacelike_verified=proof.spacelike_verified,
proof_summary=proof.proof_summary,
)
result = verify_proof_self_consistent(bad)
assert not result.ok