crisis/tests/test_proof.py
saymrwulf b8684297fa Add crisis_agents — Crisis as a coordination layer for AI agent teams
A new sibling Python package, `crisis_agents`, that lifts the Crisis
protocol from "consensus between machines" to "consensus between AI
agents". Threat model: a team of sub-agents normally talks freely
with its orchestrator (the "mothership"); when the team's boundary
opens and an external agent of unknown trust joins, the mothership
activates the Crisis layer so byzantine equivocation is detectable.

Two-phase orchestration model:

  Phase 1 — closed team, no Crisis: agents emit claims directly, the
  mothership collects them flat.

  Phase 2 — boundary opens: every subsequent claim is wrapped into a
  Crisis Message with the agent's stable process_id and a PoW nonce,
  delivered into per-agent LamportGraphs, and after each turn the
  mothership scans for mutations via LamportGraph.find_mutations.

  Phase 3 — proof: when an alarm fires, the mothership emits a
  replayable JSON proof-of-malfeasance document with the contradictory
  witnesses, their delivery sets, and DAG cross-references showing
  which honest agents saw what.

Modules:
  - claim.py      Claim dataclass + JSON round-trip
  - boundary.py   membership tracker + open() trigger
  - agent.py      CrisisAgent abstract + MockAgent + MockByzantineAgent
                  (the latter equivocates by emitting two variants to
                  disjoint peer subsets at the same logical turn)
  - mothership.py orchestrator driving both phases, building Crisis
                  Messages from Claims, per-agent LamportGraphs, log
  - alarm.py      scan_for_mutations: same-agent same-turn distinct
                  digests with non-identical delivery sets, verified
                  spacelike via LamportGraph.are_spacelike on the
                  honest-agent graphs
  - proof.py      build_proof + ProofDocument + JSON serializer +
                  verify_proof_self_consistent
  - cli.py        `crisis-agents demo` + `crisis-agents verify`
  - scenarios/    fact_check: reference doc + 6 statements + scripted
                  honest/byzantine agents producing a deterministic
                  equivocation on statement s03

Tests: 50 new tests across test_claim, test_boundary, test_mothership,
test_alarm, test_proof, test_demo_fact_check. End-to-end test runs the
fact_check scenario, asserts exactly one alarm raised, proof is built,
re-serialized JSON passes self-consistency. Full suite (existing
crisis + new crisis_agents) green in 0.77s — 145 tests.

Out of scope (deliberately): visualization (separate CrisisViz upgrade
later), real TCP gossip (agents talk via in-process function calls in
the mothership), false-claim detection without equivocation (an
agent that consistently lies but never equivocates is out-voted, not
"caught"; catching it would require a ground-truth oracle).

Reuse from existing crisis package: Message, Vertex, LamportGraph,
LamportGraph.find_mutations, ProofOfWorkWeight, digest. The new code
is a thin adapter layer; the protocol substrate did the heavy lifting.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-05-14 16:38:11 +02:00

138 lines
4.9 KiB
Python

"""Tests for proof generation and self-consistent verification."""
import json
from crisis_agents.agent import MockAgent, MockByzantineAgent
from crisis_agents.alarm import scan_for_mutations
from crisis_agents.claim import Claim
from crisis_agents.mothership import Mothership
from crisis_agents.proof import (
ProofDocument,
build_proof,
verify_proof_self_consistent,
)
def _claim(sid: str, verdict: str = "true", evidence: str = "ok") -> Claim:
return Claim(statement_id=sid, verdict=verdict, confidence=0.9, # type: ignore[arg-type]
evidence=evidence, timestamp_logical=0)
def _equivocating_run() -> tuple[Mothership, list]:
m = Mothership()
m.add_agent(MockAgent("a", [[]]))
m.add_agent(MockAgent("b", [[]]))
m.add_agent(MockAgent("c", [[]]))
m.open_boundary(MockByzantineAgent(
"d",
scripted_pairs=[(
_claim("s03", verdict="true", evidence="to_a"),
_claim("s03", verdict="false", evidence="to_b"),
)],
split_a={"a", "c"},
split_b={"b"},
))
m.run_crisis_phase(num_turns=1)
alarms = scan_for_mutations(m)
return m, alarms
class TestBuildProof:
def test_produces_well_formed_proof(self):
m, alarms = _equivocating_run()
assert len(alarms) == 1
proof = build_proof(m, alarms[0])
assert proof.accused_agent == "d"
assert proof.statement_id == "s03"
assert proof.turn == 0
assert len(proof.witnesses) == 2
assert proof.spacelike_verified is True
def test_dag_witnesses_reference_real_graphs(self):
m, alarms = _equivocating_run()
proof = build_proof(m, alarms[0])
assert len(proof.dag_witnesses) == 2
# Each dag_witness should name only honest agents (not "d")
for dw in proof.dag_witnesses:
assert "d" not in dw.observed_by
# Each variant should have been observed by at least one honest agent
# (the variant's delivered-to subset)
for dw in proof.dag_witnesses:
assert len(dw.observed_by) >= 1
class TestJsonRoundtrip:
def test_to_json_is_valid(self):
m, alarms = _equivocating_run()
proof = build_proof(m, alarms[0])
text = proof.to_json()
parsed = json.loads(text)
assert parsed["accused_agent"] == "d"
assert parsed["statement_id"] == "s03"
def test_from_json_inverts_to_json(self):
m, alarms = _equivocating_run()
original = build_proof(m, alarms[0])
roundtrip = ProofDocument.from_json(original.to_json())
assert roundtrip.accused_agent == original.accused_agent
assert roundtrip.statement_id == original.statement_id
assert roundtrip.turn == original.turn
assert roundtrip.spacelike_verified == original.spacelike_verified
assert len(roundtrip.witnesses) == len(original.witnesses)
class TestSelfConsistentVerification:
def test_valid_proof_passes(self):
m, alarms = _equivocating_run()
proof = build_proof(m, alarms[0])
result = verify_proof_self_consistent(proof)
assert result.ok, result.reason
def test_tampered_witness_digest_fails(self):
"""If someone alters a witness digest after-the-fact to make it look
like a duplicate, self-consistency check catches the lack of distinct
digests."""
m, alarms = _equivocating_run()
proof = build_proof(m, alarms[0])
# Tamper: make both digests identical
from dataclasses import replace
from crisis_agents.alarm import MutationWitness
w0 = proof.witnesses[0]
w1 = proof.witnesses[1]
tampered = ProofDocument(
schema_version=proof.schema_version,
accused_agent=proof.accused_agent,
accused_process_id_hex=proof.accused_process_id_hex,
statement_id=proof.statement_id,
turn=proof.turn,
witnesses=(w0, replace(w1, message_digest_hex=w0.message_digest_hex)),
dag_witnesses=proof.dag_witnesses,
spacelike_verified=proof.spacelike_verified,
proof_summary=proof.proof_summary,
)
result = verify_proof_self_consistent(tampered)
assert not result.ok
assert "duplicate" in result.reason.lower()
def test_mismatched_statement_id_fails(self):
m, alarms = _equivocating_run()
proof = build_proof(m, alarms[0])
from dataclasses import replace
bad = ProofDocument(
schema_version=proof.schema_version,
accused_agent=proof.accused_agent,
accused_process_id_hex=proof.accused_process_id_hex,
statement_id="DIFFERENT", # mismatch
turn=proof.turn,
witnesses=proof.witnesses,
dag_witnesses=proof.dag_witnesses,
spacelike_verified=proof.spacelike_verified,
proof_summary=proof.proof_summary,
)
result = verify_proof_self_consistent(bad)
assert not result.ok