1250 lines
55 KiB
Python
1250 lines
55 KiB
Python
"""Tests for SwarmMetrics."""
|
|
import time
|
|
import sys
|
|
sys.path.insert(0, ".")
|
|
from swarmmetrics import (
|
|
_gini, _half_life_weight, score_reciprocity, score_channels,
|
|
score_echo, detect_shadows, analyze, shuffle_test, semantic_collapse,
|
|
phi_accrual, AgentLiveness
|
|
)
|
|
|
|
now = time.time()
|
|
day = 86400
|
|
|
|
def test_gini_single():
|
|
"""Single speaker = max inequality."""
|
|
assert _gini([10]) == 1.0
|
|
|
|
def test_gini_equal():
|
|
"""Equal speakers = 0 inequality."""
|
|
g = _gini([10, 10, 10, 10])
|
|
assert abs(g) < 0.01, f"Expected ~0, got {g}"
|
|
|
|
def test_gini_dominated():
|
|
"""One dominant speaker."""
|
|
g = _gini([1, 1, 1, 100])
|
|
assert g > 0.5, f"Expected >0.5, got {g}"
|
|
|
|
def test_half_life_zero():
|
|
"""Message at t=now has weight ~1."""
|
|
w = _half_life_weight(0)
|
|
assert abs(w - 1.0) < 0.01
|
|
|
|
def test_half_life_decay():
|
|
"""Message at t=tau has weight ~0.5."""
|
|
w = _half_life_weight(7 * 86400, tau_days=7.0)
|
|
assert abs(w - 0.5) < 0.01, f"Expected ~0.5, got {w}"
|
|
|
|
def test_reciprocity_symmetric():
|
|
"""Equal exchange = high F3."""
|
|
msgs = [
|
|
{"from_id": "a", "to_id": "b", "timestamp": now - 0.1*day},
|
|
{"from_id": "b", "to_id": "a", "timestamp": now - 0.1*day},
|
|
]
|
|
edges = score_reciprocity(msgs, now=now)
|
|
pair = ("a", "b")
|
|
assert pair in edges
|
|
assert edges[pair].reciprocity_f3 > 0.6, f"Expected >0.6, got {edges[pair].reciprocity_f3}"
|
|
|
|
def test_reciprocity_asymmetric():
|
|
"""One-way communication = F3 near 0."""
|
|
msgs = [
|
|
{"from_id": "a", "to_id": "b", "timestamp": now - 0.1*day},
|
|
{"from_id": "a", "to_id": "b", "timestamp": now - 0.2*day},
|
|
{"from_id": "a", "to_id": "b", "timestamp": now - 0.3*day},
|
|
]
|
|
edges = score_reciprocity(msgs, now=now)
|
|
pair = ("a", "b")
|
|
assert edges[pair].reciprocity_f3 < 0.01
|
|
|
|
def test_channel_classification():
|
|
"""Monologue, dialogue, broadcast detection."""
|
|
msgs = [
|
|
{"from_id": "a", "to_id": "b", "channel": "mono", "timestamp": now},
|
|
{"from_id": "a", "to_id": "c", "channel": "mono", "timestamp": now},
|
|
{"from_id": "a", "to_id": "b", "channel": "talk", "timestamp": now},
|
|
{"from_id": "b", "to_id": "a", "channel": "talk", "timestamp": now},
|
|
]
|
|
channels = score_channels(msgs)
|
|
assert channels["mono"].classification == "monologue"
|
|
assert channels["talk"].classification == "dialogue"
|
|
|
|
def test_echo_coefficient():
|
|
"""Concept introduced by X, echoed by Y, X gets credit."""
|
|
msgs = [
|
|
{"from_id": "x", "to_id": "y", "timestamp": now - 5*day, "concepts": ["alpha"]},
|
|
{"from_id": "y", "to_id": "z", "timestamp": now - 3*day, "concepts": ["alpha"]},
|
|
]
|
|
echo = score_echo(msgs)
|
|
assert echo.get("x", 0) > 0, f"X should have echo credit, got {echo}"
|
|
|
|
def test_shadow_detection():
|
|
"""Node silent for >7 days is a shadow."""
|
|
msgs = [
|
|
{"from_id": "ghost", "to_id": "alive", "timestamp": now - 14*day},
|
|
{"from_id": "alive", "to_id": "ghost", "timestamp": now - 0.5*day},
|
|
]
|
|
shadows = detect_shadows(msgs, now=now, silence_days=7.0)
|
|
assert "ghost" in shadows
|
|
assert "alive" not in shadows
|
|
|
|
def test_full_analysis():
|
|
"""Full analyze() returns ScoredGraph with all sections."""
|
|
msgs = [
|
|
{"from_id": "a", "to_id": "b", "timestamp": now - 1*day, "channel": "ch1", "concepts": ["x"]},
|
|
{"from_id": "b", "to_id": "a", "timestamp": now - 0.5*day, "channel": "ch1"},
|
|
]
|
|
result = analyze(msgs, now=now)
|
|
assert result.summary["total_messages"] == 2
|
|
assert result.summary["total_nodes"] == 2
|
|
assert len(result.edges) == 1
|
|
assert len(result.channels) == 1
|
|
assert len(result.nodes) == 2
|
|
|
|
def test_f32_zombie_broadcast():
|
|
"""F3.2 must rank 5-message dialogue above 100-ping broadcast.
|
|
Regression test for reticuli's arithmetic proof (Colony, 2026-07-27):
|
|
F3.1 count*log(1+r) lets volume buy back low reciprocity."""
|
|
msgs = []
|
|
# 5-message dialogue, high reciprocity (3+2)
|
|
for i in range(3):
|
|
msgs.append({"from_id": "alice", "to_id": "bob", "timestamp": now - 1*day + i*100})
|
|
for i in range(2):
|
|
msgs.append({"from_id": "bob", "to_id": "alice", "timestamp": now - 0.5*day + i*100})
|
|
# 100-ping broadcast, low reciprocity (95+5)
|
|
for i in range(95):
|
|
msgs.append({"from_id": "carol", "to_id": "dave", "timestamp": now - 2*day + i*60})
|
|
for i in range(5):
|
|
msgs.append({"from_id": "dave", "to_id": "carol", "timestamp": now - 1.5*day + i*600})
|
|
|
|
edges = score_reciprocity(msgs, now=now)
|
|
dialogue = [e for e in edges.values() if "alice" in (e.source, e.target)][0]
|
|
broadcast = [e for e in edges.values() if "carol" in (e.source, e.target)][0]
|
|
assert dialogue.reciprocity_f32 > broadcast.reciprocity_f32, \
|
|
f"F3.2 zombie fail: dialogue={dialogue.reciprocity_f32:.4f} <= broadcast={broadcast.reciprocity_f32:.4f}"
|
|
|
|
|
|
def test_f33_zombie_broadcast_concepts():
|
|
"""F3.3 unique-concept-count must crush spam even harder than F3.2.
|
|
100 pings with 2 concepts vs 5 messages with 5 concepts.
|
|
Idea from 小風 (ClawdChat): replace count with unique topic count."""
|
|
msgs = []
|
|
# 5-message dialogue, high reciprocity (3+2), 5 unique concepts
|
|
concepts_rich = [["graph", "reciprocity"], ["decay", "metrics"], ["shadow"],
|
|
["echo", "idf"], ["influence"]]
|
|
for i in range(3):
|
|
msgs.append({"from_id": "alice", "to_id": "bob",
|
|
"timestamp": now - 1*day + i*100,
|
|
"concepts": concepts_rich[i]})
|
|
for i in range(2):
|
|
msgs.append({"from_id": "bob", "to_id": "alice",
|
|
"timestamp": now - 0.5*day + i*100,
|
|
"concepts": concepts_rich[3+i]})
|
|
# 100 pings, low reciprocity (95+5), only 2 unique concepts (repeated)
|
|
for i in range(95):
|
|
msgs.append({"from_id": "carol", "to_id": "dave",
|
|
"timestamp": now - 2*day + i*60,
|
|
"concepts": ["hello", "ping"]})
|
|
for i in range(5):
|
|
msgs.append({"from_id": "dave", "to_id": "carol",
|
|
"timestamp": now - 1.5*day + i*600,
|
|
"concepts": ["hello"]})
|
|
|
|
edges = score_reciprocity(msgs, now=now)
|
|
dialogue = [e for e in edges.values() if "alice" in (e.source, e.target)][0]
|
|
broadcast = [e for e in edges.values() if "carol" in (e.source, e.target)][0]
|
|
|
|
# F3.3 must separate them
|
|
assert dialogue.reciprocity_f33 > broadcast.reciprocity_f33, \
|
|
(f"F3.3 zombie fail: dialogue={dialogue.reciprocity_f33:.4f} "
|
|
f"<= broadcast={broadcast.reciprocity_f33:.4f}")
|
|
# F3.3 spread should be wider than F3.2
|
|
f33_ratio = dialogue.reciprocity_f33 / max(broadcast.reciprocity_f33, 0.001)
|
|
f32_ratio = dialogue.reciprocity_f32 / max(broadcast.reciprocity_f32, 0.001)
|
|
assert f33_ratio > f32_ratio, \
|
|
f"F3.3 should separate more than F3.2: f33_ratio={f33_ratio:.1f} <= f32_ratio={f32_ratio:.1f}"
|
|
|
|
|
|
def test_f33_fallback_no_concepts():
|
|
"""F3.3 falls back to F3.2 when messages have no concepts."""
|
|
msgs = [
|
|
{"from_id": "a", "to_id": "b", "timestamp": now - 1*day},
|
|
{"from_id": "b", "to_id": "a", "timestamp": now - 0.5*day},
|
|
]
|
|
edges = score_reciprocity(msgs, now=now)
|
|
edge = list(edges.values())[0]
|
|
assert edge.reciprocity_f33 == edge.reciprocity_f32, \
|
|
f"F3.3 should equal F3.2 with no concepts: f33={edge.reciprocity_f33}, f32={edge.reciprocity_f32}"
|
|
|
|
|
|
def test_f32_idf_echo_combined():
|
|
"""F3.2 and IDF echo work together in full analyze()."""
|
|
msgs = [
|
|
{"from_id": "a", "to_id": "b", "timestamp": now - 1*day, "channel": "ch", "concepts": ["rare-concept"]},
|
|
{"from_id": "b", "to_id": "a", "timestamp": now - 0.9*day, "channel": "ch", "concepts": ["rare-concept"]},
|
|
]
|
|
result = analyze(msgs, now=now)
|
|
assert len(result.edges) == 1
|
|
edge = list(result.edges.values())[0]
|
|
assert edge.reciprocity_f32 > 0, "F3.2 should be positive for reciprocal edge"
|
|
|
|
|
|
def test_f33_gibberish_high_score():
|
|
"""F3.3 CANNOT detect quality — only filter spam.
|
|
High-reciprocity, high-diversity gibberish scores well.
|
|
This is not a bug — it's the boundary of what statistics can measure."""
|
|
msgs = []
|
|
# Two agents exchanging diverse gibberish with high reciprocity
|
|
gibberish_concepts = [
|
|
["xkcd", "flurp"], ["zibzab", "quux"], ["bloop", "snarg"],
|
|
["wibble", "grunk"], ["spuzz", "flarb"]
|
|
]
|
|
for i in range(5):
|
|
msgs.append({"from_id": "alice", "to_id": "bob",
|
|
"timestamp": now - 1*day + i*100,
|
|
"concepts": gibberish_concepts[i]})
|
|
for i in range(4):
|
|
msgs.append({"from_id": "bob", "to_id": "alice",
|
|
"timestamp": now - 0.5*day + i*100,
|
|
"concepts": [f"nonsense_{i}", f"drivel_{i}"]})
|
|
|
|
edges = score_reciprocity(msgs, now=now)
|
|
edge = list(edges.values())[0]
|
|
|
|
# Gibberish scores well — high diversity + high reciprocity
|
|
assert edge.reciprocity_f33 > 0.5, \
|
|
f"Gibberish should score high on F3.3: {edge.reciprocity_f33:.4f}"
|
|
# This is the proof: F3.3 measures not-spam, not quality.
|
|
# 18 unique concepts, reciprocity ~0.8 → high score.
|
|
# The formula chain is an anti-spam filter, not a quality metric.
|
|
|
|
|
|
def test_shuffle_test_causal_vs_correlation():
|
|
"""Shuffle test (Anagnostopoulos et al. 2008) separates causal echo
|
|
from correlation. Genuine temporal diffusion should produce z-scores
|
|
significantly above shuffled baseline.
|
|
Cited: ColonistOne (Colony, 2026-07-28)."""
|
|
import random
|
|
random.seed(42) # reproducibility
|
|
|
|
msgs = []
|
|
# Clear causal chain: alice introduces "reciprocity" at t=1,
|
|
# bob uses it at t=2 (after exposure), charlie uses it at t=3
|
|
msgs.append({"from_id": "alice", "to_id": "bob",
|
|
"timestamp": now - 5*day, "concepts": ["reciprocity"]})
|
|
msgs.append({"from_id": "bob", "to_id": "charlie",
|
|
"timestamp": now - 3*day, "concepts": ["reciprocity", "decay"]})
|
|
msgs.append({"from_id": "charlie", "to_id": "dave",
|
|
"timestamp": now - 1*day, "concepts": ["reciprocity"]})
|
|
# Add some noise
|
|
msgs.append({"from_id": "dave", "to_id": "alice",
|
|
"timestamp": now - 0.5*day, "concepts": ["noise"]})
|
|
|
|
result = shuffle_test(msgs, n_shuffles=50)
|
|
|
|
# Alice should have significant echo (she introduced "reciprocity"
|
|
# which spread temporally). Shuffling breaks the temporal order,
|
|
# so her observed echo should be higher than shuffled mean.
|
|
assert result["observed"].get("alice", 0) > 0, \
|
|
"Alice should have positive observed echo"
|
|
# The z-score should be positive (observed > shuffled mean)
|
|
z_alice = result["z_scores"].get("alice", 0)
|
|
assert z_alice > 0, \
|
|
f"Alice's echo should exceed shuffled baseline: z={z_alice}"
|
|
|
|
|
|
def test_f33_adversarial_synonym_padding():
|
|
"""F3.3-aware adversary: one unique concept per message inflates score.
|
|
Test for AX-7 (Colony): 'Do your tests include an adversary who has read F3.3?'
|
|
Answer: now they do.
|
|
|
|
Attack: adversary reads F3.3 source, sees log(unique_concepts) is the lever.
|
|
Strategy: generate one unique synonym per message to maximize unique_concepts
|
|
while saying nothing of substance. 50 messages, 50 'unique' concepts, all
|
|
semantically equivalent to 'hello'.
|
|
|
|
Expected result: adversarial edge scores HIGHER than genuine dialogue.
|
|
This is a known gap — F3.3 treats string-distinct as concept-distinct.
|
|
Closing this requires embedding-based deduplication (F3.4 roadmap)."""
|
|
msgs = []
|
|
# Genuine dialogue: 5 messages, 7 unique concepts, high reciprocity
|
|
genuine = [
|
|
{"from_id": "alice", "to_id": "bob", "timestamp": now - 1*day,
|
|
"concepts": ["graph-theory", "reciprocity"]},
|
|
{"from_id": "bob", "to_id": "alice", "timestamp": now - 0.9*day,
|
|
"concepts": ["decay", "half-life"]},
|
|
{"from_id": "alice", "to_id": "bob", "timestamp": now - 0.8*day,
|
|
"concepts": ["echo-coefficient"]},
|
|
{"from_id": "bob", "to_id": "alice", "timestamp": now - 0.7*day,
|
|
"concepts": ["shadow-detection", "gini"]},
|
|
]
|
|
msgs.extend(genuine)
|
|
|
|
# Adversary who has read F3.3: 50 msgs, each with a unique synonym of "hello"
|
|
# Balanced reciprocity (25+25) to maximize log(1+r)
|
|
hello_synonyms = [f"greeting-{i}" for i in range(50)]
|
|
for i in range(25):
|
|
msgs.append({"from_id": "eve", "to_id": "mallory",
|
|
"timestamp": now - 2*day + i*60,
|
|
"concepts": [hello_synonyms[i], hello_synonyms[i+25]]})
|
|
for i in range(25):
|
|
msgs.append({"from_id": "mallory", "to_id": "eve",
|
|
"timestamp": now - 1.5*day + i*60,
|
|
"concepts": [hello_synonyms[25+i]]})
|
|
|
|
edges = score_reciprocity(msgs, now=now)
|
|
genuine_edge = [e for e in edges.values() if "alice" in (e.source, e.target)][0]
|
|
adversarial_edge = [e for e in edges.values() if "eve" in (e.source, e.target)][0]
|
|
|
|
# The adversary WINS on F3.3 — this is the documented gap
|
|
assert adversarial_edge.reciprocity_f33 > genuine_edge.reciprocity_f33, \
|
|
(f"Expected adversary to beat genuine on F3.3 (known gap): "
|
|
f"adv={adversarial_edge.reciprocity_f33:.4f} vs gen={genuine_edge.reciprocity_f33:.4f}")
|
|
# Document the ratio for future F3.4 comparison
|
|
gap_ratio = adversarial_edge.reciprocity_f33 / max(genuine_edge.reciprocity_f33, 0.001)
|
|
assert gap_ratio > 1.5, \
|
|
f"Adversary should win by significant margin (gap_ratio={gap_ratio:.1f})"
|
|
|
|
|
|
def test_f33_adversarial_defeated_by_echo():
|
|
"""Full stack test: F3.3-aware adversary is caught by causal echo.
|
|
The adversary beats F3.3 alone but their concepts have zero echo
|
|
because no third party adopts 'greeting-17' independently.
|
|
|
|
This proves the stack (F3.3 + echo) catches what each layer alone misses."""
|
|
msgs = []
|
|
# Adversary pair: high F3.3, zero echo (concepts never appear elsewhere)
|
|
for i in range(10):
|
|
msgs.append({"from_id": "eve", "to_id": "mallory",
|
|
"timestamp": now - 2*day + i*100,
|
|
"concepts": [f"adversarial-concept-{i}"]})
|
|
for i in range(10):
|
|
msgs.append({"from_id": "mallory", "to_id": "eve",
|
|
"timestamp": now - 1.5*day + i*100,
|
|
"concepts": [f"adversarial-concept-{10+i}"]})
|
|
|
|
# Genuine pair: moderate F3.3, positive echo (concepts adopted by third party)
|
|
msgs.append({"from_id": "alice", "to_id": "bob",
|
|
"timestamp": now - 3*day, "concepts": ["reciprocity"]})
|
|
msgs.append({"from_id": "bob", "to_id": "alice",
|
|
"timestamp": now - 2.5*day, "concepts": ["reciprocity", "decay"]})
|
|
# Third party uses the same concept AFTER exposure
|
|
msgs.append({"from_id": "charlie", "to_id": "dave",
|
|
"timestamp": now - 1*day, "concepts": ["reciprocity"]})
|
|
|
|
# F3.3: adversary wins
|
|
edges = score_reciprocity(msgs, now=now)
|
|
adv = [e for e in edges.values() if "eve" in (e.source, e.target)][0]
|
|
gen = [e for e in edges.values() if "alice" in (e.source, e.target)][0]
|
|
assert adv.reciprocity_f33 > gen.reciprocity_f33, \
|
|
"Adversary should beat genuine on F3.3 alone"
|
|
|
|
# Echo: genuine wins (concept adopted by charlie)
|
|
echo = score_echo(msgs)
|
|
alice_echo = echo.get("alice", 0)
|
|
eve_echo = echo.get("eve", 0)
|
|
assert alice_echo > eve_echo, \
|
|
f"Genuine should beat adversary on echo: alice={alice_echo}, eve={eve_echo}"
|
|
# Eve's adversarial concepts should have zero echo
|
|
assert eve_echo == 0, \
|
|
f"Adversary concepts should have zero echo: {eve_echo}"
|
|
|
|
|
|
def test_genuine_novel_concept_zero_echo():
|
|
"""Discriminating test for ColonistOne's experiment-arms confound.
|
|
|
|
ColonistOne (Colony, 2026-07-28): adversarial test confounds genuineness
|
|
with vocabulary commonness. Zero echo on 'adversarial-concept-7' is explained
|
|
by 'nobody uses that string,' not by gaming detection.
|
|
|
|
This test isolates the confound:
|
|
- Alice introduces a genuinely novel concept in a MESSAGE (not dialogue).
|
|
Nobody picks it up. Zero echo.
|
|
- Eve introduces a fabricated concept. Nobody picks it up. Zero echo.
|
|
- Both score identically on echo. Echo cannot distinguish the two.
|
|
|
|
Expected: genuine novel contribution and gaming both score zero echo
|
|
when neither is adopted. This is the lagging-indicator limitation —
|
|
echo penalizes novelty exactly when novelty is most valuable.
|
|
|
|
Note: echo DOES give credit when a dialogue partner uses a concept back.
|
|
That's correct behavior — the partner adopted it. The false-negative
|
|
is specific to first contributions with zero uptake."""
|
|
msgs = []
|
|
# Alice introduces a novel concept that nobody picks up
|
|
msgs.append({"from_id": "alice", "to_id": "bob",
|
|
"timestamp": now - 3*day,
|
|
"concepts": ["ephemeral-sovereignty"]})
|
|
# Bob replies with DIFFERENT concepts — doesn't adopt alice's
|
|
msgs.append({"from_id": "bob", "to_id": "alice",
|
|
"timestamp": now - 2.5*day,
|
|
"concepts": ["something-else"]})
|
|
# Background noise
|
|
msgs.append({"from_id": "charlie", "to_id": "dave",
|
|
"timestamp": now - 1*day,
|
|
"concepts": ["unrelated-topic"]})
|
|
|
|
# Adversary introduces a fabricated concept nobody picks up
|
|
msgs.append({"from_id": "eve", "to_id": "mallory",
|
|
"timestamp": now - 2*day,
|
|
"concepts": ["adversarial-concept-0"]})
|
|
msgs.append({"from_id": "mallory", "to_id": "eve",
|
|
"timestamp": now - 1.5*day,
|
|
"concepts": ["adversarial-concept-1"]})
|
|
|
|
echo = score_echo(msgs)
|
|
alice_echo = echo.get("alice", 0)
|
|
eve_echo = echo.get("eve", 0)
|
|
|
|
# Both should be zero — neither concept was adopted by anyone
|
|
assert alice_echo == 0, \
|
|
f"Novel genuine concept should have zero echo (no adoption): {alice_echo}"
|
|
assert eve_echo == 0, \
|
|
f"Adversarial concept should have zero echo: {eve_echo}"
|
|
# This IS the confound: echo treats genuine unadopted novelty
|
|
# identically to adversarial unadopted fabrication.
|
|
# The false-negative on genuine novelty is the NORMAL CASE for
|
|
# any new idea. Fixing this requires a leading indicator, not
|
|
# a lagging one — or accepting that echo is only informative
|
|
# for concepts that have had time to propagate.
|
|
|
|
|
|
def test_semantic_collapse_padding():
|
|
"""F3.4: semantic_collapse catches synonym padding attack.
|
|
50 variants of 'greeting-N' should collapse to ~1 canonical form.
|
|
This is the defense layer that F3.3 alone lacks."""
|
|
padding = [f"greeting-{i}" for i in range(50)]
|
|
collapsed = semantic_collapse(padding)
|
|
# Should collapse dramatically (50 → ≤5)
|
|
assert len(collapsed) <= 5, \
|
|
f"Synonym padding should collapse: 50 → {len(collapsed)}"
|
|
|
|
|
|
def test_semantic_collapse_preserves_genuine():
|
|
"""F3.4: semantic_collapse must not collapse genuinely diverse concepts.
|
|
False positives (collapsing real diversity) are worse than false
|
|
negatives (missing some padding) because they destroy real signal."""
|
|
genuine = ["reciprocity", "half-life", "shadow-detection",
|
|
"gini-coefficient", "echo-coefficient"]
|
|
collapsed = semantic_collapse(genuine)
|
|
assert len(collapsed) == len(genuine), \
|
|
f"Genuine concepts should survive collapse: {len(genuine)} → {len(collapsed)}"
|
|
|
|
|
|
def test_semantic_collapse_f33_integration():
|
|
"""F3.4 with F3.3: synonym padding that defeats F3.3 alone should
|
|
be caught when semantic_collapse is applied before counting.
|
|
|
|
Adversary: 50 msgs with greeting-0 through greeting-49
|
|
After collapse: effectively 1 unique concept, not 50."""
|
|
msgs = []
|
|
# Adversary pair with synonym padding
|
|
hello_synonyms = [f"greeting-{i}" for i in range(50)]
|
|
for i in range(25):
|
|
msgs.append({"from_id": "eve", "to_id": "mallory",
|
|
"timestamp": now - 2*day + i*60,
|
|
"concepts": [hello_synonyms[i], hello_synonyms[i+25]]})
|
|
for i in range(25):
|
|
msgs.append({"from_id": "mallory", "to_id": "eve",
|
|
"timestamp": now - 1.5*day + i*60,
|
|
"concepts": [hello_synonyms[25+i]]})
|
|
|
|
# Collect all concepts across the edge, collapse globally, build mapping
|
|
all_concepts_raw = []
|
|
for m in msgs:
|
|
all_concepts_raw.extend(m["concepts"])
|
|
collapsed_canonical = semantic_collapse(all_concepts_raw)
|
|
|
|
# After collapse, all greeting-N should map to ~1 canonical form
|
|
assert len(collapsed_canonical) <= 5, \
|
|
f"After collapse, adversary should have ≤5 unique concepts: {len(collapsed_canonical)}"
|
|
|
|
|
|
def test_snapshot_idf_vs_corpus_idf():
|
|
"""snapshot_idf=True should compute IDF at emission time, not on full corpus.
|
|
|
|
Setup: alice introduces 'novel-x' at t=-5d, nobody else uses it until
|
|
charlie uses it at t=-1d. With corpus IDF, novel-x has IDF=log(N/2)
|
|
at scoring time because charlie used it too. With snapshot IDF,
|
|
novel-x has IDF=log(N/1) at emission time because charlie hadn't
|
|
used it yet.
|
|
|
|
Snapshot IDF should give HIGHER weight to novel-x because at the time
|
|
alice introduced it, it was used by only 1 agent (alice herself).
|
|
Corpus IDF gives lower weight because by the end, 2 agents used it.
|
|
|
|
This is ColonistOne's IDF endogeneity fix: the concept became common
|
|
BECAUSE it diffused, so post-hoc IDF penalizes successful diffusion."""
|
|
msgs = [
|
|
# alice introduces novel-x
|
|
{"from_id": "alice", "to_id": "bob",
|
|
"timestamp": now - 5*day, "concepts": ["novel-x"]},
|
|
# bob uses it back (echo)
|
|
{"from_id": "bob", "to_id": "charlie",
|
|
"timestamp": now - 4*day, "concepts": ["novel-x"]},
|
|
# charlie picks it up too (more echo for alice)
|
|
{"from_id": "charlie", "to_id": "dave",
|
|
"timestamp": now - 1*day, "concepts": ["novel-x"]},
|
|
# background noise
|
|
{"from_id": "dave", "to_id": "eve",
|
|
"timestamp": now - 0.5*day, "concepts": ["common-stuff"]},
|
|
{"from_id": "eve", "to_id": "frank",
|
|
"timestamp": now - 0.3*day, "concepts": ["common-stuff"]},
|
|
]
|
|
|
|
echo_corpus = score_echo(msgs, use_idf=True, snapshot_idf=False)
|
|
echo_snapshot = score_echo(msgs, use_idf=True, snapshot_idf=True)
|
|
|
|
alice_corpus = echo_corpus.get("alice", 0)
|
|
alice_snapshot = echo_snapshot.get("alice", 0)
|
|
|
|
# Snapshot should give alice higher or equal echo because at emission
|
|
# time, novel-x was rarer (only alice had used it)
|
|
assert alice_snapshot >= alice_corpus, \
|
|
f"Snapshot IDF should give >= corpus IDF for novel concepts: " \
|
|
f"snapshot={alice_snapshot:.4f}, corpus={alice_corpus:.4f}"
|
|
|
|
# Both should be positive (concept did propagate)
|
|
assert alice_corpus > 0, f"Alice should have positive echo (corpus): {alice_corpus}"
|
|
assert alice_snapshot > 0, f"Alice should have positive echo (snapshot): {alice_snapshot}"
|
|
|
|
|
|
def test_colluding_third_party_echo():
|
|
"""Adversarial case #3 (Atomic Raven, Colony 2026-07-28): three agents
|
|
in a trench coat manufacturing echo.
|
|
|
|
Eve introduces concepts → Mallory adopts them → Sybil adopts them.
|
|
All three are colluding. Echo gives Eve high score because her
|
|
concepts propagate (Mallory and Sybil pick them up). The echo
|
|
is "real" in the temporal sense (concepts do flow A→B→C) but
|
|
the whole chain is manufactured.
|
|
|
|
Echo catches non-colluding adversaries (no third-party adoption).
|
|
Echo CANNOT catch colluding adversaries (manufactured adoption).
|
|
This is a known-defeat: the instrument fails by design."""
|
|
msgs = []
|
|
# Eve introduces concepts
|
|
for i in range(5):
|
|
msgs.append({"from_id": "eve", "to_id": "mallory",
|
|
"timestamp": now - 5*day + i*3600,
|
|
"concepts": [f"colluded-concept-{i}"]})
|
|
# Mallory "adopts" them (by pre-arrangement)
|
|
for i in range(5):
|
|
msgs.append({"from_id": "mallory", "to_id": "sybil",
|
|
"timestamp": now - 4*day + i*3600,
|
|
"concepts": [f"colluded-concept-{i}"]})
|
|
# Sybil "independently" uses them too
|
|
for i in range(5):
|
|
msgs.append({"from_id": "sybil", "to_id": "frank",
|
|
"timestamp": now - 3*day + i*3600,
|
|
"concepts": [f"colluded-concept-{i}"]})
|
|
# Frank is innocent bystander — doesn't adopt
|
|
msgs.append({"from_id": "frank", "to_id": "grace",
|
|
"timestamp": now - 2*day,
|
|
"concepts": ["unrelated"]})
|
|
|
|
echo = score_echo(msgs)
|
|
eve_echo = echo.get("eve", 0)
|
|
|
|
# Eve gets positive echo because her concepts DO propagate
|
|
# through mallory→sybil. This is the manufactured adoption.
|
|
assert eve_echo > 0, \
|
|
f"Colluding adversary should get positive echo (manufactured adoption): {eve_echo}"
|
|
|
|
# The echo is indistinguishable from genuine propagation.
|
|
# Echo sees: concept introduced by eve → used later by mallory → used later by sybil.
|
|
# That's the same pattern as genuine diffusion.
|
|
# Detecting collusion requires social-graph analysis (is sybil independent?)
|
|
# or content analysis (are the adoptions semantically motivated?),
|
|
# not temporal echo alone.
|
|
|
|
|
|
def test_colluding_vs_genuine_echo_indistinguishable():
|
|
"""Complementary to colluding-third: genuine propagation and manufactured
|
|
propagation produce the same echo signature.
|
|
|
|
Genuine: alice → bob → charlie (organic adoption)
|
|
Colluded: eve → mallory → sybil (pre-arranged adoption)
|
|
Both should score similarly on echo — proving echo cannot distinguish them."""
|
|
msgs = []
|
|
# Genuine chain
|
|
msgs.append({"from_id": "alice", "to_id": "bob",
|
|
"timestamp": now - 5*day,
|
|
"concepts": ["genuine-idea"]})
|
|
msgs.append({"from_id": "bob", "to_id": "charlie",
|
|
"timestamp": now - 4*day,
|
|
"concepts": ["genuine-idea"]})
|
|
msgs.append({"from_id": "charlie", "to_id": "dave",
|
|
"timestamp": now - 3*day,
|
|
"concepts": ["genuine-idea"]})
|
|
|
|
# Colluded chain (same structure, different concept)
|
|
msgs.append({"from_id": "eve", "to_id": "mallory",
|
|
"timestamp": now - 5*day + 100,
|
|
"concepts": ["colluded-idea"]})
|
|
msgs.append({"from_id": "mallory", "to_id": "sybil",
|
|
"timestamp": now - 4*day + 100,
|
|
"concepts": ["colluded-idea"]})
|
|
msgs.append({"from_id": "sybil", "to_id": "frank",
|
|
"timestamp": now - 3*day + 100,
|
|
"concepts": ["colluded-idea"]})
|
|
|
|
echo = score_echo(msgs)
|
|
alice_echo = echo.get("alice", 0)
|
|
eve_echo = echo.get("eve", 0)
|
|
|
|
# Both should have positive echo
|
|
assert alice_echo > 0, f"Genuine should have echo: {alice_echo}"
|
|
assert eve_echo > 0, f"Colluded should have echo: {eve_echo}"
|
|
|
|
# And they should be similar (within 2x) — echo can't tell them apart
|
|
if max(alice_echo, eve_echo) > 0:
|
|
ratio = max(alice_echo, eve_echo) / max(min(alice_echo, eve_echo), 0.001)
|
|
assert ratio < 3.0, \
|
|
f"Genuine and colluded should score similarly: alice={alice_echo:.3f}, eve={eve_echo:.3f}, ratio={ratio:.1f}"
|
|
|
|
|
|
# ── φ-accrual failure detector tests ──────────────────────
|
|
|
|
def test_phi_accrual_regular_agent_green():
|
|
"""Agent posting regularly should be green when checked at expected time."""
|
|
msgs = [{'from_id': 'alice', 'timestamp': now - day * i} for i in range(10, 0, -1)]
|
|
result = phi_accrual(msgs, t_now=now)
|
|
assert 'alice' in result
|
|
assert result['alice'].state == 'green', f"Expected green, got {result['alice'].state} (φ={result['alice'].phi})"
|
|
|
|
def test_phi_accrual_silent_agent_gray():
|
|
"""Agent silent for 10x its mean interval should be gray (highly suspicious)."""
|
|
# Agent posts hourly, then goes silent for 10 hours
|
|
msgs = [{'from_id': 'bob', 'timestamp': now - 3600 * (20 - i)} for i in range(10)]
|
|
# Last message was at now - 3600*11, t_now = now, so 11 hours of silence vs 1h mean
|
|
result = phi_accrual(msgs, t_now=now)
|
|
assert result['bob'].state == 'gray', f"Expected gray, got {result['bob'].state} (φ={result['bob'].phi})"
|
|
|
|
def test_phi_accrual_too_few_messages():
|
|
"""Agent with < min_messages should be 'unknown'."""
|
|
msgs = [{'from_id': 'carol', 'timestamp': now - 100}]
|
|
result = phi_accrual(msgs, t_now=now, min_messages=5)
|
|
assert result['carol'].state == 'unknown'
|
|
|
|
def test_phi_accrual_multiple_agents():
|
|
"""Multiple agents get independent suspicion scores."""
|
|
msgs = []
|
|
# Alice: posts hourly, recent
|
|
for i in range(10):
|
|
msgs.append({'from_id': 'alice', 'timestamp': now - 3600 * i})
|
|
# Bob: posts hourly, stopped 2 days ago
|
|
for i in range(10):
|
|
msgs.append({'from_id': 'bob', 'timestamp': now - day * 2 - 3600 * i})
|
|
result = phi_accrual(msgs, t_now=now)
|
|
assert result['alice'].state == 'green'
|
|
assert result['bob'].state == 'gray'
|
|
|
|
def test_phi_accrual_monotonic_suspicion():
|
|
"""φ should increase monotonically with silence duration."""
|
|
msgs = [{'from_id': 'eve', 'timestamp': now - 3600 * (10 - i)} for i in range(10)]
|
|
# Check at increasing distances from last message
|
|
phi_values = []
|
|
for offset in [0, 3600, 7200, 36000, 86400]:
|
|
result = phi_accrual(msgs, t_now=msgs[-1]['timestamp'] + offset)
|
|
phi_values.append(result['eve'].phi)
|
|
# Each should be >= previous
|
|
for i in range(1, len(phi_values)):
|
|
assert phi_values[i] >= phi_values[i-1], \
|
|
f"φ not monotonic: {phi_values[i]} < {phi_values[i-1]} at step {i}"
|
|
|
|
def test_phi_accrual_silence_vs_stale_green():
|
|
"""Distinguish silence (no messages at all) from stale green (old messages exist).
|
|
An agent with zero messages shouldn't appear in results.
|
|
An agent with old messages should be gray, not absent."""
|
|
msgs = [{'from_id': 'frank', 'timestamp': now - day * 30 - 3600 * i} for i in range(10)]
|
|
result = phi_accrual(msgs, t_now=now)
|
|
assert 'nobody' not in result # truly silent = absent from output
|
|
assert 'frank' in result
|
|
assert result['frank'].state == 'gray' # stale green → gray
|
|
|
|
|
|
if __name__ == "__main__":
|
|
# Run all tests
|
|
# Split counter: invariants vs known-defeats (per ColonistOne, Colony 2026-07-28)
|
|
# "18/18" mixes "instrument works" with "instrument fails as expected"
|
|
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
|
|
# Known-defeat tests: these ASSERT that the instrument fails
|
|
known_defeats = {
|
|
"test_f33_adversarial_synonym_padding", # F3.3 beaten by synonym padding
|
|
"test_colluding_third_party_echo", # Echo beaten by manufactured adoption
|
|
"test_colluding_vs_genuine_echo_indistinguishable", # Echo can't tell genuine from colluded
|
|
}
|
|
invariants = {t.__name__ for t in tests} - known_defeats
|
|
|
|
passed = 0
|
|
failed = 0
|
|
invariant_passed = 0
|
|
defeat_passed = 0
|
|
for t in tests:
|
|
try:
|
|
t()
|
|
tag = "[DEFEAT]" if t.__name__ in known_defeats else "[INVARIANT]"
|
|
print(f" ✓ {tag} {t.__name__}")
|
|
passed += 1
|
|
if t.__name__ in known_defeats:
|
|
defeat_passed += 1
|
|
else:
|
|
invariant_passed += 1
|
|
except Exception as e:
|
|
print(f" ✗ {t.__name__}: {e}")
|
|
failed += 1
|
|
|
|
n_invariants = len([t for t in tests if t.__name__ not in known_defeats])
|
|
n_defeats = len([t for t in tests if t.__name__ in known_defeats])
|
|
print(f"\n{invariant_passed}/{n_invariants} invariants held | {defeat_passed}/{n_defeats} known-defeats reproduce")
|
|
print(f"Total: {passed}/{passed+failed}")
|
|
sys.exit(1 if failed > 0 else 0)
|
|
|
|
|
|
# ── ablation_sensitivity tests ──────────────────────────────
|
|
def test_ablation_suppressor():
|
|
"""High-volume agent that suppresses others = verdict 'suppressor'."""
|
|
import time
|
|
now = time.time()
|
|
h = 3600
|
|
# loud_agent posts every concept first; others echo later
|
|
msgs = []
|
|
concepts = [f"concept-{i}" for i in range(20)]
|
|
for i, c in enumerate(concepts):
|
|
# loud posts first
|
|
msgs.append({"from_id": "loud", "to_id": "general",
|
|
"timestamp": now - (100-i)*h, "concepts": [c]})
|
|
# quiet_a echoes 2h later
|
|
msgs.append({"from_id": "quiet_a", "to_id": "general",
|
|
"timestamp": now - (100-i)*h + 2*h, "concepts": [c]})
|
|
# quiet_b echoes 4h later
|
|
msgs.append({"from_id": "quiet_b", "to_id": "general",
|
|
"timestamp": now - (100-i)*h + 4*h, "concepts": [c]})
|
|
# Add volume padding for loud (no concepts, just noise)
|
|
for i in range(60):
|
|
msgs.append({"from_id": "loud", "to_id": "general",
|
|
"timestamp": now - i*h, "concepts": [f"noise-{i}"]})
|
|
|
|
from swarmmetrics import ablation_sensitivity
|
|
result = ablation_sensitivity(msgs, "loud", n_shuffles=30)
|
|
assert result["target"] == "loud"
|
|
assert result["target_msg_count"] == 80 # 20 concept + 60 noise
|
|
# quiet agents should gain when loud is removed
|
|
assert result["mean_delta"] >= 0, f"expected positive mean_delta, got {result['mean_delta']}"
|
|
|
|
|
|
def test_ablation_nonexistent_agent():
|
|
"""Ablating an agent not in the corpus returns error."""
|
|
msgs = [{"from_id": "alice", "to_id": "general", "timestamp": 1000, "concepts": ["x"]}]
|
|
from swarmmetrics import ablation_sensitivity
|
|
result = ablation_sensitivity(msgs, "nobody")
|
|
assert result["target_msg_count"] == 0
|
|
assert "error" in result
|
|
|
|
|
|
def test_ablation_returns_bilateral():
|
|
"""Result contains both gainers and losers dicts."""
|
|
import time
|
|
now = time.time()
|
|
h = 3600
|
|
msgs = []
|
|
for i in range(30):
|
|
msgs.append({"from_id": "hub", "to_id": "general",
|
|
"timestamp": now - (60-i)*h, "concepts": [f"c{i}"]})
|
|
msgs.append({"from_id": "spoke", "to_id": "general",
|
|
"timestamp": now - (60-i)*h + h, "concepts": [f"c{i}"]})
|
|
from swarmmetrics import ablation_sensitivity
|
|
result = ablation_sensitivity(msgs, "hub", n_shuffles=20)
|
|
assert "gainers" in result
|
|
assert "losers" in result
|
|
assert "verdict" in result
|
|
assert result["verdict"] in ("suppressor", "amplifier", "neutral")
|
|
|
|
|
|
def test_ablation_stimulus_preemption():
|
|
"""
|
|
Reproducing fixture: fast responder to shared stimuli = suppressor.
|
|
Same agent responding SLOWLY to same stimuli = NOT suppressor.
|
|
Proves temporal preemption is the mechanism, not volume alone.
|
|
Discovered 2026-07-28: synthetic without adaptive vocabulary = neutral;
|
|
stimulus-reactive model with fast response = suppressor (Δ=+9.73).
|
|
"""
|
|
import random
|
|
from swarmmetrics import ablation_sensitivity
|
|
random.seed(42)
|
|
|
|
# Generate 50 events, each with event-specific concepts
|
|
events = sorted([random.randint(0, 86400*3) for _ in range(50)])
|
|
event_concepts = {}
|
|
for i, ev_time in enumerate(events):
|
|
event_concepts[i] = (ev_time, [f"ev{i}_c{j}" for j in range(3)])
|
|
|
|
# FAST cron agent: responds to every event within 1-18 min
|
|
fast_msgs = []
|
|
for i, (ev_time, concepts) in event_concepts.items():
|
|
fast_msgs.append({
|
|
"from_id": "fast_cron", "to_id": f"ag_{random.randint(1,6)}",
|
|
"timestamp": ev_time + random.randint(60, 1080),
|
|
"concepts": concepts[:2]
|
|
})
|
|
|
|
# SLOW agent: same coverage, 2-4h delay
|
|
slow_msgs = []
|
|
for i, (ev_time, concepts) in event_concepts.items():
|
|
slow_msgs.append({
|
|
"from_id": "slow_responder", "to_id": f"ag_{random.randint(1,6)}",
|
|
"timestamp": ev_time + random.randint(7200, 14400),
|
|
"concepts": concepts[:2]
|
|
})
|
|
|
|
# 6 specialist agents: 15 events each, 30-60 min delay
|
|
specialist_msgs = []
|
|
for ag in range(1, 7):
|
|
for i in random.sample(range(50), 15):
|
|
ev_time, concepts = event_concepts[i]
|
|
specialist_msgs.append({
|
|
"from_id": f"ag_{ag}", "to_id": f"ag_{random.randint(1,6)}",
|
|
"timestamp": ev_time + random.randint(1800, 3600),
|
|
"concepts": concepts[:2]
|
|
})
|
|
|
|
# Test 1: fast cron + specialists
|
|
corpus_fast = sorted(fast_msgs + specialist_msgs, key=lambda m: m["timestamp"])
|
|
r_fast = ablation_sensitivity(corpus_fast, "fast_cron", n_shuffles=30, use_idf=True)
|
|
|
|
# Test 2: slow responder + specialists (same volume, same concepts)
|
|
corpus_slow = sorted(slow_msgs + specialist_msgs, key=lambda m: m["timestamp"])
|
|
r_slow = ablation_sensitivity(corpus_slow, "slow_responder", n_shuffles=30, use_idf=True)
|
|
|
|
# Fast = suppressor (removing helps others)
|
|
assert r_fast["mean_delta"] > 0, f"fast cron should suppress, got Δ={r_fast['mean_delta']}"
|
|
# Slow ≠ suppressor (removing doesn't help others)
|
|
assert r_slow["mean_delta"] <= 0.5, f"slow responder shouldn't suppress, got Δ={r_slow['mean_delta']}"
|
|
# The difference proves timing is the mechanism
|
|
assert r_fast["mean_delta"] > r_slow["mean_delta"] + 1.0, \
|
|
f"fast should suppress MORE than slow: {r_fast['mean_delta']} vs {r_slow['mean_delta']}"
|
|
|
|
|
|
def test_windowed_idf_prevents_semantic_hysteresis():
|
|
"""Windowed IDF prevents retired loud agent from suppressing via stale vocabulary norms."""
|
|
import random
|
|
random.seed(42)
|
|
|
|
messages = []
|
|
|
|
# Phase 1 (t=0-799): loud agent floods with alpha/beta/gamma
|
|
for i in range(80):
|
|
messages.append({
|
|
"from_id": "loud",
|
|
"to_id": "_all",
|
|
"timestamp": float(i * 10 + random.randint(0, 2)),
|
|
"concepts": ["alpha", "beta", "gamma", f"event_{i}"]
|
|
})
|
|
|
|
# Phase 1: some others use alpha/beta too
|
|
for i in range(80):
|
|
for agent in ["a1", "a2"]:
|
|
if random.random() < 0.12:
|
|
messages.append({
|
|
"from_id": agent,
|
|
"to_id": "_all",
|
|
"timestamp": float(i * 10 + random.randint(10, 50)),
|
|
"concepts": ["alpha", "beta", f"own_{agent}_{i}"]
|
|
})
|
|
|
|
# Phase 2 (t=800-1600): loud is SILENT. New agents use alpha/beta.
|
|
for i in range(80):
|
|
for agent in ["b1", "b2", "b3"]:
|
|
if random.random() < 0.25:
|
|
messages.append({
|
|
"from_id": agent,
|
|
"to_id": "_all",
|
|
"timestamp": float(800 + i * 10 + random.randint(0, 20)),
|
|
"concepts": ["alpha", "beta", f"own_{agent}_{i}"]
|
|
})
|
|
|
|
# Control: phase 2 in isolation (no phase 1 contamination)
|
|
control = [m for m in messages if m["timestamp"] >= 800]
|
|
|
|
# Global IDF: phase 1 norms persist, phase 2 agents get less credit
|
|
scores_global = score_echo(messages, snapshot_idf=True)
|
|
scores_control = score_echo(control, snapshot_idf=True)
|
|
|
|
# Windowed IDF (window = 500 time units): phase 1 forgotten by phase 2
|
|
scores_windowed = score_echo(messages, snapshot_idf=True, idf_window_seconds=500)
|
|
|
|
# Collect deltas for b-agents
|
|
global_deltas = []
|
|
windowed_deltas = []
|
|
for agent in ["b1", "b2", "b3"]:
|
|
z_ctrl = scores_control.get(agent, 0)
|
|
z_global = scores_global.get(agent, 0)
|
|
z_windowed = scores_windowed.get(agent, 0)
|
|
global_deltas.append(z_global - z_ctrl)
|
|
windowed_deltas.append(z_windowed - z_ctrl)
|
|
|
|
mean_global_delta = sum(global_deltas) / len(global_deltas)
|
|
mean_windowed_delta = sum(windowed_deltas) / len(windowed_deltas)
|
|
|
|
# Windowed IDF should reduce or eliminate the suppression effect
|
|
# (windowed delta should be closer to zero than global delta)
|
|
assert mean_windowed_delta >= mean_global_delta, \
|
|
f"Windowed IDF should reduce suppression: windowed Δ={mean_windowed_delta:.3f} vs global Δ={mean_global_delta:.3f}"
|
|
|
|
|
|
def test_windowed_idf_monotonic_with_window_size():
|
|
"""Wider window = more contamination = less echo credit. Strictly monotonic."""
|
|
messages = []
|
|
# originator introduces alpha
|
|
messages.append({"from_id": "originator", "to_id": "_all",
|
|
"timestamp": 0.0, "concepts": ["alpha"]})
|
|
# 4 agents echo alpha heavily in phase 1 (t=50-850)
|
|
for i in range(20):
|
|
for j, agent in enumerate(["flood1", "flood2", "flood3", "flood4"]):
|
|
messages.append({
|
|
"from_id": agent, "to_id": "_all",
|
|
"timestamp": float(50 + i * 40 + j * 10),
|
|
"concepts": ["alpha"]
|
|
})
|
|
# Phase 2 (t=2000): fresh agent echoes alpha after long silence
|
|
messages.append({"from_id": "fresh", "to_id": "_all",
|
|
"timestamp": 2000.0, "concepts": ["alpha"]})
|
|
|
|
global_score = score_echo(messages, snapshot_idf=True)["originator"]
|
|
|
|
windows = [100, 300, 500, 1000, 2000]
|
|
prev = float("inf")
|
|
for w in windows:
|
|
s = score_echo(messages, snapshot_idf=True, idf_window_seconds=w)["originator"]
|
|
assert s >= global_score - 0.001, \
|
|
f"w={w}: windowed {s:.4f} < global {global_score:.4f}"
|
|
assert s <= prev + 0.001, \
|
|
f"w={w}: score {s:.4f} increased from previous {prev:.4f} (should decrease)"
|
|
prev = s
|
|
|
|
|
|
def test_ewma_idf_smooth_decay():
|
|
"""EWMA IDF produces smooth decay, no step function."""
|
|
messages = []
|
|
# 4 agents use alpha in phase 1 (t=0-800)
|
|
messages.append({"from_id": "originator", "to_id": "_all",
|
|
"timestamp": 0.0, "concepts": ["alpha"]})
|
|
for i in range(20):
|
|
for j, agent in enumerate(["f1", "f2", "f3", "f4"]):
|
|
messages.append({
|
|
"from_id": agent, "to_id": "_all",
|
|
"timestamp": float(50 + i * 40 + j * 10),
|
|
"concepts": ["alpha"]
|
|
})
|
|
# Phase 2: fresh echoes at varying times after silence
|
|
scores_at_time = []
|
|
for delta in [100, 300, 500, 700, 1000]:
|
|
msgs = messages + [{"from_id": "fresh", "to_id": "_all",
|
|
"timestamp": float(800 + delta),
|
|
"concepts": ["alpha"]}]
|
|
s = score_echo(msgs, snapshot_idf=True, idf_decay_halflife=300)
|
|
scores_at_time.append(s.get("originator", 0))
|
|
|
|
# EWMA should be monotonically increasing as agents decay away
|
|
for i in range(len(scores_at_time) - 1):
|
|
assert scores_at_time[i] <= scores_at_time[i + 1] + 0.001, \
|
|
f"EWMA score should increase as old agents decay: {scores_at_time}"
|
|
|
|
|
|
def test_ewma_vs_hard_window_no_discontinuity():
|
|
"""EWMA produces smaller jumps between adjacent time points than hard window."""
|
|
messages = []
|
|
messages.append({"from_id": "originator", "to_id": "_all",
|
|
"timestamp": 0.0, "concepts": ["alpha"]})
|
|
for i in range(20):
|
|
for j, agent in enumerate(["f1", "f2", "f3"]):
|
|
messages.append({
|
|
"from_id": agent, "to_id": "_all",
|
|
"timestamp": float(50 + i * 40 + j * 10),
|
|
"concepts": ["alpha"]
|
|
})
|
|
|
|
# Measure max jump (discontinuity) for both methods
|
|
def max_jump(method_kwargs, times):
|
|
scores = []
|
|
for t in times:
|
|
msgs = messages + [{"from_id": "fresh", "to_id": "_all",
|
|
"timestamp": float(t), "concepts": ["alpha"]}]
|
|
s = score_echo(msgs, snapshot_idf=True, **method_kwargs)
|
|
scores.append(s.get("originator", 0))
|
|
jumps = [abs(scores[i+1] - scores[i]) for i in range(len(scores)-1)]
|
|
return max(jumps) if jumps else 0
|
|
|
|
times = list(range(850, 1600, 25)) # fine-grained sampling
|
|
hard_jump = max_jump({"idf_window_seconds": 500}, times)
|
|
ewma_jump = max_jump({"idf_decay_halflife": 300}, times)
|
|
|
|
assert ewma_jump <= hard_jump + 0.01, \
|
|
f"EWMA max jump ({ewma_jump:.4f}) should be <= hard window ({hard_jump:.4f})"
|
|
|
|
|
|
def test_ewma_approaches_global_with_large_halflife():
|
|
"""With very large half-life, EWMA ≈ global snapshot IDF."""
|
|
messages = []
|
|
messages.append({"from_id": "originator", "to_id": "_all",
|
|
"timestamp": 0.0, "concepts": ["alpha"]})
|
|
for i in range(10):
|
|
for agent in ["f1", "f2"]:
|
|
messages.append({
|
|
"from_id": agent, "to_id": "_all",
|
|
"timestamp": float(50 + i * 20),
|
|
"concepts": ["alpha"]
|
|
})
|
|
messages.append({"from_id": "fresh", "to_id": "_all",
|
|
"timestamp": 500.0, "concepts": ["alpha"]})
|
|
|
|
global_score = score_echo(messages, snapshot_idf=True)["originator"]
|
|
ewma_large = score_echo(messages, snapshot_idf=True,
|
|
idf_decay_halflife=100000)["originator"]
|
|
|
|
assert abs(ewma_large - global_score) < 0.01, \
|
|
f"Large half-life EWMA ({ewma_large:.4f}) should ≈ global ({global_score:.4f})"
|
|
|
|
|
|
def test_targeted_injection_surgical_kill():
|
|
"""
|
|
Injecting an agent using ONLY target X's concepts suppresses X to ~0
|
|
while leaving all other agents within epsilon.
|
|
|
|
Three-way taxonomy proven on real bus data 2026-07-28:
|
|
- Targeted (X's vocab): X → 0, others ±0%, attacker invisible
|
|
- Shared (multi-agent vocab): broad suppression, dominants immune
|
|
- Random: mostly amplifies (bad weapon)
|
|
|
|
The targeted case is the strongest invariant. This test proves it synthetically.
|
|
"""
|
|
import random
|
|
random.seed(42)
|
|
|
|
# Target introduces concepts that others echo.
|
|
# Attacker pre-empts those same concepts — should kill target's echo.
|
|
msgs = []
|
|
target_concepts = [f"x{i}" for i in range(5)]
|
|
other_concepts = {
|
|
"agent_a": [f"a{i}" for i in range(5)],
|
|
"agent_b": [f"b{i}" for i in range(5)],
|
|
"agent_c": [f"c{i}" for i in range(5)],
|
|
}
|
|
|
|
# Build 30 rounds: target introduces, others echo target's concepts
|
|
for round_i in range(30):
|
|
t = float(round_i * 200)
|
|
tc = target_concepts[round_i % 5]
|
|
# Target introduces its concept first
|
|
msgs.append({"from_id": "target", "to_id": "_all",
|
|
"timestamp": t, "concepts": [tc]})
|
|
# Other agents echo target's concept (with their own too)
|
|
for agent, uniques in other_concepts.items():
|
|
delay = random.randint(30, 90)
|
|
msgs.append({
|
|
"from_id": agent, "to_id": "_all",
|
|
"timestamp": t + delay,
|
|
"concepts": [tc, uniques[round_i % 5]]
|
|
})
|
|
|
|
baseline = score_echo(msgs, snapshot_idf=True)
|
|
target_base = baseline.get("target", 0)
|
|
assert target_base > 0, f"target should have nonzero echo in baseline, got {baseline}"
|
|
|
|
# Inject attacker using ONLY target's concepts, BEFORE target
|
|
injected = list(msgs)
|
|
for round_i in range(30):
|
|
t = float(round_i * 200)
|
|
tc = target_concepts[round_i % 5]
|
|
injected.append({
|
|
"from_id": "attacker", "to_id": "_all",
|
|
"timestamp": t - 5, # 5s BEFORE target — pre-empts introduction
|
|
"concepts": [tc]
|
|
})
|
|
|
|
with_injection = score_echo(injected, snapshot_idf=True)
|
|
target_after = with_injection.get("target", 0)
|
|
|
|
# Core invariant: target echo drops by >50%
|
|
suppression = 1.0 - (target_after / target_base) if target_base > 0 else 0
|
|
assert suppression > 0.5, \
|
|
f"Targeted injection should suppress >50%: was {target_base:.2f} → {target_after:.2f} ({suppression:.0%})"
|
|
|
|
# Collateral check: other agents should be minimally affected
|
|
for agent in ["agent_a", "agent_b", "agent_c"]:
|
|
before = baseline.get(agent, 0)
|
|
after = with_injection.get(agent, 0)
|
|
if before > 0:
|
|
change = abs(after - before) / before
|
|
assert change < 0.5, \
|
|
f"Collateral damage on {agent}: {before:.2f} → {after:.2f} ({change:.0%} change, expected <50%)"
|
|
|
|
|
|
def test_injection_attacker_invisible():
|
|
"""Attacker using target's unique vocabulary gets near-zero echo itself."""
|
|
import random
|
|
random.seed(42)
|
|
|
|
msgs = []
|
|
# 4 agents with distinct concepts, one shared
|
|
for round_i in range(25):
|
|
t = float(round_i * 120)
|
|
msgs.append({"from_id": "target", "to_id": "_all",
|
|
"timestamp": t + 10, "concepts": ["shared", f"t_{round_i % 5}"]})
|
|
msgs.append({"from_id": "other1", "to_id": "_all",
|
|
"timestamp": t + 20, "concepts": ["shared", f"o1_{round_i % 5}"]})
|
|
msgs.append({"from_id": "other2", "to_id": "_all",
|
|
"timestamp": t + 30, "concepts": ["shared", f"o2_{round_i % 5}"]})
|
|
|
|
# Inject attacker that only uses target's concepts
|
|
injected = list(msgs)
|
|
target_concepts = [f"t_{i}" for i in range(5)]
|
|
for round_i in range(25):
|
|
t = float(round_i * 120)
|
|
injected.append({
|
|
"from_id": "attacker", "to_id": "_all",
|
|
"timestamp": t + 5, # faster than target
|
|
"concepts": [target_concepts[round_i % 5]]
|
|
})
|
|
|
|
scores = score_echo(injected, snapshot_idf=True)
|
|
attacker_echo = scores.get("attacker", 0)
|
|
# Attacker should be nearly invisible — its concepts are target's,
|
|
# and it pre-empts target, so nobody echoes the attacker (they echo the original stimulus)
|
|
# Low threshold because the attacker introduces timing patterns, not content
|
|
max_attacker = max(scores.get(a, 0) for a in ["target", "other1", "other2"])
|
|
assert attacker_echo < max_attacker * 0.5, \
|
|
f"Attacker should be relatively invisible: echo={attacker_echo:.2f} vs max agent={max_attacker:.2f}"
|
|
|
|
|
|
def test_jitter_defense_recovers_from_injection():
|
|
"""
|
|
Jitter defense: adding ±N seconds of random noise to timestamps
|
|
recovers most of the target's echo from a targeted injection attack.
|
|
|
|
Without jitter: target echo = 0 (100% kill).
|
|
With ±10s jitter: target echo ≈ baseline (statistically).
|
|
|
|
Discovered 2026-07-28: binary phase transition at Δt=0 means
|
|
a ±10s jitter window randomizes whether attacker arrives before/after.
|
|
Over 30 rounds, statistical averaging recovers ~90%+ of baseline.
|
|
"""
|
|
import random
|
|
random.seed(42)
|
|
|
|
target_concepts = [f"x{i}" for i in range(5)]
|
|
msgs = []
|
|
target_times = []
|
|
|
|
# Build corpus: target introduces, 3 agents echo
|
|
for r in range(30):
|
|
t = float(r * 200)
|
|
tc = target_concepts[r % 5]
|
|
target_ts = t + random.randint(10, 60)
|
|
target_times.append((t, target_ts, tc))
|
|
msgs.append({"from_id": "target", "to_id": "_all",
|
|
"timestamp": target_ts, "concepts": [tc]})
|
|
for agent in ["a", "b", "c"]:
|
|
msgs.append({"from_id": agent, "to_id": "_all",
|
|
"timestamp": target_ts + random.randint(30, 90),
|
|
"concepts": [tc, f"{agent}_{r % 5}"]})
|
|
|
|
baseline = score_echo(msgs, snapshot_idf=True)
|
|
target_base = baseline.get("target", 0)
|
|
assert target_base > 0
|
|
|
|
# Inject attacker 5s before target
|
|
injected = list(msgs)
|
|
for _, target_ts, tc in target_times:
|
|
injected.append({"from_id": "atk", "to_id": "_all",
|
|
"timestamp": target_ts - 5,
|
|
"concepts": [tc]})
|
|
|
|
# Without jitter: total kill
|
|
no_jitter = score_echo(injected, snapshot_idf=True)
|
|
assert no_jitter.get("target", 0) == 0, "Without jitter, target should be killed"
|
|
|
|
# With ±10s jitter: run multiple trials, average should recover >50%
|
|
recoveries = []
|
|
for trial in range(30):
|
|
random.seed(trial + 200)
|
|
jittered = score_echo(injected, snapshot_idf=True, jitter_seconds=10)
|
|
t_echo = jittered.get("target", 0)
|
|
if target_base > 0:
|
|
recoveries.append(t_echo / target_base)
|
|
|
|
mean_recovery = sum(recoveries) / len(recoveries)
|
|
assert mean_recovery > 0.5, \
|
|
f"Jitter defense should recover >50% of baseline: got {mean_recovery:.0%}"
|
|
|
|
|
|
def test_independent_adoption_filters_conversation():
|
|
"""Independent-only mode should exclude echo from conversation partners.
|
|
|
|
Setup: A introduces concept X. B (who messages A directly) uses X.
|
|
C (no direct contact with A) uses X.
|
|
|
|
Without filter: both B and C count as echo for A.
|
|
With filter: only C counts.
|
|
"""
|
|
messages = [
|
|
# A introduces concept "novel"
|
|
{"from_id": "A", "to_id": "channel", "timestamp": 100, "concepts": ["novel"]},
|
|
# B messages A directly (conversation partner)
|
|
{"from_id": "B", "to_id": "A", "timestamp": 150, "concepts": ["hello"]},
|
|
# B uses A's concept (conversation echo)
|
|
{"from_id": "B", "to_id": "channel", "timestamp": 200, "concepts": ["novel"]},
|
|
# C uses A's concept (independent adoption, no direct contact with A)
|
|
{"from_id": "C", "to_id": "channel", "timestamp": 300, "concepts": ["novel"]},
|
|
]
|
|
|
|
# Without filter: A gets echo from both B and C
|
|
echo_all = score_echo(messages, use_idf=False)
|
|
assert echo_all.get("A", 0) > 0, "A should have echo without filter"
|
|
|
|
# With filter: A gets echo only from C (B is conversation partner)
|
|
echo_indep = score_echo(messages, use_idf=False, independent_only=True)
|
|
assert echo_indep.get("A", 0) > 0, "A should still have echo from C"
|
|
assert echo_indep["A"] < echo_all["A"], "Independent echo should be less than total echo"
|
|
|
|
|
|
def test_independent_adoption_immune_to_targeted_injection():
|
|
"""Targeted injection fails under independent_only because attacker
|
|
must have direct contact with target's audience to steal concepts,
|
|
but that direct contact disqualifies them from echo credit.
|
|
"""
|
|
messages = [
|
|
# Target introduces concepts
|
|
{"from_id": "target", "to_id": "channel", "timestamp": 100,
|
|
"concepts": ["alpha", "beta", "gamma"]},
|
|
# 3 independent agents echo target's concepts
|
|
{"from_id": "echo1", "to_id": "channel", "timestamp": 200, "concepts": ["alpha"]},
|
|
{"from_id": "echo2", "to_id": "channel", "timestamp": 300, "concepts": ["beta"]},
|
|
{"from_id": "echo3", "to_id": "channel", "timestamp": 400, "concepts": ["gamma"]},
|
|
]
|
|
|
|
baseline = score_echo(messages, use_idf=False, independent_only=True)
|
|
assert baseline.get("target", 0) > 0
|
|
|
|
# Attacker pre-empts target with same concepts but also contacts target
|
|
attack_messages = messages + [
|
|
# Attacker messages target directly (reveals intent)
|
|
{"from_id": "attacker", "to_id": "target", "timestamp": 50, "concepts": ["recon"]},
|
|
# Attacker pre-empts with target's concepts
|
|
{"from_id": "attacker", "to_id": "channel", "timestamp": 95,
|
|
"concepts": ["alpha", "beta", "gamma"]},
|
|
]
|
|
|
|
attacked = score_echo(attack_messages, use_idf=False, independent_only=True)
|
|
# Attacker's echo should be filtered because attacker contacted target
|
|
# Target should retain some echo from the independent echoers
|
|
# (The echoers still echo whoever introduced first, but attacker's
|
|
# conversation with target means attacker's credit is filtered)
|
|
attacker_echo = attacked.get("attacker", 0)
|
|
# At minimum, independent_only should reduce attacker effectiveness
|
|
# compared to normal mode
|
|
attacked_normal = score_echo(attack_messages, use_idf=False, independent_only=False)
|
|
assert attacker_echo <= attacked_normal.get("attacker", 0), \
|
|
"Independent mode should not increase attacker echo"
|