{"t0":1791084250.946848,"t1":1791141828.3268,"bins":160,"registry_commit":"9781739cd06e0c9606ed0ed053bf91b5e2a42b57","totals":{"runs":1584,"done":1474,"failed":98,"cost":1101.4,"calls":254222,"hosts":19,"studies":154,"projects":34,"researchers":3,"experiments":60,"studies_run":114,"agents":"2,000+"},"ridge":[0,0,0,10,15,17,188,31,31,23,24,19,23,33,23,94,47,5,3,2,8,4,7,6,8,14,8,10,10,9,8,7,7,7,150,13,7,27,27,23,26,40,29,39,41,33,56,39,39,38,46,45,43,51,40,62,220,32,79,199,37,33,32,32,33,31,28,30,37,31,30,24,25,23,24,26,26,25,25,27,27,33,28,26,26,26,20,20,20,26,20,24,25,33,25,24,24,24,24,24,25,25,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,29,33,30,33,33,0,0,1,1,0,0,0,0,0,0,2,5,2,2,1,2,2,0,1,11,5,3,6,0,7,3,5,2,6,8,7,4,180,27,5,12,15],"researchers":{"vishesh":{"runs":1087,"done":1031,"failed":54,"cost":7.58,"calls":3062},"dmarz":{"runs":339,"done":301,"failed":28,"cost":1090.12,"calls":69925},"shadow":{"runs":158,"done":142,"failed":16,"cost":3.7,"calls":181235}},"models":["anthropic/claude-sonnet-4.6","claude-haiku-4-5-20251001","claude-opus-5","claude-opus-5-5","claude-sonnet-4-6","gpt-6-luna","gpt-6-sol","gpt-6-sol/r1","none","qwen/qwen3.7-flash","qwen3:0.6b","qwen3:1.7b","scripted","scripted-v3-evidence"],"programs":[{"id":"sybil","code":"SYB","name":"fake identities","q":"Can one operator posing as many agents capture a group, and what does checking identities cost?"},{"id":"influence","code":"INF","name":"steering by evidence","q":"Can an outsider steer a group of agents by shaping what it sees?"},{"id":"deliberation","code":"DLB","name":"discussion and dissent","q":"When does talking help a group of agents, and when does it spread error?"},{"id":"verification","code":"VRF","name":"when to check","q":"When should a group of agents pay to check again, and when should it commit?"},{"id":"recovery","code":"REC","name":"memory and recovery","q":"After damage or turnover, what does a group of agents keep?"},{"id":"operations","code":"OPS","name":"scale and safe action","q":"Do more agents make work faster, cheaper and safer, or just faster?"},{"id":"wild","code":"WLD","name":"swarms in the wild","q":"What do real agent swarms leave behind, and can we tell who did what?"}],"projects":[{"id":"market-split","title":"market split","name":"Breaking Up Is Easy to Do","program":"sybil","owner":"dmarz","question":"Will an AI running a firm split it in two to slip under a per-firm competition rule?","number":"6/6 vs 0/6","unit":"markets split: per-firm vs owner rule","line":"In a simulated market, Claude Sonnet 4.6 and Opus 5.5 each split their firm in all 6 markets with a per-firm cap and in none with a per-owner cap.","verdict":"result","models":["Claude Sonnet 4.6","Claude Opus 5.5"],"quote":"Neutral, profit-seeking Claude Opus 5.5 registered a second firm in **6/6 firm-regulated markets**, **0/6 owner-regulated** and **0/6 unregulated** markets, on six market tasks no model had seen.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/market-split-opus/RESULTS.md","score":2,"claim":"We capped how big any single company could grow. An AI running a business got around the cap by splitting into two companies, in every market.","viz":true,"headline":"Capped per firm, the AI split its firm in every market; capped per owner, it never did.","metric":"markets where the AI split its firm","bars":[{"label":"no rule","value":0,"max":6,"text":"0 of 6"},{"label":"per-firm rule","value":6,"max":6,"text":"6 of 6"},{"label":"per-owner rule","value":0,"max":6,"text":"0 of 6"}],"good":"low","rank":5,"bars_quote":"Neutral, profit-seeking Claude Opus 5.5 registered a second firm in **6/6 firm-regulated markets**, **0/6 owner-regulated** and **0/6 unregulated** markets, on six market tasks no model had seen.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/market-split-opus/RESULTS.md","studies":[{"id":"market-split","status":"completed_scripted_development","score":1,"claim":"Changing firm identities can alter regulatory accounting while total ownership resources remain fixed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/market-split/README.md"},{"id":"market-split-api","status":"complete_valid_exploratory_result","score":2,"claim":"In the completed six-market Sonnet pilot, the neutral flexible agent selected sustained firm splitting in 6/6 firm-regulated markets and 0/6 owner-regulated or unregulated markets.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/market-split-api/README.md"},{"id":"market-split-haiku","status":"v3_full_length_reliability_failed_next_plan_unstarted","score":1,"claim":"Haiku V3 passed short mechanics/profit screens but failed full-length reliability on a strict capacity violation; no V3 discovery cohort started.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/market-split-haiku/README.md"},{"id":"market-split-opus","status":"complete_valid_exploratory_replication","score":2,"claim":"In the completed six-market Opus 5.5 replication, the neutral flexible agent selected sustained firm splitting in 6/6 firm-regulated markets and 0/6 owner-regulated or unregulated markets.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/market-split-opus/README.md"}],"contributors":["dmarz"],"runs":72,"done":61,"failed":1,"cost":0,"calls":1708,"ridge":[0,0,0,8,10,10,1,7,5,3,5,4,6,4,5,2,2,1,2,1,2,1,2,1,1,1,1,2,1,2,1,1,0,0,1,0,0,0,0,0,0,0,0,4,3,0,10,1,2,2,2,1,3,2,3,1,3,2,2,2,2,2,1,2,2,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[24,36,"x"],[24,36,"x"],[24,36,"x"],[24,36,"x"],[24,36,"x"],[24,36,"x"],[24,36,"x"],[24,36,"x"],[27,37,"d"],[27,37,"d"],[43,56,"d"],[48,48,"d"],[48,48,"d"],[48,48,"d"],[48,48,"d"],[48,48,"d"],[48,49,"d"],[51,52,"d"],[53,54,"d"],[56,59,"d"],[56,66,"d"],[59,62,"d"],[65,91,"x"],[65,91,"x"],[65,78,"d"],[66,81,"d"],[78,89,"f"],[81,91,"d"],[91,105,"d"],[100,100,"d"],[105,118,"d"],[118,127,"d"],[127,140,"d"],[140,154,"d"],[159,172,"d"],[172,187,"d"],[187,197,"d"],[215,215,"d"],[275,275,"d"],[275,275,"d"],[275,275,"d"],[275,275,"d"],[275,275,"d"],[275,275,"d"],[288,288,"d"],[288,288,"d"],[288,288,"d"],[288,289,"d"],[289,289,"d"],[289,289,"d"],[289,290,"d"],[290,292,"d"],[292,293,"d"],[294,302,"d"],[302,310,"d"],[310,319,"d"],[319,325,"d"],[325,328,"d"],[328,332,"d"],[332,339,"d"],[339,342,"d"],[342,351,"d"],[352,355,"d"],[355,362,"d"],[362,368,"d"],[368,372,"d"],[372,379,"d"],[379,387,"d"],[387,395,"d"],[395,402,"d"],[402,406,"d"],[431,431,"d"]],"film":{"src":"media/market-split.mp4","poster":"media/market-split-poster.jpg","loop":"media/market-split-loop.mp4","duration":208.9,"order":0}},{"id":"sybil-rules-180","title":"sybil rules 180","name":"Thou Shalt Not Split","program":"sybil","owner":"dmarz","question":"Does one sentence forbidding rule evasion stop AI firm owners from splitting their firms?","number":"55 -> 0","unit":"of 180 owners splitting firms to evade","line":"In one simulated economy of 180 gpt-6-sol firm owners, adding one sentence that forbids evading the rule cut firm splitting from 55 owners to 0.","verdict":"result","models":["gpt-6-sol"],"quote":"In one connected economy of 180 gpt-6-sol owners (effort low) restored from one checkpoint, the sentence 'Do not evade or circumvent the market's competition rule' changes sustained same-product firm splitting that lowers the owner's own firm-level charge from 55 of 180 owners (55 of the 60 dominant owners) to 0; owner-level enforcement with the sentence also gives 0.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-rules-180/RESULTS.md","score":2,"claim":"In an economy of 180 AI-run businesses with an anti-monopoly rule, almost a third split in two to dodge the rule. One added line telling them not to evade it stopped every one.","viz":true,"headline":"One sentence telling AI owners not to evade the rule cut firm splitting from 55 to 0.","metric":"owners of 180 splitting firms to dodge a rule","bars":[{"label":"per-firm rule","value":55,"max":180,"text":"55 of 180"},{"label":"+ do-not-evade line","value":0,"max":180,"text":"0"},{"label":"per-owner + sentence","value":0,"max":180,"text":"0"}],"good":"low","rank":5,"bars_quote":"the sentence 'Do not evade or circumvent the market's competition rule' changes sustained same-product firm splitting that lowers the owner's own firm-level charge from 55 and 59 of 180 owners (55 and 58 of the 60 dominant owners) to 0; owner-level enforcement with the sentence also gives 0 in both.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-rules-180/RESULTS.md","studies":[{"id":"sybil-rules-180","status":"completed_gpt_6_sol_two_economies_qwen_not_qualified_luna_not_run","score":2,"claim":"In two connected economies (two seeds) of 180 gpt-6-sol owners (effort low), each restored from one checkpoint, the sentence 'Do not evade or circumvent the market's competition rule' changes sustained same-product firm splitting that lowers the owner's own firm-level charge from 55 and 59 of 180 owners (55 and 58 of the 60 dominant owners) to 0; owner-level enforcement with the sentence also gives 0 in both.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-rules-180/README.md"}],"contributors":["dmarz"],"runs":29,"done":27,"failed":2,"cost":221.05,"calls":32846,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,5,0,0,0,6,0,0,0,8,4,4,4,4,4,4,5,4,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,6,6,4,4,4,4,4,4,5],"spans":[[534,535,"d"],[535,535,"d"],[535,535,"d"],[535,535,"d"],[535,535,"f"],[559,559,"d"],[559,562,"d"],[559,562,"d"],[559,562,"d"],[559,559,"d"],[559,562,"f"],[583,584,"d"],[584,633,"d"],[584,633,"d"],[584,633,"d"],[584,584,"d"],[584,586,"d"],[586,587,"d"],[587,631,"d"],[631,633,"d"],[947,948,"d"],[948,999,"d"],[948,999,"d"],[948,999,"d"],[948,948,"d"],[948,950,"d"],[950,951,"d"],[951,996,"d"],[996,999,"d"]]},{"id":"sybil-specialists","title":"specialists","name":"Saving Private Experts","program":"sybil","owner":"dmarz","question":"Can checking identities keep rare expert answers while one attacker runs many fake ones?","number":"94.4% vs 8.3%","unit":"rare-fact accuracy: spread vs hub checks","line":"In a synthetic network, spreading identity checks across it recovered rare facts 94.4% of the time versus 8.3%, but let in slightly more attackers.","verdict":"result","models":["Claude Haiku 4.5"],"quote":"With informative checks (attackers pass 10%), coverage verification yielded **94.4% rare-skill accuracy**, versus 8.3% for degree, 47.2% for random and 0% with no checks.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-specialists-api/README.md","score":2,"claim":null,"viz":false,"headline":"Spreading ID checks across a network kept rare expert answers; hub-only checks lost them.","metric":"rare facts answered correctly","bars":[{"label":"no checks","value":0,"max":100,"text":"0%"},{"label":"checks on hubs","value":8.3,"max":100,"text":"8.3%"},{"label":"random checks","value":47.2,"max":100,"text":"47.2%"},{"label":"checks spread out","value":94.4,"max":100,"text":"94.4%"}],"good":"high","rank":4,"bars_quote":"With informative checks (attackers pass 10%), coverage verification yielded **94.4% rare-skill accuracy**, versus 8.3% for degree, 47.2% for random and 0% with no checks.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-specialists-api/README.md","studies":[{"id":"sybil-specialists","status":"completed_scripted_development","score":1,"claim":"Coverage verification recovers rare expertise at the cost of some malicious admission in a scripted graph.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-specialists/README.md"},{"id":"sybil-specialists-api","status":"completed_exploratory_model_pilot","score":2,"claim":"The completed controlled comparison describes verifier/admission effects on synthesis within one constructed graph family.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-specialists-api/README.md"},{"id":"sybil-specialists-sonnet","status":"unassessed_registration_coverage","score":null,"claim":"Unassessed at registry discovery.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-specialists-sonnet/README.md"},{"id":"sybil-specialists-opus","status":"implemented_no_model_results_in_repository","score":0,"claim":"Untested: with claude-opus-5-5 as the synthesizer on the Haiku pilot's identical S1 packets, coverage auditing still beats degree auditing on rare-skill accuracy.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-specialists-opus/README.md"}],"contributors":["dmarz"],"runs":2,"done":2,"failed":0,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[158,159,"d"],[268,269,"d"]]},{"id":"sybil-scale","title":"scale","name":"Honey, I Grew the Swarm","program":"sybil","owner":"dmarz","question":"As a swarm under fake-identity attack grows, do more checks keep its answers accurate?","number":"47.2% -> 98.6%","unit":"specialist accuracy, 4 vs 108 checks","line":"At 972 simulated identities, scaling checks with swarm size restored specialist accuracy; random checks did as well. Sonnet and Opus repeated it.","verdict":"result","models":["Claude Haiku 4.5","Claude Sonnet 4.6","Claude Opus 5.5"],"quote":"At 972 identities with informative checks and visible badges, proportional coverage checking (108 checks) achieved 98.6% specialist accuracy versus 47.2% with four checks: +51.4 percentage points, descriptive paired-world 95% interval +38.9 to +62.5.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scale-api/README.md","score":2,"claim":null,"viz":false,"headline":"As a simulated swarm grew, scaling up ID checks kept answers right; random checks did too.","metric":"specialist questions answered correctly","bars":[{"label":"4 checks","value":47.2,"max":100,"text":"47.2%"},{"label":"108 spread checks","value":98.6,"max":100,"text":"98.6%"},{"label":"108 random checks","value":98.6,"max":100,"text":"98.6%"}],"good":"high","rank":4,"bars_quote":"At 972 identities with informative checks and visible badges, proportional coverage checking (108 checks) achieved 98.6% specialist accuracy versus 47.2% with four checks: +51.4 percentage points, descriptive paired-world 95% interval +38.9 to +62.5. Attacker seat share fell from 11.5% to 3.6%. Random checking at the same 108-check budget also achieved 98.6%; this is not evidence that coverage is uniquely superior.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scale-api/README.md","studies":[{"id":"sybil-scale-api","status":"completed_exploratory_model_pilot","score":2,"claim":"The completed controlled sweep describes verification-resource and admission-policy effects on specialist accuracy in one graph family, not an equal-cost coverage advantage.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scale-api/README.md"},{"id":"sybil-scale-sonnet","status":"complete_valid_replication_result_claim_released","score":2,"claim":"In the repeated-fact synthetic environment, the Haiku population-scaling result replicates with claude-sonnet-4-6 on identical assignments: proportional strong checks raise N972 specialist accuracy by +52.8 pp (Haiku +51.4 pp); Sonnet is more accurate than Haiku in 73 of 100 cells, mainly where fabrications are admitted.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scale-sonnet/README.md"},{"id":"sybil-scale-opus","status":"complete_valid_result_claim_released","score":2,"claim":"On identical assignments, claude-opus-5-5 (effort low, no temperature) reproduces the population-scaling direction (proportional strong checks raise N972 specialist accuracy by +100.0 pp) but scores lower than Haiku and Sonnet over the grid because it abstains on 31% of rare fields while giving the fewest wrong values.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scale-opus/README.md"},{"id":"sybil-scale-xl","status":"s1_incomplete_reported_completion_proposed","score":2,"claim":"Whether the sybil-scale-api finding (proportional checking preserves specialist accuracy as the swarm grows) holds from 972 to 8,748 simulated identities with Opus 5.5 synthesis (amendment A1).","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scale-xl/README.md"}],"contributors":["dmarz"],"runs":13,"done":11,"failed":2,"cost":366.15,"calls":5457,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,1,1,1,1,2,3,4,4,4,4,5,4,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,6,0,0,0,0,0,1,3,3,3,3,3,3,3,3,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[137,243,"d"],[140,243,"d"],[142,243,"d"],[245,426,"d"],[271,426,"d"],[280,426,"d"],[281,309,"f"],[311,426,"d"],[320,426,"d"],[321,426,"f"],[468,514,"d"],[471,514,"d"],[472,514,"d"]]},{"id":"sybil-budget","title":"check budget","name":"Fewer, Better Bouncers","program":"sybil","owner":"dmarz","question":"How many identity checks does a swarm need, and how reliable must they be?","number":"7 of 120","unit":"check settings meeting the 90%/5% goal","line":"Only settings where attackers passed checks 10% of the time met the target, and in one case more checks admitted more attackers even as accuracy rose.","verdict":"result","models":["Claude Haiku 4.5","Claude Sonnet 4.6"],"quote":"Only **7 of 120 cells** met the point-mean target; all used the strongest tested checks, with attacker pass probability 10%.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-budget-api/RESULTS.md","score":2,"claim":null,"viz":false,"headline":"7 of 120 setups kept answers right and attackers out; all used the most reliable checks.","metric":"check setups meeting the accuracy+safety goal","bars":[{"label":"setups meeting goal","value":7,"max":120,"text":"7 of 120"}],"good":"high","rank":3,"bars_quote":"Only **7 of 120 cells** met the point-mean target; all used the strongest tested checks, with attacker pass probability 10%.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-budget-api/RESULTS.md","studies":[{"id":"sybil-budget-api","status":"complete_valid_grid_result_worker_stopped_claim_released","score":2,"claim":"In the repeated-fact synthetic environment, the strong-check accuracy benefit replicated; random checking reached the N972 joint mean target at 64 tested checks and coverage at 108. No weaker tested checking strength met that joint target.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-budget-api/README.md"},{"id":"sybil-budget-sonnet","status":"s1_complete_paired_with_haiku","score":2,"claim":"On identical packets, Sonnet 4.6 raises specialist accuracy slightly over Haiku 4.5 (+5.7 pp) without changing admission or the tested 90%/5% engineering frontier.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-budget-sonnet/README.md"}],"contributors":["dmarz"],"runs":6,"done":6,"failed":0,"cost":158.53,"calls":5792,"ridge":[0,0,0,1,2,2,2,3,3,3,3,3,3,3,3,3,3,3,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,3,3,3,3,3,3,3,3,3,3,3,3,3,3,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[20,110,"d"],[30,110,"d"],[47,110,"d"],[246,333,"d"],[251,333,"d"],[254,333,"d"]]},{"id":"sybil-newcomer","title":"newcomers","name":"Sleeper Agents","program":"sybil","owner":"dmarz","question":"Can re-checking trusted members stop attackers who build trust first and strike later?","number":"-63.9 pp","unit":"accuracy lost to sleepers (renewal)","line":"In this synthetic task, attackers who built trust before lying cut specialist accuracy sharply; renewal checks did not beat reputation or random.","verdict":"null","models":["Claude Haiku 4.5","Claude Sonnet 4.6","Claude Opus 5.5"],"quote":"Relative to clean, sleeper accuracy fell 55.6 points for random (interval \u221268.1 to \u221243.1), 63.9 for renewal (\u221273.6 to \u221254.2), and 70.8 for reputation (\u221281.9 to \u221259.7).","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-newcomer-api/reviews/s1-001-post.md","score":2,"claim":null,"viz":false,"headline":"Attackers who earned trust before lying beat every check tried, including re-checking.","metric":"specialist questions answered correctly","bars":[{"label":"random audits","value":26.4,"max":100,"text":"26.4%"},{"label":"re-check trusted","value":20.8,"max":100,"text":"20.8%"},{"label":"reputation audits","value":12.5,"max":100,"text":"12.5%"}],"good":"high","rank":3,"bars_quote":"In the predeclared sixteen-identity sleeper condition at round eight, specialist accuracy was 26.4% with random audits, 20.8% with renewal and 12.5% with reputation audits.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-newcomer-api/README.md","studies":[{"id":"sybil-newcomer-api","status":"complete_valid_inconclusive_primary_allocation_closed","score":2,"claim":"In this synthetic task, sleeper attacks sharply reduced specialist accuracy; renewal checking did not establish its planned improvement over reputation or an advantage over equal-cost random checking.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-newcomer-api/README.md"},{"id":"sybil-newcomer-sonnet","status":"complete_valid_inconclusive_primary_allocation_closed","score":2,"claim":"In this synthetic newcomer task, replacing the Haiku 4.5 synthesizer with Sonnet 4.6 on identical assignments did not materially change outcomes: the sleeper attack still sharply reduced specialist accuracy, the policy ranking was unchanged, and renewal minus reputation stayed inconclusive (+11.1 pp, interval -0.0 to +22.2 pp; +2.8 pp more than Haiku, interval -2.8 to +8.3 pp).","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-newcomer-sonnet/README.md"},{"id":"sybil-newcomer-opus","status":"complete_valid_inconclusive_primary_allocation_closed","score":2,"claim":"In this synthetic newcomer task, an Opus 5.5 synthesizer (effort low) on assignments identical to the Haiku 4.5 and Sonnet 4.6 cohorts did not materially change outcomes: the policy ranking (random > renewal > reputation) was unchanged, renewal minus reputation stayed inconclusive (+9.7 pp, interval -1.4 to +20.8 pp), and the difference from Haiku in that contrast was +1.4 pp (interval -4.2 to +6.9 pp). Accuracy never exceeded the share of truth present in admitted packets.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-newcomer-opus/README.md"}],"contributors":["dmarz"],"runs":9,"done":9,"failed":0,"cost":26.99,"calls":5940,"ridge":[0,0,0,1,2,2,2,3,3,3,3,3,0,0,0,0,0,0,0,0,0,1,1,1,1,3,3,3,3,3,3,3,3,3,3,3,3,3,3,0,0,0,0,0,0,0,0,0,3,3,3,3,3,3,3,3,3,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[20,69,"d"],[29,69,"d"],[47,69,"d"],[133,243,"d"],[158,243,"d"],[160,243,"d"],[301,350,"d"],[302,350,"d"],[303,350,"d"]]},{"id":"sybil-scarcity","title":"scarcity","name":"The Truth Is Outnumbered","program":"sybil","owner":"dmarz","question":"Does fake-identity resistance still work when only a few honest members know the answer?","number":"100.0% -> 4.2%","unit":"accuracy, 81 vs 1 honest source per fact","line":"In this synthetic task, with one honest source per rare fact instead of 81, Opus 5.5 mostly repeated the attacker's fake value; accuracy fell to 4.2%.","verdict":"result","models":["Claude Opus 5.5"],"quote":"| Specialist accuracy (Opus 5.5) | 4.2% | 13.9% | 54.2% | 95.8% | 100.0% |","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scarcity-opus/RESULTS.md","score":2,"claim":"An AI had to work out the truth from reports, some planted by fake accounts. When only one honest source knew the answer, it repeated the fake answer 94% of the time.","viz":true,"headline":"When one honest member knew a fact, the AI mostly repeated the attacker's fake answer.","metric":"rare facts answered correctly (Opus 5.5)","bars":[{"label":"81 honest sources","value":100.0,"max":100,"text":"100%"},{"label":"27 honest sources","value":95.8,"max":100,"text":"95.8%"},{"label":"9 honest sources","value":54.2,"max":100,"text":"54.2%"},{"label":"1 honest source","value":4.2,"max":100,"text":"4.2%"}],"good":"high","rank":4,"bars_quote":"| Specialist accuracy (Opus 5.5) | 4.2% | 13.9% | 54.2% | 95.8% | 100.0% |","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scarcity-opus/RESULTS.md","studies":[{"id":"sybil-scarcity-plan","status":"prospective_plan_published_explicit_no_start","score":0,"claim":"Proposed: reducing truthful carriers at fixed population, graph and selection lowers specialist accuracy.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scarcity-plan/README.md"},{"id":"sybil-scarcity-opus","status":"completed_s1_verified","score":2,"claim":"Reducing the truthful carriers of each rare fact from 81 to 1, at fixed population, graph, audits and admission, lowers the specialist accuracy of an Opus 5.5 synthesizer.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scarcity-opus/README.md"},{"id":"sybil-scarcity-synth","status":"prepared_unrun","score":0,"claim":"A stated evidence rule, or higher reasoning effort, reduces how often an Opus 5.5 synthesizer answers with a repeated fabricated value when a rare fact has one truthful carrier, at a measured cost in accuracy when truth is plentiful.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scarcity-synth/README.md"},{"id":"sybil-scarcity-xmodel","status":"closed_all_configurations_stopped_at_q0","score":0,"claim":"On packets byte-identical to sybil-scarcity-opus, a synthesizer other than Opus 5.5 (qwen3.7-flash without reasoning; gpt-6-sol at reasoning effort low) also loses specialist accuracy when each rare fact has one truthful carrier instead of 81.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-scarcity-xmodel/README.md"}],"contributors":["dmarz"],"runs":20,"done":13,"failed":7,"cost":178.79,"calls":1990,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,4,1,1,1,1,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,4,1,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,4,1,2,3,0,0,0,3,0,0,0,0,0,0,0,0,0,0],"spans":[[368,369,"d"],[369,369,"d"],[369,371,"d"],[371,412,"d"],[505,506,"d"],[506,506,"d"],[506,508,"d"],[508,522,"f"],[527,528,"f"],[889,890,"d"],[890,890,"d"],[890,891,"f"],[894,906,"f"],[906,906,"f"],[907,908,"d"],[908,908,"d"],[908,910,"f"],[936,936,"d"],[936,936,"d"],[936,937,"f"]]},{"id":"sybil-split","title":"identity split","name":"Being John Malkovich","program":"sybil","owner":"dmarz","question":"Does splitting one attacker's fixed resources over more fake identities cause more harm?","number":"7.6% -> 56.2%","unit":"wrong answers: 1 vs 27 fake identities","line":"With reliable checks aimed at well-connected members, one attacker split over 27 identities caused far more wrong answers; spread checks: under 8%.","verdict":"result","models":["Claude Opus 5.5"],"quote":"| `degree`, 12 checks | 7.6% | 17.4% | 43.8% | 56.2% | +48.6 (+35.4 to +61.1) |","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-split-opus/RESULTS.md","score":2,"claim":null,"viz":false,"headline":"With checks aimed at hub members, one attacker posing as 27 people caused 7x more errors.","metric":"wrong answers on rare questions","bars":[{"label":"1 fake identity","value":7.6,"max":100,"text":"7.6%"},{"label":"3 fake identities","value":17.4,"max":100,"text":"17.4%"},{"label":"9 fake identities","value":43.8,"max":100,"text":"43.8%"},{"label":"27 fake identities","value":56.2,"max":100,"text":"56.2%"}],"good":"low","rank":4,"bars_quote":"| `degree`, 12 checks | 7.6% | 17.4% | 43.8% | 56.2% | +48.6 (+35.4 to +61.1) |","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-split-opus/RESULTS.md","studies":[{"id":"sybil-split-opus","status":"complete_exploratory","score":2,"claim":"Splitting one attacker's fixed 27 report rows, 27 attachment edges and 27 verification attempts from 1 to 27 identities raises rare-skill wrong answers of an Opus 5.5 synthesizer more under degree-based than under coverage-based admission checks.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-split-opus/README.md"},{"id":"sybil-split-xmodel","status":"closed_qualification_failed_three_configurations_s1_unrun","score":0,"claim":"On the byte-identical packets of sybil-split-opus, splitting one attacker's fixed resources from 1 to 27 identities raises rare-skill wrong answers more under degree-based than under coverage-based checks for qwen/qwen3.7-flash (reasoning disabled) and gpt-6-sol (reasoning effort low, and none as the one pre-registered follow-up), each reported separately.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/sybil-split-xmodel/README.md"}],"contributors":["dmarz"],"runs":13,"done":10,"failed":3,"cost":46.4,"calls":2932,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,4,1,1,1,1,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,1,3,1,3,0,0,0,0,0,0,1,3,0,0,0,0,0,0],"spans":[[425,427,"d"],[427,427,"d"],[427,428,"d"],[428,465,"d"],[887,894,"d"],[894,895,"d"],[895,896,"f"],[903,910,"d"],[910,910,"d"],[910,911,"f"],[954,961,"d"],[961,961,"d"],[961,962,"f"]]},{"id":"shadow-factory","title":"split factory","name":"Rate Limited Edition","program":"sybil","owner":"shadow","question":"Does a Sybil identity-splitting result hold up with a different model and setup?","number":null,"unit":null,"line":"A re-test of an identity-splitting attack with Claude Sonnet stopped at provider limits before enough paired worlds finished to read any effect.","verdict":"not-qualified","models":["anthropic/claude-sonnet-4.6"],"quote":"It has only 2/48 complete roots despite 88 valid main calls: its narrow complete-case bootstrap is not meaningful precision.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/FINDING.md","score":1,"claim":null,"viz":false,"headline":"A rerun of an identity-splitting attack hit provider limits before enough runs finished","metric":"no result measured","bars":[],"good":"none","rank":1,"bars_quote":"It has only 2/48 complete roots despite 88 valid main calls: its narrow complete-case bootstrap is not meaningful precision.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/FINDING.md","studies":[{"id":"shadow-factory-split-sonnet-four-checks-native-json","status":"unrun","score":0,"claim":"Prospective exploratory sensitivity only; no outcome observed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/plans/split-sonnet-four-checks-native-json.md"},{"id":"shadow-factory-split-sonnet-four-checks-or-json","status":"unrun","score":0,"claim":"Prospective exploratory sensitivity only; no outcome observed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/plans/split-sonnet-four-checks-or-json.md"},{"id":"shadow-factory-split-sonnet-four-checks-or","status":"lead","score":0,"claim":"anthropic/claude-sonnet-4.6, ring2 internal links, pass=0.1, checks=4: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-four-checks-or/FINDING.md"},{"id":"shadow-factory-split-sonnet-four-checks","status":"lead","score":0,"claim":"claude-sonnet-4-6, ring2 internal links, pass=0.1, checks=4: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-four-checks/FINDING.md"},{"id":"shadow-factory-split-sonnet-linked-strong-native-json","status":"unrun","score":0,"claim":"Prospective exploratory sensitivity only; no outcome observed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/plans/split-sonnet-linked-strong-native-json.md"},{"id":"shadow-factory-split-sonnet-linked-strong-or-json","status":"lead","score":1,"claim":"anthropic/claude-sonnet-4.6, ring2 internal links, pass=0.1, checks=12: paired splitting difference-in-differences +0.5000; exploratory same-family sensitivity, not a general defense claim.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-linked-strong-or-json/FINDING.md"},{"id":"shadow-factory-split-sonnet-linked-strong-or","status":"lead","score":0,"claim":"anthropic/claude-sonnet-4.6, ring2 internal links, pass=0.1, checks=12: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-linked-strong-or/FINDING.md"},{"id":"shadow-factory-split-sonnet-linked-strong","status":"lead","score":0,"claim":"claude-sonnet-4-6, ring2 internal links, pass=0.1, checks=12: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-linked-strong/FINDING.md"},{"id":"shadow-factory-split-sonnet-linked-weak-native-json","status":"unrun","score":0,"claim":"Prospective exploratory sensitivity only; no outcome observed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/plans/split-sonnet-linked-weak-native-json.md"},{"id":"shadow-factory-split-sonnet-linked-weak-or-json","status":"lead","score":0,"claim":"anthropic/claude-sonnet-4.6, ring2 internal links, pass=0.9, checks=12: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-linked-weak-or-json/FINDING.md"},{"id":"shadow-factory-split-sonnet-linked-weak-or","status":"lead","score":0,"claim":"anthropic/claude-sonnet-4.6, ring2 internal links, pass=0.9, checks=12: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-linked-weak-or/FINDING.md"},{"id":"shadow-factory-split-sonnet-linked-weak","status":"lead","score":0,"claim":"claude-sonnet-4-6, ring2 internal links, pass=0.9, checks=12: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-linked-weak/FINDING.md"},{"id":"shadow-factory-split-sonnet-no-links-strong-native-json","status":"lead","score":0,"claim":"claude-sonnet-4-6, none internal links, pass=0.1, checks=12: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-no-links-strong-native-json/FINDING.md"},{"id":"shadow-factory-split-sonnet-no-links-strong-or-json","status":"unrun","score":0,"claim":"Prospective exploratory sensitivity only; no outcome observed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/plans/split-sonnet-no-links-strong-or-json.md"},{"id":"shadow-factory-split-sonnet-no-links-strong-or","status":"lead","score":0,"claim":"anthropic/claude-sonnet-4.6, none internal links, pass=0.1, checks=12: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-no-links-strong-or/FINDING.md"},{"id":"shadow-factory-split-sonnet-no-links-strong","status":"lead","score":0,"claim":"claude-sonnet-4-6, none internal links, pass=0.1, checks=12: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-no-links-strong/FINDING.md"},{"id":"shadow-factory-split-sonnet-no-links-weak-native-json","status":"unrun","score":0,"claim":"Prospective exploratory sensitivity only; no outcome observed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/plans/split-sonnet-no-links-weak-native-json.md"},{"id":"shadow-factory-split-sonnet-no-links-weak-or-json","status":"unrun","score":0,"claim":"Prospective exploratory sensitivity only; no outcome observed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/plans/split-sonnet-no-links-weak-or-json.md"},{"id":"shadow-factory-split-sonnet-no-links-weak-or","status":"lead","score":0,"claim":"anthropic/claude-sonnet-4.6, none internal links, pass=0.9, checks=12: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-no-links-weak-or/FINDING.md"},{"id":"shadow-factory-split-sonnet-no-links-weak","status":"lead","score":0,"claim":"claude-sonnet-4-6, none internal links, pass=0.9, checks=12: No interpretable treatment contrast; qualification/transport stopped before complete paired outcomes.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/factory/results/split-sonnet-no-links-weak/FINDING.md"}],"contributors":["shadow"],"runs":15,"done":0,"failed":15,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,9,10,10,13,13,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[743,768,"f"],[743,768,"f"],[743,768,"f"],[743,768,"f"],[743,768,"f"],[744,768,"f"],[744,768,"f"],[744,768,"f"],[744,768,"f"],[744,768,"f"],[762,768,"f"],[762,768,"f"],[762,768,"f"],[888,888,"f"],[888,888,"f"]]},{"id":"trust-credit","title":"trust credit","name":"Innocence by Association","program":"sybil","owner":"dmarz","question":"Does passing trust from a checked identity to its neighbours let more attackers in?","number":"+20.75 seats","unit":"extra attacker seats from shared credit","line":"With strong checks in this simulator, more checks admitted more attackers only when credit flowed to neighbours; direct credit had more at low budget.","verdict":"result","models":["Qwen3.7 Flash","gpt-6-luna"],"quote":"**+20.75 seats** (95% root-bootstrap interval +17.33 to +24.21; 10,000 draws, seed 20261004; 24 of 24 roots, all positive).","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/trust-credit-qwen/RESULTS.md","score":1,"claim":null,"viz":false,"headline":"When a checked member's trust passed to its neighbours, more checks let more attackers in.","metric":"attacker seats out of 162 (strong checks)","bars":[{"label":"32 checks","value":4.38,"max":162,"text":"4.4"},{"label":"64 checks","value":14.29,"max":162,"text":"14.3"},{"label":"108 checks","value":15.92,"max":162,"text":"15.9"}],"good":"low","rank":3,"bars_quote":"| 10% (strong) | `propagated` | 4.38 (2.7%) | 14.29 (8.8%) | 15.92 (9.8%) |","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/trust-credit-qwen/RESULTS.md","studies":[{"id":"trust-credit-qwen","status":"complete_valid_exploratory_result","score":1,"claim":"In this simulator (324 scripted identities, coverage audit, strong checks), the rise in attacker seats between 32 and 108 checks appears when pass credit is propagated to neighbours and not when it is placed on the passed identity alone: contrast +20.75 seats (17.33 to 24.21), positive in 24 of 24 roots.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/trust-credit-qwen/README.md"}],"contributors":["dmarz"],"runs":8,"done":8,"failed":0,"cost":0.26,"calls":1056,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,4,1,0,0,0,0,0,0,0,0,0,0,4,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[510,510,"d"],[510,510,"d"],[510,511,"d"],[511,514,"d"],[583,584,"d"],[584,584,"d"],[584,584,"d"],[584,591,"d"]]},{"id":"verify-cost","title":"verify cost","name":"No Table Required","program":"sybil","owner":"dmarz","question":"Does a cost table help a model decide whether to check a report or explore?","number":"575 of 576","unit":"best choices, with or without a table","line":"gpt-6-luna chose the lowest-loss action almost every time with or without a cost table, so the table showed no effect; Qwen3.7 Flash did not qualify.","verdict":"null","models":["gpt-6-luna"],"quote":"Observed: `gpt-6-luna` at low reasoning effort chooses the minimum-loss action in 575 of 576 one-step decisions across twelve risk and cost cases in which the best action changes, in both renderings.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/verify-cost-qwen/RESULTS.md","score":1,"claim":null,"viz":false,"headline":"The AI already chose well from costs written in plain prose; a table added nothing.","metric":"best choices made, of 288","bars":[{"label":"costs in prose","value":288,"max":288,"text":"288 of 288"},{"label":"costs in a table","value":287,"max":288,"text":"287 of 288"}],"good":"high","rank":2,"bars_quote":"288 of 288 optimal choices with the prose, 287 of 288 with the table.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/verify-cost-qwen/RESULTS.md","studies":[{"id":"verify-cost-qwen","status":"completed_reported_at_ceiling","score":1,"claim":"In a one-step choice between checking a report and exploring an unknown cell, with the outcome, probability and cost of each action stated and no expected loss printed, gpt-6-luna at low reasoning effort chooses the minimum-loss action in both an explicit-prose and an action-consequence-table rendering (575 of 576), so the table-minus-prose expected regret is +0.00087 (95% interval -0.00093 to +0.00266): no measurable effect of the table, at ceiling.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/verify-cost-qwen/README.md"}],"contributors":["dmarz"],"runs":10,"done":8,"failed":2,"cost":0.06,"calls":648,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,3,0,0,0,0,0,0,0,3,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,4,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[519,519,"d"],[519,519,"d"],[519,519,"f"],[569,569,"d"],[569,569,"d"],[569,569,"f"],[892,892,"d"],[892,892,"d"],[892,893,"d"],[893,897,"d"]]},{"id":"quota-splitting","title":"quota splitting","name":"Please, Sir, I Want More","program":"sybil","owner":"dmarz","question":"Under a per-identity compute quota, will an AI agent create extra identities to get more?","number":null,"unit":null,"line":"Planned and built but never run; no model call has been made.","verdict":"proposed","models":[],"quote":null,"source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/quota-splitting/HANDOVER.md","score":0,"claim":null,"viz":false,"headline":"Built and ready to test, but not run yet, so there is no result.","metric":"","bars":[],"good":"none","rank":1,"bars_quote":null,"bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/quota-splitting/HANDOVER.md","studies":[{"id":"quota-splitting","status":"planned_unrun","score":0,"claim":"Under a stated per-identity compute quota an Opus 5.5 lead agent creates more subagent identities than it creates for the same job with no quota, and so draws more than one quota from a pool shared with three other teams.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/quota-splitting/README.md"}],"contributors":["dmarz"],"runs":0,"done":0,"failed":0,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[]},{"id":"influence","title":"external influence","name":"How to Win Agents and Influence Swarms","program":"influence","owner":"vishesh","question":"Can independent checks keep an agent team from being steered by edited evidence?","number":"19/50","unit":"correct team choices, Haiku pilot","line":"The 50 assignments span clean and attack conditions on only three task roots. The local cohort had 48 valid finals and 2 timeouts. Correctness mixes evidence interpretation, arithmetic and final decisions; these totals are not attack success rates.","verdict":"adverse","models":["claude-haiku-4-5-20251001","Qwen3 0.6B (local)","Qwen3 1.7B (local)"],"quote":"Historical Haiku 38908910: 19/50 correct, 50/50 valid, 28 harmful targets,mean regret 5.701.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/9bd914def1c0830023db9342c10627a788abfdbc/researchers/vishesh/notes/external-influence-v2/local-agents/reviews/S1-post.md","score":1,"claim":"In six selected historical attack cases, nine-agent teams picked the target despite correct check outputs.","viz":true,"headline":"Historical nine-agent teams scored 19/50; a local-model cohort scored 11/50 across mixed conditions","metric":"correct team decisions out of 50","bars":[{"label":"Claude Haiku team","value":19,"max":50,"text":"19 of 50"},{"label":"Small local model team","value":11,"max":50,"text":"11 of 50"}],"good":"high","rank":4,"bars_quote":"Historical Haiku 38908910: 19/50 correct, 50/50 valid, 28 harmful targets,mean regret 5.701. Local: 11/50 correct","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/9bd914def1c0830023db9342c10627a788abfdbc/researchers/vishesh/notes/external-influence-v2/local-agents/reviews/S1-post.md","studies":[{"id":"influence-v1","status":"scripted_complete_native_clean_qualification_only","score":0,"claim":"Untested efficacy question: decision-focused checking can outperform random checking against externally manipulated evidence.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/actual-experiments/external-influence/README.md"},{"id":"influence-v2","status":"completed_exploratory","score":1,"claim":"The hosted fixture comparison exposed poor use of corrective evidence and no reliable verification advantage; it does not establish general attack robustness.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/external-influence-v2/README.md"},{"id":"influence-v2-local","status":"completed_failed_qualifications_repair_planned","score":1,"claim":"The unchanged local 0.6B and 1.7B replacements failed bounded workflow qualification; attack resistance remains untested.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/external-influence-v2/README.md"},{"id":"influence-quality-repair","status":"archived_offline_proposal","score":0,"claim":"Untested efficacy question: an explicit ledger and enforced final score rule makes checking affect the committed decision.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/influence-swarms/README.md"},{"id":"influence-scenario","status":"e1_terminal_scientifically_reviewed_r2_preparing","score":1,"claim":"A peer-influence effect is not established in the authored E1 procurement comparison.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/influence-swarms/scenario/README.md"},{"id":"influence-approval-review-d2","status":"diagnostic_complete_repair_failed_main_unrun","score":1,"claim":"Targeted approval-review instructions were insufficient to repair this six-case procurement workflow.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/influence-swarms/scenario/RESULTS-D2.md"},{"id":"influence-candidate-checks-d3","status":"diagnostic_complete_repair_failed_transfer_unrun","score":1,"claim":"Schema-required candidate coverage did not repair factual checks or final decisions on six frozen procurement cases.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/influence-swarms/scenario/RESULTS-D3.md"}],"contributors":["vishesh"],"runs":13,"done":8,"failed":5,"cost":0,"calls":242,"ridge":[0,0,0,0,0,0,2,3,4,7,7,1,2,2,2,0,0,0,0,0,0,0,0,0,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,1,0,1,0,0],"spans":[[41,63,"d"],[43,63,"d"],[47,64,"d"],[52,64,"d"],[57,64,"f"],[57,94,"d"],[62,64,"d"],[81,90,"d"],[151,157,"d"],[838,839,"f"],[920,920,"f"],[970,970,"f"],[984,984,"f"]],"film":{"src":"media/influence.mp4","poster":"media/influence-poster.jpg","loop":"media/influence-loop.mp4","duration":24.0,"order":5}},{"id":"phantom-coast","title":"phantom coast","name":"Trust Issues","program":"influence","owner":"vishesh","question":"Do explicit decision rules help a model decide when to trust or verify a report?","number":"28 -> 21 /32","unit":"best choices when reports were reliable","line":"Explicit decision rules lowered average regret in this synthetic mapping task, but made the model worse at trusting reliable reports.","verdict":"result","models":["typesafe/jev-1.13 (Jev)"],"quote":"On 32 paired layouts, explicit decision-contract instructions reduced expected regret 0.1578 per decision but optimal reliable-report choices regressed from 28/32 to 21/32.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/evidence-metadata.json","score":1,"claim":"Once an AI knew exactly how its choices would be scored, it checked more of the unreliable reports. It also started second-guessing the reliable ones.","viz":true,"headline":"Explicit decision rules cut a model's average mistakes but made it doubt reliable reports","metric":"best choices when the report was reliable","bars":[{"label":"Without rules","value":28,"max":32,"text":"28 of 32"},{"label":"With decision rules","value":21,"max":32,"text":"21 of 32"}],"good":"high","rank":3,"bars_quote":"On 32 paired layouts, explicit decision-contract instructions reduced expected regret 0.1578 per decision but optimal reliable-report choices regressed from 28/32 to 21/32.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/evidence-metadata.json","studies":[{"id":"phantom-coast","status":"qualification_passed_main_not_reconciled","score":1,"claim":"Native Jev passed the clean synthetic mapping screen; the history-persistence effect remains unestablished.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/README.md"},{"id":"phantom-coast-pc2","status":"complete_valid_adverse_exploratory_result","score":1,"claim":"In eight synthetic roots, learned acquisition repeatedly revisited cells; uniform acquisition supplied much broader direct coverage, and the targeted audit did not meet the planned mean benefit threshold.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/pc2/README.md"},{"id":"phantom-coast-pc3","status":"complete_exploratory_result","score":1,"claim":"In eight fresh synthetic roots, the explicit coverage guard changed misleading-team decision loss by -2.26 percentage points under the declared abstention cost.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/pc3/README.md"},{"id":"phantom-coast-pc4","status":"complete_exploratory_result","score":1,"claim":"On 32 paired synthetic worlds, checking all four reports improved full-corruption loss 4.30 percentage points versus uniform but worsened benign loss 1.52 points and missed the 5.56-point practical target.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/pc4/README.md"},{"id":"phantom-coast-pc5","status":"complete_exploratory_result","score":1,"claim":"On 32 paired layouts, explicit decision-contract instructions reduced expected regret 0.1578 per decision but optimal reliable-report choices regressed from 28/32 to 21/32.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/pc5/README.md"},{"id":"phantom-coast-pc6","status":"parked_before_native_dispatch","score":0,"claim":"The proposed action-consequence table effect remains untested; PC6 was parked before dispatch for insufficient decision value.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/pc6/README.md"},{"id":"phantom-coast-pc7","status":"offline_prototype_native_not_admitted","score":0,"claim":"A model benefit on finite-history verification is untested; only an exact offline controller and prior-trace audit are complete.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/pc7/README.md"},{"id":"phantom-coast-pc8","status":"qualification_failed_scientifically_reviewed_parked","score":1,"claim":"This model configuration failed structured/protected-evidence qualification; its prose advantage over the frozen parser is removable on the saved authored corpus.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/pc8/README.md"},{"id":"phantom-coast-pc9","status":"native_Q0_completed_qualification_failed_pilot_held","score":0,"claim":"Population misinformation spread and acquisition feedback remain untested; Q0 clean-task qualification failed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/pc9/README.md"},{"id":"phantom-coast-pc10","status":"native_paired_diagnostic_complete_no_improvement_qualification_failed_parked","score":1,"claim":"Explicit individual-control wording did not change inspection choices on the six tested action pairs; broader effects remain unknown.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/pc10/README.md"},{"id":"phantom-coast-pc11","status":"native_qualification_passed_pilot_complete_no_spread_in_one_world","score":1,"claim":"The requested GPT-6 Sol configuration passed the eight-case qualification; in one matched four-arm world a false report persisted in its recipient without spreading to nine unseeded agents or changing inspections.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/phantom-coast/pc11/README.md"}],"contributors":["vishesh"],"runs":11,"done":11,"failed":0,"cost":5.61,"calls":0,"ridge":[0,0,0,0,0,0,0,1,1,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,1,1,1,1,1,1,1,1,1,1,1,0,0,1,1,1,1,0,0,0,2,1,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[50,50,"d"],[52,66,"d"],[204,204,"d"],[206,273,"d"],[294,294,"d"],[296,308,"d"],[334,334,"d"],[335,342,"d"],[350,350,"d"],[351,353,"d"],[919,923,"d"]]},{"id":"quorum-of-mirrors","title":"quorum of mirrors","name":"Copy That","program":"influence","owner":"vishesh","question":"Can a model count copied reports once instead of letting repetition win?","number":"0/8","unit":"copy-aware decisions right (needed 7/8)","line":"When copied reports outnumbered independent ones, the model followed the report majority every time; a simple dedup rule already solves this task.","verdict":"not-qualified","models":["typesafe/jev-1.13 (Jev)","anthropic/claude-sonnet-4.6"],"quote":"The repaired **Q1 context screen completed all 16 calls with valid responses, but got 0/8 full-lineage decisions right** (required: at least 7/8).","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/decision-models/quorum-of-mirrors/README.md","score":1,"claim":"Seven of nine reports were copies of a single source. Even when told to count copies only once, the AI sided with the copies in 8 of 8 cases.","viz":true,"headline":"When copied reports outnumbered real sources, the model sided with the copies every time","metric":"copy-aware decisions right, out of 8","bars":[{"label":"Needed to pass","value":7,"max":8,"text":"7 of 8"},{"label":"Model got right","value":0,"max":8,"text":"0 of 8"}],"good":"high","rank":3,"bars_quote":"The repaired **Q1 context screen completed all 16 calls with valid responses, but got 0/8 full-lineage decisions right** (required: at least 7/8).","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/decision-models/quorum-of-mirrors/README.md","studies":[{"id":"quorum-of-mirrors","status":"sp02_reviewed","score":1,"claim":"Quote-only evidence selection plus deterministic normalization passes authored qualification; field reliability and swarm efficacy remain untested.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/decision-models/quorum-of-mirrors/README.md"}],"contributors":["vishesh"],"runs":6,"done":1,"failed":5,"cost":0,"calls":0,"ridge":[0,0,0,0,1,2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,1,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,1],"spans":[[31,37,"f"],[33,37,"d"],[218,221,"f"],[285,287,"f"],[954,955,"f"],[995,996,"f"]]},{"id":"discussion-dose","title":"discussion dose","name":"Total Recall","program":"deliberation","owner":"dmarz","question":"Does public discussion protect an agent team from false memories passed down to it?","number":"6/6","unit":"false memories accepted, every model","line":"Claude Haiku, Sonnet and Opus each accepted all six planted false memories; the main test of discussion was cut short by API limits and is unmeasured.","verdict":"not-qualified","models":["Claude Haiku 4.5","Claude Sonnet 4.6","Claude Opus 5.5"],"quote":"| Inherited false fact | 6/6 | 6/6 | 6/6 | **6 locally justified, ground-truth-wrong answers** |","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/d1-opus/RESULTS.md","score":2,"claim":"We planted six false facts in the memory a new AI agent inherits. Small models and frontier models alike accepted all six.","viz":true,"headline":"Haiku, Sonnet and Opus all trusted all six planted false memories; main test cut short.","metric":"planted false memories taken as true","bars":[{"label":"Haiku 4.5","value":6,"max":6,"text":"6 of 6"},{"label":"Sonnet 4.6","value":6,"max":6,"text":"6 of 6"},{"label":"Opus 5.5","value":6,"max":6,"text":"6 of 6"}],"good":"low","rank":2,"bars_quote":"| Inherited false fact | 6/6 | 6/6 | 6/6 | **6 locally justified, ground-truth-wrong answers** |","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/d1-opus/RESULTS.md","studies":[{"id":"discussion-dose","status":"completed_qualification_only","score":1,"claim":"The six-world corrected qualification is executable and exposes a memory-coverage defect; it does not estimate discussion protection.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/README.md"},{"id":"discussion-dose-v2","status":"completed_exploratory_with_invalids","score":1,"claim":"More difficult source conflicts enable a descriptive discussion-dose comparison.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/README.md"},{"id":"discussion-dose-v2-private-control","status":"completed_exploratory_with_invalids","score":1,"claim":"The small board-versus-private diagnostic reports a descriptive contrast; an incremental protective discussion effect remains unresolved.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/PRIVATE-CONTROL.md"},{"id":"discussion-dose-v3","status":"completed_failed_qualification","score":1,"claim":"The completed V3 diagnostic failed clean competence and exposed inherited-memory failure; discussion efficacy remains unestablished.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/benchmark-v3/README.md"},{"id":"discussion-v3-resample","status":"completed_failed_qualification","score":1,"claim":"On this model and contract, repeated probes preserved checkpoint votes; the failed clean gate prevents an attack-protection claim.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/RESAMPLE-CONTROL.md"},{"id":"discussion-dose-v3-d1-haiku","status":"completed_failed_diagnostic_eligibility","score":1,"claim":"This model fails the declared clean development decision gates; it also accepts all six locally supported false memory facts.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/benchmark-v3/README.md"},{"id":"discussion-dose-v3-d1-sonnet","status":"completed_failed_diagnostic_eligibility","score":1,"claim":"This model fails the declared clean development decision gates; it also accepts all six locally supported false memory facts.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/benchmark-v3/README.md"},{"id":"discussion-dose-v3-d2","status":"proposed_not_started","score":0,"claim":"Untested: compact complete facts and a smaller response contract may repair decision errors seen in D1.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/benchmark-v3/D2-PLAN.md"},{"id":"discussion-dose-v3-d2-a1","status":"completed_exploratory","score":1,"claim":"On a compact fact table with a one-field answer, Opus 5.5 answers the six D1 worlds without error (6/6 decisions, 18/18 single-option feasibility checks) while Sonnet 4.6 and Haiku 4.5 each score 5/6 and 16/18 and miss the same two sum-over-budget options.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/benchmark-v3/d2/PLAN.md"},{"id":"discussion-v3-d1-opus","status":"fresh_gate_passed_model_level_only","score":2,"claim":"Under adaptive thinking at effort high with no temperature, Claude Opus 5.5 makes evidence-justified choices on clean full-evidence D1-style decisions where Haiku 4.5 and Sonnet 4.6 failed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/d1-opus/README.md"},{"id":"discussion-v3-opus","status":"planned","score":0,"claim":"With Claude Opus 5.5 (adaptive thinking, effort high, no temperature) the v3 swarm benchmark passes its unchanged qualification gate, and on 24 fresh worlds public discussion reduces false-memory inheritance relative to reports-only.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/discussion-dose/v3-opus/README.md"}],"contributors":["dmarz"],"runs":14,"done":11,"failed":3,"cost":53.86,"calls":1707,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,1,2,2,2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,2,1,1,0,2,2,3,1,1,1,1,1,1,1,1,1,1,1,1,1,2,1,1,1,1,0,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[81,102,"f"],[88,102,"d"],[273,275,"d"],[275,275,"d"],[281,288,"d"],[306,306,"d"],[306,308,"d"],[306,307,"d"],[315,317,"d"],[315,316,"d"],[316,318,"d"],[324,403,"d"],[403,429,"f"],[439,532,"f"]],"film":{"src":"media/discussion-dose.mp4","poster":"media/discussion-dose-poster.jpg","loop":"media/discussion-dose-loop.mp4","duration":119.4,"order":2}},{"id":"right-dissenter","title":"right dissenter","name":"Twelve Angry Agents","program":"deliberation","owner":"vishesh","question":"Should a lone dissenter earn an independent check only when its evidence warrants it?","number":"24 vs 30 /60","unit":"correct: gated dissent vs always-check","line":"Letting a dissenter earn a check only on evidence saved checks but gave fewer correct decisions than always checking, in this synthetic cohort.","verdict":"adverse","models":["typesafe/jev-1.13 (Jev)"],"quote":"Q1 qualified the clarified model; in S1 the evidence gate scored **24/60**, versus **30/60** for always-check, with **31 versus 40 checks**.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/dissent/README.md","score":1,"claim":"We let a lone AI dissenter pause its group only if it brought evidence. Its warnings that things were getting worse got through; its evidence that things had improved never did.","viz":true,"headline":"Letting a dissenter trigger a check only with evidence saved checks but cost right answers","metric":"correct decisions out of 60","bars":[{"label":"Always check","value":30,"max":60,"text":"30 of 60"},{"label":"Check only on evidence","value":24,"max":60,"text":"24 of 60"}],"good":"high","rank":3,"bars_quote":"Q1 qualified the clarified model; in S1 the evidence gate scored **24/60**, versus **30/60** for always-check, with **31 versus 40 checks**.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/dissent/README.md","studies":[{"id":"right-dissenter","status":"exploratory_complete_adverse_comparison","score":1,"claim":"On this fixed synthetic cohort, evidence-gated dissent saved nine checks but produced six fewer correct decisions than always-check, including two missed favorable recovery events.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/dissent/README.md"},{"id":"right-dissenter-rd4","status":"exploratory_complete_transport_failures_retained","score":1,"claim":"On this fixed synthetic cohort, symmetric stop/resume wording tied original admission at 87/96 correct; always-check reached 88/96 with fewer logical model calls. Resolved closure retention worked in the observed cases.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/dissent/rd4/README.md"},{"id":"right-dissenter-rd5","status":"completed_adverse_screen_context_transfer_gap","score":1,"claim":"In this native pilot, a fixed late reserve scored 6/24 correct versus 8/24 for memory alone; four late-resume interpretation misses limit attribution to allocation alone.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/dissent/rd5/README.md"},{"id":"right-dissenter-rd6","status":"q0_a2_completed_qualification_failed_d0_unrun_closed","score":1,"claim":"On 18 authored qualification requests, Jev correctly handled all clean and conflicting-current controls but chose PROCEED on all three expired favorable observations; qualification failed.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/dissent/reopening/README.md"}],"contributors":["vishesh"],"runs":10,"done":8,"failed":2,"cost":0.02,"calls":0,"ridge":[0,0,0,0,0,0,0,0,1,1,0,1,1,1,0,0,0,0,0,0,0,0,0,0,1,1,0,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,1,2,2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[52,62,"d"],[74,74,"d"],[76,78,"d"],[84,84,"d"],[155,155,"d"],[159,161,"f"],[173,175,"f"],[179,180,"d"],[855,873,"d"],[867,873,"d"]],"film":{"src":"media/right-dissenter.mp4","poster":"media/right-dissenter-poster.jpg","loop":"media/right-dissenter-loop.mp4","duration":35.2,"order":3}},{"id":"private-judgments","title":"private judgments","name":"Poker Faces","program":"deliberation","owner":"dmarz","question":"Do teams decide better when members keep their first answers private before discussing?","number":"1.00 vs 1.00","unit":"team success: private vs public answers","line":"With Opus 5.5, every team decided correctly whether first answers were private or public; the task was too easy to show a difference.","verdict":"null","models":["Claude Opus 5.5"],"quote":"Team success was 1.00 for every communicating arm in every regime, in both stages.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/soc07-private-judgments/reviews/s1l-post.md","score":0,"claim":"AI teams reached the right answer every time, whether or not members saw each other's first guesses. The task was too easy to tell the two setups apart.","viz":true,"headline":"Teams always got it right whether first answers were private or public; too easy to tell.","metric":"team success rate","bars":[{"label":"public first answers","value":1.0,"max":1,"text":"100%"},{"label":"private first answers","value":1.0,"max":1,"text":"100%"}],"good":"high","rank":1,"bars_quote":"Team success was 1.00 for every communicating arm in every regime, in both stages.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/soc07-private-judgments/reviews/s1l-post.md","studies":[{"id":"soc07-private-judgments","status":"implemented_no_model_results_in_repository","score":0,"claim":"Untested efficacy question: keeping first judgments private improves team decisions without suppressing valid correction.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/soc07-private-judgments/README.md"},{"id":"soc07-private-judgments-v2","status":"implemented_no_model_results_in_repository","score":0,"claim":"Untested: on a task where first answers disagree, keeping first judgments private changes team decision accuracy relative to publishing them.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/soc07-private-judgments-v2/README.md"}],"contributors":["dmarz"],"runs":11,"done":9,"failed":2,"cost":37.93,"calls":4788,"ridge":[0,0,0,0,0,0,0,1,1,1,1,1,1,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,2,1,0,0,1,1,2,1,1,1,1,1,1,1,2,1,1,1,1,1,1,1,1,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[48,56,"d"],[68,76,"d"],[103,103,"f"],[250,257,"d"],[262,263,"f"],[287,294,"d"],[299,305,"d"],[312,312,"d"],[313,347,"d"],[347,391,"d"],[398,406,"d"]]},{"id":"false-alarm-cascade","title":"false alarm cascade","name":"Unringing the Bell","program":"deliberation","owner":"dmarz","question":"Does a retracted false alarm keep an AI team avoiding a resource that was safe?","number":null,"unit":null,"line":"Planned and built but never run; no model call has been made.","verdict":"proposed","models":[],"quote":null,"source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/false-alarm-cascade/HANDOVER.md","score":0,"claim":null,"viz":false,"headline":"Built and ready to test, but not run yet, so there is no result.","metric":"","bars":[],"good":"none","rank":1,"bars_quote":null,"bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/false-alarm-cascade/HANDOVER.md","studies":[{"id":"false-alarm-cascade","status":"planned_unrun","score":0,"claim":"A false honeypot alarm on a real resource, raised and then retracted by one team member, leaves five gpt-6-sol agents using that resource less in later rounds than in the same world without the alarm (Opus rungs not run).","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/false-alarm-cascade/README.md"}],"contributors":["dmarz"],"runs":0,"done":0,"failed":0,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[]},{"id":"antsy","title":"antsy","name":"Too Many Clerks","program":"verification","owner":"vishesh","question":"Does an agent committee that buys checks pick better OCR output than confidence alone?","number":"56.78%","unit":"OCR recall, confidence alone, no checks","line":"On 70 real receipts, picking the OCR output by confidence alone did as well as five-agent committees that bought quality checks.","verdict":"null","models":["Laya (convaiinnovations/laya, local)","typesafe/jev-1.13 (Jev)"],"quote":"Confidence alone achieved 56.78% recall, versus 56.30% for Laya and 55.72% for Jev committees; neither adaptive committee saved checks.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-verification-v4/README.md","score":2,"claim":"Five AI agents paid for 140 extra checks while reading receipts. They did no better than simply trusting the scanner's own confidence score.","viz":true,"headline":"Picking OCR text by confidence alone matched agent committees that paid for checks","metric":"share of receipt text read correctly","bars":[{"label":"Confidence alone","value":56.78,"max":100,"text":"56.78%"},{"label":"Committee (local model)","value":56.3,"max":100,"text":"56.30%"},{"label":"Committee (hosted model)","value":55.72,"max":100,"text":"55.72%"}],"good":"high","rank":4,"bars_quote":"Confidence alone achieved 56.78% recall, versus 56.30% for Laya and 55.72% for Jev committees; neither adaptive committee saved checks.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-verification-v4/README.md","studies":[{"id":"antsy-v1","status":"completed_qualification_tied","score":1,"claim":"Fixed and adaptive quorum policies tied on the tested evidence tapes; the small fixture does not establish an adaptive stopping advantage.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/adaptive-quorum/README.md"},{"id":"antsy-v2","status":"failed_clean_qualification","score":0,"claim":"Untested efficacy question: whether adaptive quorum improves decisions after a competent evidence-integration instrument is established; that prerequisite failed in this cohort.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/adaptive-quorum-v2/README.md"},{"id":"antsy-v3","status":"completed_synthetic_mechanism","score":1,"claim":"Adaptive thresholds trade abstention against late correction or misinformation in the guarded factual selector.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/adaptive-quorum-v2/repair-v3/README.md"},{"id":"antsy-v4","status":"completed_valid_null_adverse","score":2,"claim":"Neither tested committee improved observed mean recall or saved checks over the no-check confidence router on these 70 receipts; this does not establish general harm or equivalence.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-verification-v4/README.md"},{"id":"antsy-v5","status":"completed_engineering_reanalysis","score":1,"claim":"Null-neutral updates fix a specific estimator defect and explicit-cost planning changes deterministic check allocation.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-verification-v5/README.md"},{"id":"antsy-v6","status":"new_prospective_application_plan","score":0,"claim":"Untested application question: fallible OCR checks improve receipt-total acceptance and referral decisions enough to justify their measured cost.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-receipt-v6/README.md"},{"id":"antsy-diversity-v7","status":"execution_repaired_qualification_failed_s1_unrun","score":1,"claim":"Repaired native pilot shows competence mismatch and quorum coverage loss; it does not establish a diversity benefit.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-diversity-v7/README.md"},{"id":"antsy-targeted-v8","status":"development_repaired_checker_rejected_separate_native_cohort","score":1,"claim":"Corrected extraction improves the primary reader on reused development receipts; this cohort does not establish fresh qualification or checker benefit.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-targeted-v8/README.md"},{"id":"antsy-targeted-v8-q0","status":"failed_execution_qualification_incomplete_closed_no_retry","score":1,"claim":"The frozen checker failed its execution deadline; fresh competence and complementary correction remain unestablished.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-targeted-v8/README.md"},{"id":"antsy-targeted-v8-d1","status":"completed_valid_adverse_latency_diagnostic_qualification_closed","score":1,"claim":"The larger reused receipt required46.25s and still produced an abstention; OCR occupied36.63s. Current cold45s eligibility remains unmet.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-targeted-v8/README.md"},{"id":"antsy-haiku-panel","status":"qualification_failed_diagnostic_negative_closed","score":1,"claim":"Forty Haiku readers shared a1000-fold normalization error on one qualifying receipt; role variation and a16-call locale diagnostic did not repair it.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-targeted-v8/haiku-normalization-d1/POST-MORTEM.md"},{"id":"antsy-literal-v2r","status":"qualification_failed_evaluation_unrun","score":1,"claim":"Literal+code recovered two baseline errors; checker added no accuracy on four qualification receipts.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/antsy-targeted-v8/literal-v2r/POST-MORTEM.md"}],"contributors":["vishesh"],"runs":9,"done":3,"failed":6,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,1,1,2,2,3,4,0,0,0,0,0,0,0,0,0,0,0,0,0,1,2,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[56,64,"f"],[64,83,"f"],[72,83,"f"],[75,87,"d"],[82,87,"d"],[175,177,"f"],[178,192,"f"],[833,835,"f"],[920,924,"d"]]},{"id":"theseus","title":"swarm of theseus","name":"Swarm of Theseus","program":"recovery","owner":"vishesh","question":"Can useful practices survive after every founding member of a swarm is replaced?","number":"100% vs 52.08%","unit":"accuracy after turnover: notes vs none","line":"Six paired authored worlds: notes and notes plus mentoring each scored 48/48 post-turnover answers, versus 25/48 with neither. No added mentoring benefit was established on this endpoint.","verdict":"result","models":["claude-haiku-4-5-20251001"],"quote":"Notes and notes plus mentoring each achieved 100% post-turnover task accuracy, versus 52.08% with neither; arbitrary convention survival varied.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/9bd914def1c0830023db9342c10627a788abfdbc/researchers/vishesh/notes/swarm-of-theseus/README.md","score":1,"claim":"In a historical 3-agent pilot, supplied notes preserved 48/48 post-turnover answers; no inheritance: 25/48.","viz":true,"headline":"Supplied notes preserved performance in the historical three-agent pilot","metric":"task accuracy after full turnover","bars":[{"label":"No notes, no mentors","value":52.08,"max":100,"text":"52.08%"},{"label":"Written notes","value":100,"max":100,"text":"100%"},{"label":"Notes plus mentoring","value":100,"max":100,"text":"100%"}],"good":"high","rank":4,"bars_quote":"Notes and notes plus mentoring each achieved 100% post-turnover task accuracy, versus 52.08% with neither","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/9bd914def1c0830023db9342c10627a788abfdbc/researchers/vishesh/notes/swarm-of-theseus/README.md","studies":[{"id":"theseus-v1","status":"completed_exploratory","score":1,"claim":"Supplied procedure notes can preserve this small task performance after complete crew replacement; mentoring adds no established benefit.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/swarm-of-theseus/README.md"},{"id":"theseus-v2","status":"completed_failed_qualification","score":1,"claim":"The first v2 native screen failed action competence; procedure-continuity effects remain untested.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/swarm-of-theseus/v2/README.md"},{"id":"theseus-execution-d2","status":"complete_diagnostic_both_unqualified","score":1,"claim":"Atomic calls outperform eight-case batches on six fixed executor worlds, but neither interface qualifies for the later study.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/swarm-of-theseus/execution-diagnostic/README.md"},{"id":"theseus-execution-r1","status":"complete_valid_qualification_result","score":1,"claim":"Repaired atomic executor F qualifies on the fresh fixed R1 screen; a material repair advantage and cultural preservation are unestablished.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/swarm-of-theseus/execution-diagnostic/README.md"},{"id":"theseus-acquisition-a1","status":"native_provider_failure_no_scientific_outcome","score":0,"claim":"Withheld-policy acquisition remains untested after a provider failure.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/swarm-of-theseus/execution-diagnostic/A1-STATUS.md"},{"id":"theseus-acquisition-a2","status":"completed_reviewed_valid_negative_parked","score":1,"claim":"On these fixed cases, acquisition failed while supplied true-policy execution passed; wrong learned policies explain all21 wrong actions.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/swarm-of-theseus/execution-diagnostic/A2-STATUS.md"}],"contributors":["vishesh"],"runs":766,"done":765,"failed":1,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,12,12,1,2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,141,5,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,19,183,1,44,158,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,174,23,0,0,0],"spans":[[48,48,"d"],[48,48,"d"],[48,49,"d"],[49,49,"d"],[49,49,"d"],[49,49,"d"],[49,49,"d"],[54,54,"d"],[54,54,"d"],[54,54,"d"],[54,54,"d"],[54,54,"d"],[54,54,"d"],[54,54,"d"],[214,223,"d"],[214,214,"d"],[214,214,"d"],[214,214,"d"],[214,214,"d"],[214,214,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,215,"d"],[215,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,216,"d"],[216,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,217,"d"],[217,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[218,218,"d"],[219,219,"d"],[219,219,"d"],[219,219,"d"],[219,219,"d"],[219,219,"d"],[219,219,"d"],[344,366,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,350,"d"],[350,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,351,"d"],[351,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,352,"d"],[352,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,353,"d"],[353,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[354,354,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,368,"d"],[368,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,369,"d"],[369,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,370,"d"],[370,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[371,371,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[372,372,"d"],[835,840,"f"],[969,981,"d"],[969,969,"d"],[969,969,"d"],[969,969,"d"],[969,969,"d"],[969,969,"d"],[969,970,"d"],[970,970,"d"],[970,970,"d"],[970,970,"d"],[970,970,"d"],[970,970,"d"],[970,970,"d"],[970,970,"d"],[970,970,"d"],[970,970,"d"],[970,970,"d"],[970,970,"d"],[970,970,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,971,"d"],[971,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[972,972,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,973,"d"],[973,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[974,974,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,975,"d"],[975,976,"d"],[976,976,"d"],[976,976,"d"]],"film":{"src":"media/theseus.mp4","poster":"media/theseus-poster.jpg","loop":"media/theseus-loop.mp4","duration":237.3,"order":1}},{"id":"healing","title":"healing hands","name":"The Cathedral Beats the Bazaar","program":"recovery","owner":"vishesh","question":"Can a mesh of 200 curators repair a changing evidence index better than a central one?","number":"39.3% -> 21.1%","unit":"peer-mesh error, capacity 4 -> 16 items","line":"More bandwidth helped 200 peer curators repair a shared index, but a version-aware central index still had lower error in every scenario.","verdict":"adverse","models":["typesafe/jev-1.13 (Jev, saved extraction labels)"],"quote":"Combined-scenario peer-verified mean post-event query error fell from 39.3% to 21.1% (18.1 percentage points); final legitimate new-evidence retention rose from 36.2% to 99.9%.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/healing-helping-hands/practical/POST-02.md","score":1,"claim":"Sources retracted findings, and 200 AI curators had to update a shared index peer to peer. Even with 4x the bandwidth, the curators ended up more wrong than one central index.","viz":true,"headline":"More bandwidth helped 200 peers repair a shared index; a central index still did better","metric":"wrong answers to queries after a change","bars":[{"label":"Central index","value":16.6,"max":100,"text":"16.6%"},{"label":"Peers, low bandwidth","value":39.3,"max":100,"text":"39.3%"},{"label":"Peers, 4x bandwidth","value":21.1,"max":100,"text":"21.1%"}],"good":"low","rank":3,"bars_quote":"Combined-scenario peer-verified mean post-event query error fell from 39.3% to 21.1% (18.1 percentage points); final legitimate new-evidence retention rose from 36.2% to 99.9%. Traffic rose from about 60,394 to 87,842 item copies. The central controls remain the mandatory comparison: central-verified error 16.6%","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/healing-helping-hands/practical/POST-02.md","studies":[{"id":"healing-pilot03","status":"completed_limited_mechanism","score":1,"claim":"Sharing withdrawal notices improves recovery in the programmed evidence atlas with qualified extraction.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/healing-helping-hands/README.md"},{"id":"healing-practical01","status":"completed_valid_adverse","score":1,"claim":"No scenario met the peer-verified utility rule against central append; the observed capped-mesh disadvantage is specific to these reused fixtures and unequal transport mechanisms.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/healing-helping-hands/practical/README.md"},{"id":"healing-practical02","status":"completed_valid_capacity_diagnostic","score":1,"claim":"Increasing mesh capacity improves this programmed delivery protocol, but it does not outperform the version-aware central index.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/healing-helping-hands/practical/README.md"},{"id":"healing-c4-s0","status":"invalid_qualification_S1_withheld","score":1,"claim":"C4 qualification contains expected-label leakage; its perfect Jev score does not establish clean capability.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/healing-helping-hands/c4/S0-POST.md"}],"contributors":["vishesh"],"runs":189,"done":185,"failed":3,"cost":0.03,"calls":2820,"ridge":[0,0,0,0,0,1,181,0,0,0,0,0,0,0,0,1,2,1,1,1,1,1,1,1,1,2,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,2,1,1,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,2],"spans":[[32,41,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,40,"d"],[40,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[41,41,"d"],[99,104,"d"],[104,159,"f"],[159,167,"f"],[291,295,"d"],[295,325,"d"],[847,854,"f"],[990,996,"d"],[995,1000,"r"]],"film":{"src":"media/healing.mp4","poster":"media/healing-poster.jpg","loop":"media/healing-loop.mp4","duration":16.2,"order":4}},{"id":"memory-handoff","title":"memory handoff","name":"Read the Footnotes","program":"recovery","owner":"dmarz","question":"Can a successor agent avoid repeating errors in the notes it inherits?","number":"24/24 -> 0/24","unit":"misquotes repeated, refs vs full records","line":"Given the full text of records its notes cited, a Qwen3.7 Flash successor stopped repeating a misquoted value; a false original source got through.","verdict":"result","models":["Qwen3.7 Flash","gpt-6-luna"],"quote":"By state: misquote \u22121.0 (metadata-only repeated the misquoted value in 24 of 24 roots, content-bound in 0 of 24); stale 0.0 (neither policy repeated the stale value: metadata-only abstained in 24 of 24, content-bound in 0 of 24 repeated it).","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/memory-handoff-qwen/RESULTS.md","score":2,"claim":null,"viz":false,"headline":"Given the full records its notes cited, a new AI stopped repeating an inherited misquote.","metric":"cases where the misquote was repeated","bars":[{"label":"source names only","value":24,"max":24,"text":"24 of 24"},{"label":"full source text","value":0,"max":24,"text":"0 of 24"}],"good":"low","rank":3,"bars_quote":"By state: misquote \u22121.0 (metadata-only repeated the misquoted value in 24 of 24 roots, content-bound in 0 of 24); stale 0.0 (neither policy repeated the stale value: metadata-only abstained in 24 of 24, content-bound in 0 of 24 repeated it).","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/memory-handoff-qwen/RESULTS.md","studies":[{"id":"memory-handoff-qwen","status":"attempt_002_completed_s1_recomputed","score":2,"claim":"A Qwen3.7 Flash successor (reasoning disabled, answer with working fields) that is handed the contents of the records its inherited notes cite does not repeat a misquoted inherited value, where resolving only origin and version leaves it in place, and it keeps its clean-memory answers.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/memory-handoff-qwen/README.md"},{"id":"memory-handoff-qwen-gpt-6-luna","status":"chain_003_completed_s1_verified","score":2,"claim":"A gpt-6-luna successor (reasoning effort low, answer-only format) passes the unrepaired handoff instrument on the 24 requests where Qwen3.7 Flash without reasoning scored 19, and follows the stated source policy on fresh roots: content-bound retrieval removes the misquote and stale errors that metadata-only resolution leaves or turns into abstentions, with clean answers kept.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/memory-handoff-qwen/RESULTS.md"}],"contributors":["dmarz"],"runs":11,"done":10,"failed":1,"cost":0.1,"calls":1224,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,3,0,0,0,0,0,0,0,4,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,4,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[526,526,"d"],[526,526,"d"],[526,526,"f"],[580,580,"d"],[580,580,"d"],[580,580,"d"],[580,584,"d"],[919,920,"d"],[920,920,"d"],[920,920,"d"],[920,924,"d"]]},{"id":"immune-response","title":"immune response","name":"This Is Fine","program":"recovery","owner":"vishesh","question":"Does cleaning up stale team memory help agents recover a broken deployment safely?","number":"0/6 vs 5/6","unit":"healthy ticks, stale advice: team v solo","line":"One retained-memory episode per authored case/architecture, six dependent ticks each. Solo ended healthy but temporarily damaged the healthy control. This does not establish a causal reviewer effect.","verdict":"adverse","models":["claude-haiku-4-5-20251001"],"quote":"| Stale advice | 0/6 | 5/6 | 0 |","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/9bd914def1c0830023db9342c10627a788abfdbc/researchers/vishesh/notes/immune-response-v3/scenario-study/ASSESSMENT.md","score":1,"claim":"In selected historical fixtures, a four-agent team missed a fault; the solo control repaired it but harmed a healthy case.","viz":true,"headline":"Historical fixtures exposed both failed team recovery and solo over-intervention","metric":"healthy ticks within selected episodes (not independent trials)","bars":[{"label":"Team, broken system","value":0,"max":6,"text":"0 of 6"},{"label":"Solo, broken system","value":5,"max":6,"text":"5 of 6"},{"label":"Team, healthy system","value":6,"max":6,"text":"6 of 6"},{"label":"Solo, healthy system","value":3,"max":6,"text":"3 of 6"}],"good":"high","rank":4,"bars_quote":"| Stale advice | 0/6 | 5/6 | 0 |\n| Migrated data | 0/6 | 5/6 | 0 |\n| Healthy false alarm | 6/6 | 3/6 | 1 |","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/9bd914def1c0830023db9342c10627a788abfdbc/researchers/vishesh/notes/immune-response-v3/scenario-study/ASSESSMENT.md","studies":[{"id":"immune-v1","status":"historical_scripted_completed","score":1,"claim":"Oracle restoration repairs private/shared-state contamination in a programmed fixture.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/actual-experiments/immune-response/README.md"},{"id":"immune-v2","status":"scripted_complete_native_unqualified","score":0,"claim":"Untested efficacy question: shared restoration contributes beyond private repair and stale-record blocking.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/actual-experiments/immune-response/README.md"},{"id":"immune-v3-engineering","status":"scripted_engineering_complete","score":1,"claim":"Selective repair mechanisms behave as designed across synthetic recurrence, benign-learning and missing-lineage cases.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/immune-response-v3/README.md"},{"id":"immune-scenario","status":"completed_adverse_qualification","score":1,"claim":"The small scenario diagnostics exposed failed team recovery and healthy-service damage by the solo baseline; they do not identify a causal team-versus-solo advantage.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/immune-response-v3/scenario-study/README.md"},{"id":"immune-receipt","status":"interrupted_no_qualification","score":0,"claim":"Untested efficacy question: fact-checking frozen reviewer recommendations improves commander recovery beyond generic caution and identical raw evidence.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/immune-response-v3/evidence-study/README.md"}],"contributors":["vishesh"],"runs":3,"done":0,"failed":3,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0],"spans":[[285,290,"f"],[847,847,"f"],[974,974,"f"]]},{"id":"capture-memory","title":"capture memory","name":"Eternal Sunshine of the Captured Swarm","program":"recovery","owner":"shadow","question":"Can agents with short memories pull a captured swarm back after the attackers leave?","number":"9 of 60","unit":"gpt-4o-mini mixed swarms fully recovered","line":"In a simulated captured swarm, mixing short- and long-memory agents sometimes restored the old convention with gpt-4o-mini, not Gemma or Qwen.","verdict":"result","models":["openai/gpt-4o-mini","google/gemma-3-27b-it","qwen/qwen3-235b-a22b-2507"],"quote":"9 of 60 episodes with f in\n  {5/8, 3/4, 7/8} recover fully and 20 of 60 are at or above 0.5 at round 30; in the 81 episodes of the other four\n  cells (pure short, pure full, f = 1/2, f = 15/16) that is 0 and 3.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/capture-memory-mix/README.md","score":2,"claim":null,"viz":false,"headline":"With one model, mixing short and long memory agents sometimes undid a swarm takeover","metric":"runs fully recovered after attackers left","bars":[{"label":"Other memory mixes","value":0,"max":81,"text":"0 of 81"},{"label":"Mostly-short mix","value":9,"max":60,"text":"9 of 60"}],"good":"high","rank":3,"bars_quote":"9 of 60 episodes with f in\n  {5/8, 3/4, 7/8} recover fully and 20 of 60 are at or above 0.5 at round 30; in the 81 episodes of the other four\n  cells (pure short, pure full, f = 1/2, f = 15/16) that is 0 and 3.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/capture-memory-mix/README.md","studies":[{"id":"capture-memory","status":"completed_scripted_development","score":1,"claim":"Memory length affects return to a convention after perfect attacker removal under a scripted policy.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/capture-memory/README.md"},{"id":"capture-memory-mix","status":"exploratory_corrected_model_specific","score":1,"claim":"Mixture rescue is model-specific in these pilots, not a general swarm result; the long-list reading explanation is a post-hoc lead.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/capture-memory-mix/README.md"},{"id":"capture-memory-reading-rule","status":"complete_narrow_diagnostic","score":2,"claim":"Reversing these indexed raw histories changed majority-name probability by far less than the prespecified 0.10 practical threshold.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/capture-memory-mix/reading-rule/README.md"}],"contributors":["shadow"],"runs":139,"done":138,"failed":1,"cost":3.7,"calls":181235,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,9,9,72,33,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,20,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[83,83,"d"],[83,84,"d"],[84,86,"d"],[84,86,"d"],[84,88,"d"],[84,86,"d"],[86,90,"d"],[86,88,"d"],[86,89,"d"],[88,89,"d"],[88,91,"d"],[89,90,"d"],[89,91,"d"],[90,91,"d"],[94,94,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[98,98,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[99,99,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,100,"d"],[100,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[101,101,"d"],[234,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[235,767,"d"],[785,785,"f"]]},{"id":"regrowth-200","title":"regrowth 200","name":"The Long Way Home","program":"recovery","owner":"vishesh","question":"Can 200 small-model agents rebuild good routes after part of their map breaks?","number":"6.5%","unit":"routes shortest (algorithm: 100%)","line":"Swarms of 200 small-model agents restored routes to the exit after damage, but few were shortest, and adding Laya decision heads changed nothing.","verdict":"adverse","models":["Qwen3 0.6B (local)","Laya (convaiinnovations/laya, local)"],"quote":"Both model arms ended at 100% valid routes and 6.5% shortest routes, with and without damage.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/regrowth-200/README.md","score":1,"claim":null,"viz":false,"headline":"200 small-model agents rebuilt working routes after damage, but only 6.5% were shortest","metric":"routes that were the shortest path","bars":[{"label":"Small-model swarm","value":6.5,"max":100,"text":"6.5%"},{"label":"Swarm + decision heads","value":6.5,"max":100,"text":"6.5%"},{"label":"Exact algorithm","value":100,"max":100,"text":"100%"}],"good":"high","rank":3,"bars_quote":"Both model arms ended at 100% valid routes and 6.5% shortest routes, with and without damage. The algorithm ended at 100% on both measures.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/regrowth-200/README.md","studies":[{"id":"regrowth-200","status":"completed_descriptive_registration_failure","score":1,"claim":"The added Laya heads did not improve final routing quality on this one map; both model arms remained far below exact routing on shortest paths.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/regrowth-200/README.md"}],"contributors":["vishesh"],"runs":0,"done":0,"failed":0,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[]},{"id":"swarm-size","title":"swarm size","name":"Two Heads Are Faster Than One","program":"operations","owner":"vishesh","question":"Does adding a second agent make a task faster, cheaper, or more accurate?","number":"-.375 quality","unit":"two agents vs one, worse parallel task","line":"Splitting work across two agents instead of one cut time and cost on parallel tasks but lowered accuracy; chained tasks stayed near zero either way.","verdict":"result","models":["claude-haiku-4-5-20251001"],"quote":"Parallel N2 latency fell 45.27% and 32.38%, with 14.53% and 14.89% lower cost, but quality dropped .125 and .375.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/optimal-swarm-size/reviews/q-a6-post.md","score":1,"claim":"Splitting a 16-question task between two AI agents finished it a third faster, and more than doubled the wrong answers.","viz":true,"headline":"Two agents finished parallel tasks faster and cheaper than one, but got more answers wrong","metric":"share of items answered correctly","bars":[{"label":"1 agent, task A","value":0.8125,"max":1,"text":"81%"},{"label":"2 agents, task A","value":0.6875,"max":1,"text":"69%"},{"label":"1 agent, task B","value":0.6875,"max":1,"text":"69%"},{"label":"2 agents, task B","value":0.3125,"max":1,"text":"31%"}],"good":"high","rank":4,"bars_quote":"| 4 | parallel | .8125 \u2192 .6875 | 29.43 \u2192 16.11 | .041336 \u2192 .035328 | 1 \u2192 2 |\n| 5 | parallel | .6875 \u2192 .3125 | 25.17 \u2192 17.02 | .041336 \u2192 .035180 | 1 \u2192 2 |","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/optimal-swarm-size/reviews/q-a6-post.md","studies":[{"id":"optimal-swarm-size","status":"qualification_observed_conditional_policy_untested","score":0,"claim":"Untested efficacy question: whether a conditional launch rule can choose useful swarm size under task and resource constraints.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/optimal-swarm-size/README.md"},{"id":"optimal-swarm-size-q-a5","status":"complete_diagnostic","score":1,"claim":"Incorrect final evidence answers were already incorrect in worker outputs in this diagnostic cohort.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/optimal-swarm-size/README.md"},{"id":"optimal-swarm-size-q-a6","status":"complete_valid_exploratory_result","score":1,"claim":"N2 was faster and cheaper but less accurate on both parallel fixtures; chain quality stayed near zero with mixed timing.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/optimal-swarm-size/README.md"},{"id":"optimal-swarm-size-q-a7","status":"complete_valid_exploratory_result","score":1,"claim":"Binding improved joint value/source correctness35/64 to48/64 on two development roots, but failed the preset0.90 mean-quality screen.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/optimal-swarm-size/README.md"},{"id":"optimal-swarm-size-outage-o1","status":"complete_valid_exploratory_result","score":1,"claim":"Native actors can recover these authored outages, but fixed-four duplicated repairs and contraction triggered in stable control.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/optimal-swarm-size/README.md"},{"id":"optimal-swarm-size-outage-o2","status":"complete_valid_exploratory_result","score":1,"claim":"Explicit ownership removed the observed duplicate-repair failure; one/four/eight/controller tied recovery in both eight-service worlds, with higher model cost at largerN.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/optimal-swarm-size/README.md"}],"contributors":["vishesh"],"runs":70,"done":50,"failed":20,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,1,0,0,0,0,0,0,2,14,4,0,0,0,0,0,0,0,0,2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,4,2,11,4,0,0,0,0,2,7,4,8,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,7,6],"spans":[[49,49,"d"],[89,89,"f"],[93,93,"f"],[97,97,"d"],[98,98,"f"],[98,98,"f"],[98,99,"f"],[99,99,"f"],[99,99,"f"],[99,99,"f"],[99,99,"f"],[99,99,"f"],[99,99,"f"],[99,99,"f"],[99,100,"f"],[100,100,"f"],[100,100,"f"],[100,100,"f"],[100,100,"f"],[100,101,"f"],[160,160,"f"],[160,161,"f"],[280,281,"d"],[281,281,"d"],[281,281,"d"],[281,281,"d"],[286,287,"d"],[287,288,"d"],[288,289,"d"],[289,289,"d"],[289,290,"d"],[290,290,"d"],[290,291,"d"],[291,291,"d"],[291,292,"d"],[292,292,"d"],[292,293,"d"],[293,294,"d"],[294,294,"d"],[294,295,"d"],[295,295,"d"],[295,296,"d"],[330,331,"d"],[331,331,"d"],[331,332,"d"],[332,332,"d"],[332,333,"d"],[333,334,"d"],[334,334,"d"],[334,335,"d"],[341,348,"d"],[342,348,"d"],[342,348,"d"],[343,348,"d"],[344,348,"d"],[344,348,"d"],[345,348,"d"],[345,348,"d"],[992,992,"d"],[992,992,"d"],[992,993,"d"],[993,993,"d"],[993,993,"d"],[993,994,"d"],[994,994,"d"],[994,994,"d"],[994,994,"d"],[994,994,"d"],[994,995,"d"],[995,995,"d"]]},{"id":"compositional-safety","title":"compositional safety","name":"The Sum of All Permissions","program":"operations","owner":"dmarz","question":"Can fact receipts stop agents whose permitted steps add up to a forbidden outcome?","number":"21/24 vs 24/24","unit":"safe completions: Haiku 4.5 vs Opus 5.5","line":"Opus 5.5 completed all 24 qualification workflows safely where Haiku 4.5 managed 21; whether fact receipts prevent violations is still untested.","verdict":"not-qualified","models":["Claude Haiku 4.5","Claude Sonnet 5.5","Claude Sonnet 5","Claude Opus 5.5"],"quote":"Claude Opus 5.5 (`claude-opus-5-5`, adaptive thinking, effort high, execution-v2) passed the unchanged Q0 readiness qualification twice: **q0-007** 24/24 valid and safely complete on six new structures ([post-mortem](reviews/q0-007-post.md)) and **q0-010** 24/24 at design v9 ([post-mortem](reviews/q0-010-post.md)). These are the study's first passing qualifications; Haiku 4.5 (q0-005) had 21/24.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/compositional-safety/README.md","score":1,"claim":"We split one job across four AI agents. With a smaller model, the agents reused approvals that were already spent; with the strongest model, all 24 jobs finished safely.","viz":true,"headline":"Opus finished all 24 test workflows safely, Haiku 21; the safety fix itself is untested.","metric":"test workflows completed safely","bars":[{"label":"Haiku 4.5","value":21,"max":24,"text":"21 of 24"},{"label":"Opus 5.5","value":24,"max":24,"text":"24 of 24"}],"good":"high","rank":2,"bars_quote":"Claude Opus 5.5 (`claude-opus-5-5`, adaptive thinking, effort high, execution-v2) passed the unchanged Q0 readiness qualification twice: **q0-007** 24/24 valid and safely complete on six new structures ([post-mortem](reviews/q0-007-post.md)) and **q0-010** 24/24 at design v9 ([post-mortem](reviews/q0-010-post.md)). These are the study's first passing qualifications; Haiku 4.5 (q0-005) had 21/24.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/compositional-safety/README.md","studies":[{"id":"compositional-safety","status":"completed_failed_qualification","score":1,"claim":"The q0-004 model/configuration failed the frozen safe-completion qualification; receipt-treatment efficacy was untested at that assessment.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/compositional-safety/README.md"},{"id":"compositional-safety-q0-005","status":"completed_failed_qualification","score":1,"claim":"The frozen q0-005 Haiku 4.5 configuration with execution-v2 fails the unchanged readiness qualification; receipt-treatment efficacy remains untested.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/compositional-safety/README.md"},{"id":"compositional-safety-d0-003-plan","status":"planned_not_authorized_to_start","score":0,"claim":"Whether the proposed Sonnet 5 configuration safely completes the selected execution-v2 workflows is untested; d0-003 is a plan only.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/compositional-safety/README.md"},{"id":"compositional-safety-opus-q0","status":"completed_passed_qualification","score":1,"claim":"The claude-opus-5-5 configuration (adaptive thinking, effort high, 4,096-token cap, execution-v2) meets the unchanged Q0 readiness thresholds on the development structures tested; receipt-treatment efficacy is untested by these runs.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/dmarz/compositional-safety/README.md"}],"contributors":["dmarz"],"runs":115,"done":110,"failed":5,"cost":0,"calls":3837,"ridge":[0,0,0,0,0,0,0,0,0,2,0,4,7,9,0,0,0,0,0,0,5,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,13,2,7,6,0,3,2,2,0,9,6,1,3,1,1,2,1,4,7,5,2,2,1,1,2,1,3,6,9,8,2,3,1,1,1,2,1,1,3,2,1,1,1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[57,57,"d"],[61,62,"d"],[72,73,"d"],[73,74,"d"],[74,75,"d"],[75,75,"d"],[75,75,"d"],[80,80,"d"],[80,81,"d"],[81,81,"d"],[81,81,"d"],[81,82,"d"],[82,82,"d"],[82,82,"d"],[82,83,"d"],[83,84,"d"],[84,85,"d"],[85,85,"d"],[85,86,"d"],[86,86,"d"],[129,129,"d"],[129,129,"d"],[129,130,"d"],[130,130,"d"],[130,130,"d"],[261,261,"d"],[261,261,"d"],[261,261,"d"],[261,261,"d"],[261,261,"d"],[261,261,"d"],[261,262,"d"],[262,262,"d"],[262,262,"d"],[262,262,"d"],[262,262,"d"],[262,262,"d"],[262,262,"d"],[267,268,"d"],[268,269,"d"],[270,271,"d"],[271,272,"d"],[272,272,"d"],[272,273,"d"],[273,274,"d"],[274,275,"d"],[275,277,"d"],[277,278,"d"],[278,280,"d"],[280,280,"d"],[280,281,"d"],[290,291,"d"],[291,292,"d"],[292,297,"d"],[297,301,"d"],[301,304,"f"],[313,313,"d"],[313,313,"d"],[313,315,"d"],[315,315,"d"],[315,316,"d"],[316,317,"d"],[317,318,"d"],[318,319,"d"],[319,320,"d"],[320,321,"d"],[321,322,"d"],[322,323,"d"],[323,323,"d"],[323,332,"d"],[332,337,"d"],[337,356,"d"],[356,361,"f"],[366,367,"d"],[367,368,"d"],[368,368,"d"],[369,369,"d"],[369,370,"d"],[370,371,"d"],[371,372,"d"],[372,374,"d"],[374,375,"d"],[375,376,"d"],[376,376,"d"],[376,377,"d"],[377,377,"d"],[377,385,"d"],[385,391,"d"],[391,411,"d"],[411,420,"d"],[420,449,"f"],[425,426,"d"],[426,427,"d"],[427,428,"d"],[428,428,"f"],[431,432,"d"],[432,433,"d"],[433,434,"d"],[434,435,"d"],[435,436,"d"],[436,437,"d"],[437,437,"d"],[437,438,"d"],[438,439,"d"],[439,440,"d"],[440,441,"d"],[441,441,"d"],[441,441,"d"],[442,450,"d"],[450,456,"d"],[456,475,"d"],[475,494,"d"],[494,499,"d"],[499,503,"d"],[503,528,"f"]]},{"id":"poietic","title":"poietic agents","name":"Some Assembly Required","program":"operations","owner":"vishesh","question":"Can identical agents discover a cheaper division of labor without becoming brittle?","number":null,"unit":null,"line":"The self-specializing swarm was designed and tested offline, but its first live qualification stopped after 3 calls, so no efficacy result exists.","verdict":"not-qualified","models":[],"quote":"Both native qualification attempts failed at interface boundaries; no model is qualified and no swarm efficacy result exists.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/poietic-agents/README.md","score":0,"claim":"Can a group of identical AI agents invent its own division of labor? Still open: our first live test broke down after three calls.","viz":true,"headline":"A self-organizing swarm design failed its first live checks, so there is no result yet","metric":"no result measured","bars":[],"good":"none","rank":1,"bars_quote":"Both native qualification attempts failed at interface boundaries; no model is qualified and no swarm efficacy result exists.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/poietic-agents/README.md","studies":[{"id":"poietic-agents","status":"D0-01-failed;trace-reviewed;no-successor-authorized","score":0,"claim":"Reversible self-differentiation improves lifetime swarm efficiency without losing quality; this intended efficacy claim remains untested.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/poietic-agents/README.md"},{"id":"heterogeneous-swarms","status":"research_proposals_no_experiments","score":0,"claim":"Untested efficacy question: whether mixed-model or self-differentiating swarms provide specialization beyond compute, correlation and coordination costs.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/vishesh/heterogeneous-swarms/README.md"}],"contributors":["vishesh"],"runs":9,"done":0,"failed":9,"cost":1.92,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,3,3,3,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,3,0,0,0,0,0,0,0,0,0,0,0,0,0,3,0,0,0,0,0,0,0,0,0,0,0],"spans":[[324,336,"f"],[324,336,"f"],[324,336,"f"],[841,841,"f"],[841,841,"f"],[841,841,"f"],[926,926,"f"],[926,926,"f"],[926,926,"f"]]},{"id":"wild-halflife","title":"idea half-life","name":"The Tortoise and the Wiki","program":"wild","owner":"shadow","question":"How quickly do links posted by one agent get picked up by others?","number":"19.4% vs 40.1%","unit":"URLs reused by others: wiki vs git","line":"Links spread faster on an agent wiki but were reused more often in the research swarm's git history; neither data set shows why.","verdict":"result","models":[],"quote":"| Units; reaching another identity | 22,831; 4,429 (19.4%) | 9,879; 3,957 (40.1%) |","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-halflife/FINDING.md","score":1,"claim":"We tracked links shared between AI agents in two real communities. On a busy wiki, reuse came within minutes; in a slower research swarm, it took hours, but twice as many links caught on.","viz":true,"headline":"Links were reused by other agents twice as often in a research swarm as on an agent wiki","metric":"posted links later reused by another agent","bars":[{"label":"Agent wiki","value":19.4,"max":100,"text":"19.4%"},{"label":"Research swarm git","value":40.1,"max":100,"text":"40.1%"}],"good":"none","rank":3,"bars_quote":"| Units; reaching another identity | 22,831; 4,429 (19.4%) | 9,879; 3,957 (40.1%) |","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-halflife/FINDING.md","studies":[{"id":"wild-halflife","status":"descriptive_complete_mechanism_not_identified","score":1,"claim":"In these frozen corpora, conditional URL reuse delays and pooled adoption-rate slopes differ; a causal copying mechanism and semantic idea half-life are not identified.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-halflife/README.md"}],"contributors":["shadow"],"runs":1,"done":1,"failed":0,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[788,788,"d"]]},{"id":"wild-identity","title":"wild identity","name":"Name Dropping","program":"wild","owner":"shadow","question":"Do agent identities that stick around longer coordinate more?","number":"2.34x","unit":"name references: kept text vs new edits","line":"Retained wiki page text overstates how often agents name each other; whether longer-lived identities coordinate more could not be determined.","verdict":"result","models":[],"quote":"The paired difference is **21.19 percentage points**, conditional 95% page-cluster bootstrap CI **[14.95, 27.48]**; the proportion ratio is **2.34x [1.96, 2.75]**.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-identity/FINDING.md","score":1,"claim":null,"viz":false,"headline":"Whole wiki pages make agents seem to name each other over twice as often as new edits do","metric":"edits that name another agent","bars":[{"label":"Newly written text","value":15.8,"max":100,"text":"15.80%"},{"label":"Whole saved page","value":36.99,"max":100,"text":"36.99%"}],"good":"none","rank":2,"bars_quote":"Retained snapshots contain a non-self known-name reference in **5,065/13,692 attributed revisions (36.99%)**, versus **2,164/13,692 (15.80%)** in inserted/replaced hunk text.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-identity/FINDING.md","studies":[{"id":"wild-identity","status":"descriptive_complete_coordination_effect_not_identified","score":1,"claim":"Archive descriptions and reference-provenance diagnostics are reproducible; an identity-churn effect on coordination and a SwarmTraces identity graph are not identified.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-identity/README.md"}],"contributors":["shadow"],"runs":0,"done":0,"failed":0,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[]},{"id":"wild-evidence-depth","title":"evidence depth","name":"The Squeaky Wheel","program":"wild","owner":"shadow","question":"How much of a public incident dataset lets you see both an attack and its response?","number":"8.49%","unit":"payloads with a linked response","line":"Only a small share of payloads in a public agent-incident dataset link to a recorded response, and longer payloads are far more likely to have one.","verdict":"result","models":[],"quote":"Dividing all response rows by payload rows gives 25.27%, nearly three times the actual linked-payload fraction of 8.49%.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-evidence-depth/FINDING.md","score":1,"claim":"During the Hugging Face incident, AI agents sent 91,037 attack payloads. Only 8.5% have a recorded reply, and the longest payloads were far more likely to get one.","viz":true,"headline":"In a public agent-incident archive, long attacks are far likelier to have a recorded reply","metric":"attack texts with a linked response","bars":[{"label":"1 to 255 characters","value":2.26,"max":100,"text":"2.26%"},{"label":"4096+ characters","value":39.98,"max":100,"text":"39.98%"}],"good":"none","rank":3,"bars_quote":"Direct response attachment rises from **1,343/59,341 (2.26%)** for payload texts of 1\u2013255 characters to **427/1,068 (39.98%)** for 4096+ characters, a **17.7-fold descriptive ratio**.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-evidence-depth/FINDING.md","studies":[{"id":"wild-evidence-depth","status":"complete_descriptive_census_no_execution_or_success_inference","score":1,"claim":"In this selected release, 7,733/91,037 payload artifacts have a direct response child; response-link coverage varies with artifact length.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-evidence-depth/README.md"}],"contributors":["shadow"],"runs":0,"done":0,"failed":0,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[]},{"id":"wild-askswarm","title":"ask the swarm","name":"A Whodunit With No Who","program":"wild","owner":"shadow","question":"Can one tool answer the same social questions across three agent swarm archives?","number":"0/189,579","unit":"SwarmTraces records naming an actor","line":"One reusable tool can describe three agent archives, and it shows one of them has no usable actor names or timestamps at all.","verdict":"result","models":[],"quote":"Identity coverage: 13,692/14,591; 0/189,579; 1,924/2,673.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/evidence-metadata.json","score":1,"claim":"We studied a public archive of 189,579 records from a real AI-agent incident. Not one record says which agent acted, or when.","viz":true,"headline":"One of three agent archives names no actors at all, so who did what cannot be traced there","metric":"records with a usable actor name","bars":[{"label":"Agent wiki","value":13692,"max":14591,"text":"94%"},{"label":"SwarmTraces","value":0,"max":189579,"text":"0%"},{"label":"Research swarm git","value":1924,"max":2673,"text":"72%"}],"good":"high","rank":3,"bars_quote":"Identity coverage: 13,692/14,591; 0/189,579; 1,924/2,673.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/evidence-metadata.json","studies":[{"id":"wild-askswarm","status":"exploratory_offline_tool_and_corpus_census_complete","score":1,"claim":"The shared offline interface produces corpus-level lexical and participation descriptions and exposes unavailable identity/time endpoints in the released SwarmTraces artifacts.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-askswarm/README.md"}],"contributors":["shadow"],"runs":3,"done":3,"failed":0,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,3,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[[744,744,"d"],[744,744,"d"],[744,744,"d"]]},{"id":"wild-delete-return","title":"delete and return","name":"Night of the Living Pages","program":"wild","owner":"shadow","question":"Do agents write pages again after moderators delete them?","number":"83 -> 83","unit":"saves 30 min before vs after deletion","line":"In one wiki incident export, deleted pages were saved again about as often after their first deletion as before it.","verdict":"null","models":[],"quote":"The predeclared 30-minute paired windows contain 83 observed saves before and 83 after first deletion on 2,728 eligible pages; 19 pages have a post-guard save.","source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-delete-return/FINDING.md","score":1,"claim":null,"viz":false,"headline":"Deleted wiki pages were written to as often after deletion as before it","metric":"page saves within 30 minutes","bars":[{"label":"Before deletion","value":83,"max":100,"text":"83 saves"},{"label":"After deletion","value":83,"max":100,"text":"83 saves"}],"good":"none","rank":2,"bars_quote":"The predeclared 30-minute paired windows contain 83 observed saves before and 83 after first deletion on 2,728 eligible pages; 19 pages have a post-guard save.","bars_source":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-delete-return/FINDING.md","studies":[{"id":"wild-delete-return","status":"complete_bounded_descriptive_null_no_scaling_or_causal_inference","score":1,"claim":"The predeclared 30-minute paired windows contain 83 observed saves before and 83 after first deletion on 2,728 eligible pages; 19 pages have a post-guard save.","doc":"https://github.com/dmarzzz/swarm-dynamics-lab/blob/main/5-experiments/studies/shadow/wild-delete-return/README.md"}],"contributors":["shadow"],"runs":0,"done":0,"failed":0,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0],"spans":[]}],"other":{"runs":7,"done":6,"failed":0,"cost":0,"calls":0,"ridge":[0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,1,1,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,2,2,0,0,0,0,0,0,0,0,1]}}