{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UA7BKEKIBN7GPJIIWGD2J2IJC6","short_pith_number":"pith:UA7BKEKI","schema_version":"1.0","canonical_sha256":"a03e1511480b7e67a508b187a4e909178b4d5eb9432d76db26cf58c6732eb3e5","source":{"kind":"arxiv","id":"2510.20963","version":2},"attestation_state":"computed","paper":{"title":"When and Why Does Multi-Agent Debate Fail and Does It Really Underperform?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Han, Gang Niu, James Cheng, Masashi Sugiyama, Yongqiang Chen","submitted_at":"2025-10-23T19:46:00Z","abstract_excerpt":"Multi-agent debate (MAD) was proposed as a promising approach for ensembling the wisdom of multiple large language models (LLMs) to improve reasoning and provide effective supervision to superhuman LLMs. However, increasing empirical evidence suggests that MAD may not outperform or even significantly underperform single-agent approaches (SA), raising doubts about the benefits of MAD. In this work, we investigate this issue by analyzing the incentive structures of popular MAD paradigms: (i) competitive MAD (CopMAD) where agents compete by holding opposing positions; (ii) consensus-seeking MAD ("},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2510.20963","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-10-23T19:46:00Z","cross_cats_sorted":[],"title_canon_sha256":"21a5eec8b9f68642e1b58a8a9cf57a823d7e8d374d1ebcf426e308172186bbe8","abstract_canon_sha256":"3e455e817156fab3b14103b7fd304bd78c8bed8d8d556258a705a01545258974"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-15T01:21:47.620021Z","signature_b64":"pJ8NXCMVGF2KQ2uhLL/FUqPBC52TnbAXeRjYBpZuYjpR4c0W/Y1MSMteGz3vT2/Xsods4y0ak2w06ktwr0SDAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a03e1511480b7e67a508b187a4e909178b4d5eb9432d76db26cf58c6732eb3e5","last_reissued_at":"2026-07-15T01:21:47.619080Z","signature_status":"signed_v1","first_computed_at":"2026-07-15T01:21:47.619080Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When and Why Does Multi-Agent Debate Fail and Does It Really Underperform?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Han, Gang Niu, James Cheng, Masashi Sugiyama, Yongqiang Chen","submitted_at":"2025-10-23T19:46:00Z","abstract_excerpt":"Multi-agent debate (MAD) was proposed as a promising approach for ensembling the wisdom of multiple large language models (LLMs) to improve reasoning and provide effective supervision to superhuman LLMs. However, increasing empirical evidence suggests that MAD may not outperform or even significantly underperform single-agent approaches (SA), raising doubts about the benefits of MAD. In this work, we investigate this issue by analyzing the incentive structures of popular MAD paradigms: (i) competitive MAD (CopMAD) where agents compete by holding opposing positions; (ii) consensus-seeking MAD ("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.20963","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.20963/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2510.20963","created_at":"2026-07-15T01:21:47.619528+00:00"},{"alias_kind":"arxiv_version","alias_value":"2510.20963v2","created_at":"2026-07-15T01:21:47.619528+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.20963","created_at":"2026-07-15T01:21:47.619528+00:00"},{"alias_kind":"pith_short_12","alias_value":"UA7BKEKIBN7G","created_at":"2026-07-15T01:21:47.619528+00:00"},{"alias_kind":"pith_short_16","alias_value":"UA7BKEKIBN7GPJII","created_at":"2026-07-15T01:21:47.619528+00:00"},{"alias_kind":"pith_short_8","alias_value":"UA7BKEKI","created_at":"2026-07-15T01:21:47.619528+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.01251","citing_title":"Collaborative Disagreement Resolution for Scalable Oversight","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2604.02478","citing_title":"AIVV: Neuro-Symbolic LLM Agent-Integrated Verification and Validation for Trustworthy Autonomous Systems","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UA7BKEKIBN7GPJIIWGD2J2IJC6","json":"https://pith.science/pith/UA7BKEKIBN7GPJIIWGD2J2IJC6.json","graph_json":"https://pith.science/api/pith-number/UA7BKEKIBN7GPJIIWGD2J2IJC6/graph.json","events_json":"https://pith.science/api/pith-number/UA7BKEKIBN7GPJIIWGD2J2IJC6/events.json","paper":"https://pith.science/paper/UA7BKEKI"},"agent_actions":{"view_html":"https://pith.science/pith/UA7BKEKIBN7GPJIIWGD2J2IJC6","download_json":"https://pith.science/pith/UA7BKEKIBN7GPJIIWGD2J2IJC6.json","view_paper":"https://pith.science/paper/UA7BKEKI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2510.20963&json=true","fetch_graph":"https://pith.science/api/pith-number/UA7BKEKIBN7GPJIIWGD2J2IJC6/graph.json","fetch_events":"https://pith.science/api/pith-number/UA7BKEKIBN7GPJIIWGD2J2IJC6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UA7BKEKIBN7GPJIIWGD2J2IJC6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UA7BKEKIBN7GPJIIWGD2J2IJC6/action/storage_attestation","attest_author":"https://pith.science/pith/UA7BKEKIBN7GPJIIWGD2J2IJC6/action/author_attestation","sign_citation":"https://pith.science/pith/UA7BKEKIBN7GPJIIWGD2J2IJC6/action/citation_signature","submit_replication":"https://pith.science/pith/UA7BKEKIBN7GPJIIWGD2J2IJC6/action/replication_record"}},"created_at":"2026-07-15T01:21:47.619528+00:00","updated_at":"2026-07-15T01:21:47.619528+00:00"}