{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IJ3UFEIJLAZD2DWPV74KN2R2YV","short_pith_number":"pith:IJ3UFEIJ","schema_version":"1.0","canonical_sha256":"427742910958323d0ecfaff8a6ea3ac5493d2ae97780619867bbcfb570153d66","source":{"kind":"arxiv","id":"2404.10642","version":3},"attestation_state":"computed","paper":{"title":"Self-playing Adversarial Language Game Enhances LLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Han Xu, Lei Han, Nan Du, Pengyu Cheng, Tianhao Hu, Xiaolong Li, Yong Dai, Zheng Yuan, Zhisong Zhang","submitted_at":"2024-04-16T15:16:22Z","abstract_excerpt":"We explore the potential of self-play training for large language models (LLMs) in a two-player adversarial language game called Adversarial Taboo. In this game, an attacker and a defender communicate around a target word only visible to the attacker. The attacker aims to induce the defender to speak the target word unconsciously, while the defender tries to infer the target word from the attacker's utterances. To win the game, both players must have sufficient knowledge about the target word and high-level reasoning ability to infer and express in this information-reserved conversation. Hence"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.10642","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-16T15:16:22Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"acf4c5e88b6feb7959f78dfa8e7d6aedca68941c82d9faba1b8e122ab23be4ca","abstract_canon_sha256":"6386789483fadf8d2ca32304aa3cfa245144c57bbe6b4fac5ee709689755cf86"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:56.602563Z","signature_b64":"ZWrNdEaMWOq9xCa5NZhwnEv26rKHW7BkwN/KFe7zfGG2Txvc+ZsOpCC+paDtTyOgklSVUOlc3hLDTvXBCK2LCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"427742910958323d0ecfaff8a6ea3ac5493d2ae97780619867bbcfb570153d66","last_reissued_at":"2026-07-05T10:04:56.602081Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:56.602081Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-playing Adversarial Language Game Enhances LLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Han Xu, Lei Han, Nan Du, Pengyu Cheng, Tianhao Hu, Xiaolong Li, Yong Dai, Zheng Yuan, Zhisong Zhang","submitted_at":"2024-04-16T15:16:22Z","abstract_excerpt":"We explore the potential of self-play training for large language models (LLMs) in a two-player adversarial language game called Adversarial Taboo. In this game, an attacker and a defender communicate around a target word only visible to the attacker. The attacker aims to induce the defender to speak the target word unconsciously, while the defender tries to infer the target word from the attacker's utterances. To win the game, both players must have sufficient knowledge about the target word and high-level reasoning ability to infer and express in this information-reserved conversation. Hence"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.10642","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.10642/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.10642","created_at":"2026-07-05T10:04:56.602143+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.10642v3","created_at":"2026-07-05T10:04:56.602143+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.10642","created_at":"2026-07-05T10:04:56.602143+00:00"},{"alias_kind":"pith_short_12","alias_value":"IJ3UFEIJLAZD","created_at":"2026-07-05T10:04:56.602143+00:00"},{"alias_kind":"pith_short_16","alias_value":"IJ3UFEIJLAZD2DWP","created_at":"2026-07-05T10:04:56.602143+00:00"},{"alias_kind":"pith_short_8","alias_value":"IJ3UFEIJ","created_at":"2026-07-05T10:04:56.602143+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20051","citing_title":"Bootstrapping Post-training Signals for Open-ended Tasks via Rubric-based Self-play on Pre-training Text","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IJ3UFEIJLAZD2DWPV74KN2R2YV","json":"https://pith.science/pith/IJ3UFEIJLAZD2DWPV74KN2R2YV.json","graph_json":"https://pith.science/api/pith-number/IJ3UFEIJLAZD2DWPV74KN2R2YV/graph.json","events_json":"https://pith.science/api/pith-number/IJ3UFEIJLAZD2DWPV74KN2R2YV/events.json","paper":"https://pith.science/paper/IJ3UFEIJ"},"agent_actions":{"view_html":"https://pith.science/pith/IJ3UFEIJLAZD2DWPV74KN2R2YV","download_json":"https://pith.science/pith/IJ3UFEIJLAZD2DWPV74KN2R2YV.json","view_paper":"https://pith.science/paper/IJ3UFEIJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.10642&json=true","fetch_graph":"https://pith.science/api/pith-number/IJ3UFEIJLAZD2DWPV74KN2R2YV/graph.json","fetch_events":"https://pith.science/api/pith-number/IJ3UFEIJLAZD2DWPV74KN2R2YV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IJ3UFEIJLAZD2DWPV74KN2R2YV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IJ3UFEIJLAZD2DWPV74KN2R2YV/action/storage_attestation","attest_author":"https://pith.science/pith/IJ3UFEIJLAZD2DWPV74KN2R2YV/action/author_attestation","sign_citation":"https://pith.science/pith/IJ3UFEIJLAZD2DWPV74KN2R2YV/action/citation_signature","submit_replication":"https://pith.science/pith/IJ3UFEIJLAZD2DWPV74KN2R2YV/action/replication_record"}},"created_at":"2026-07-05T10:04:56.602143+00:00","updated_at":"2026-07-05T10:04:56.602143+00:00"}