{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:G2WELIVODHIE6442QHHCA2LUIR","short_pith_number":"pith:G2WELIVO","schema_version":"1.0","canonical_sha256":"36ac45a2ae19d04f739a81ce206974447f6dd3cd24232f2d105a96d831cb1ae4","source":{"kind":"arxiv","id":"2502.21017","version":2},"attestation_state":"computed","paper":{"title":"PersuasiveToM: A Benchmark for Evaluating Machine Theory of Mind in Persuasive Dialogues","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fangxu Yu, Lai Jiang, Shenyi Huang, Xinyu Dai, Zhen Wu","submitted_at":"2025-02-28T13:04:04Z","abstract_excerpt":"The ability to understand and predict the mental states of oneself and others, known as the Theory of Mind (ToM), is crucial for effective social scenarios. Although recent studies have evaluated ToM in Large Language Models (LLMs), existing benchmarks focus on simplified settings (e.g., Sally-Anne-style tasks) and overlook the complexity of real-world social interactions. To mitigate this gap, we propose PersuasiveToM, a benchmark designed to evaluate the ToM abilities of LLMs in persuasive dialogues. Our framework contains two core tasks: ToM Reasoning, which tests tracking of evolving desir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.21017","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-28T13:04:04Z","cross_cats_sorted":[],"title_canon_sha256":"64562faf21930c1bf37dd9eeda3ff45425373aa135107f17f87705dc8707a432","abstract_canon_sha256":"38f8d24e0a7e4851871966f97ea463c518dbfcb8732f261b1056db92986ccd45"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:09.374487Z","signature_b64":"U3VUPHB6W/vPpRDfP/clrWft8pwgzPaEJd7lID7gbpqIuvRSlML5pIYG5zuPymsdvcyGeOyBY8eOpkt9bSwIBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36ac45a2ae19d04f739a81ce206974447f6dd3cd24232f2d105a96d831cb1ae4","last_reissued_at":"2026-07-05T11:09:09.374002Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:09.374002Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PersuasiveToM: A Benchmark for Evaluating Machine Theory of Mind in Persuasive Dialogues","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fangxu Yu, Lai Jiang, Shenyi Huang, Xinyu Dai, Zhen Wu","submitted_at":"2025-02-28T13:04:04Z","abstract_excerpt":"The ability to understand and predict the mental states of oneself and others, known as the Theory of Mind (ToM), is crucial for effective social scenarios. Although recent studies have evaluated ToM in Large Language Models (LLMs), existing benchmarks focus on simplified settings (e.g., Sally-Anne-style tasks) and overlook the complexity of real-world social interactions. To mitigate this gap, we propose PersuasiveToM, a benchmark designed to evaluate the ToM abilities of LLMs in persuasive dialogues. Our framework contains two core tasks: ToM Reasoning, which tests tracking of evolving desir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.21017","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.21017/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.21017","created_at":"2026-07-05T11:09:09.374067+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.21017v2","created_at":"2026-07-05T11:09:09.374067+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.21017","created_at":"2026-07-05T11:09:09.374067+00:00"},{"alias_kind":"pith_short_12","alias_value":"G2WELIVODHIE","created_at":"2026-07-05T11:09:09.374067+00:00"},{"alias_kind":"pith_short_16","alias_value":"G2WELIVODHIE6442","created_at":"2026-07-05T11:09:09.374067+00:00"},{"alias_kind":"pith_short_8","alias_value":"G2WELIVO","created_at":"2026-07-05T11:09:09.374067+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05330","citing_title":"A Model of Multi-turn Human Persuadability Using Probabilistic Belief Tracing","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31916","citing_title":"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22602","citing_title":"Think Thrice Before You Speak: Dual knowledge-enhanced Theory-of-Mind Reasoning for Persuasive Agents","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10031","citing_title":"CoSToM:Causal-oriented Steering for Intrinsic Theory-of-Mind Alignment in Large Language Models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G2WELIVODHIE6442QHHCA2LUIR","json":"https://pith.science/pith/G2WELIVODHIE6442QHHCA2LUIR.json","graph_json":"https://pith.science/api/pith-number/G2WELIVODHIE6442QHHCA2LUIR/graph.json","events_json":"https://pith.science/api/pith-number/G2WELIVODHIE6442QHHCA2LUIR/events.json","paper":"https://pith.science/paper/G2WELIVO"},"agent_actions":{"view_html":"https://pith.science/pith/G2WELIVODHIE6442QHHCA2LUIR","download_json":"https://pith.science/pith/G2WELIVODHIE6442QHHCA2LUIR.json","view_paper":"https://pith.science/paper/G2WELIVO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.21017&json=true","fetch_graph":"https://pith.science/api/pith-number/G2WELIVODHIE6442QHHCA2LUIR/graph.json","fetch_events":"https://pith.science/api/pith-number/G2WELIVODHIE6442QHHCA2LUIR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G2WELIVODHIE6442QHHCA2LUIR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G2WELIVODHIE6442QHHCA2LUIR/action/storage_attestation","attest_author":"https://pith.science/pith/G2WELIVODHIE6442QHHCA2LUIR/action/author_attestation","sign_citation":"https://pith.science/pith/G2WELIVODHIE6442QHHCA2LUIR/action/citation_signature","submit_replication":"https://pith.science/pith/G2WELIVODHIE6442QHHCA2LUIR/action/replication_record"}},"created_at":"2026-07-05T11:09:09.374067+00:00","updated_at":"2026-07-05T11:09:09.374067+00:00"}