{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IQHMJQDLMZ2IV36PEEOU37WXRM","short_pith_number":"pith:IQHMJQDL","schema_version":"1.0","canonical_sha256":"440ec4c06b66748aefcf211d4dfed78b055a6acb6e2deb78205ce65bd3f74fa8","source":{"kind":"arxiv","id":"2404.13627","version":3},"attestation_state":"computed","paper":{"title":"NegotiationToM: A Benchmark for Stress-testing Machine Theory of Mind on Negotiation Surrounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Cheng Jiayang, Chunkit Chan, Haoran Li, Hongming Zhang, Wei Fan, Weiqi Wang, Xin Liu, Yangqiu Song, Yauwai Yim, Zheye Deng","submitted_at":"2024-04-21T11:51:13Z","abstract_excerpt":"Large Language Models (LLMs) have sparked substantial interest and debate concerning their potential emergence of Theory of Mind (ToM) ability. Theory of mind evaluations currently focuses on testing models using machine-generated data or game settings prone to shortcuts and spurious correlations, which lacks evaluation of machine ToM ability in real-world human interaction scenarios. This poses a pressing demand to develop new real-world scenario benchmarks. We introduce NegotiationToM, a new benchmark designed to stress-test machine ToM in real-world negotiation surrounding covered multi-dim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.13627","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-21T11:51:13Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4f46db1c6fce239f4818602ac3e5a75458b5dfe42fdde5a8eaf043a35338de42","abstract_canon_sha256":"3253bdf1daec287a318186f92404cd6fc84585bf6e9e79b8a97781cd06340d19"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:14.587132Z","signature_b64":"jZtoU1xkhfCTCqlO8Cq0CmaGmvJmr18bejWmWmP6RAXNRQ2pQXC5BoEyapZBK/BuWzuzA7VoBZa08ygLluWxCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"440ec4c06b66748aefcf211d4dfed78b055a6acb6e2deb78205ce65bd3f74fa8","last_reissued_at":"2026-07-05T09:16:14.586581Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:14.586581Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NegotiationToM: A Benchmark for Stress-testing Machine Theory of Mind on Negotiation Surrounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Cheng Jiayang, Chunkit Chan, Haoran Li, Hongming Zhang, Wei Fan, Weiqi Wang, Xin Liu, Yangqiu Song, Yauwai Yim, Zheye Deng","submitted_at":"2024-04-21T11:51:13Z","abstract_excerpt":"Large Language Models (LLMs) have sparked substantial interest and debate concerning their potential emergence of Theory of Mind (ToM) ability. Theory of mind evaluations currently focuses on testing models using machine-generated data or game settings prone to shortcuts and spurious correlations, which lacks evaluation of machine ToM ability in real-world human interaction scenarios. This poses a pressing demand to develop new real-world scenario benchmarks. We introduce NegotiationToM, a new benchmark designed to stress-test machine ToM in real-world negotiation surrounding covered multi-dim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.13627","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.13627/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.13627","created_at":"2026-07-05T09:16:14.586655+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.13627v3","created_at":"2026-07-05T09:16:14.586655+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.13627","created_at":"2026-07-05T09:16:14.586655+00:00"},{"alias_kind":"pith_short_12","alias_value":"IQHMJQDLMZ2I","created_at":"2026-07-05T09:16:14.586655+00:00"},{"alias_kind":"pith_short_16","alias_value":"IQHMJQDLMZ2IV36P","created_at":"2026-07-05T09:16:14.586655+00:00"},{"alias_kind":"pith_short_8","alias_value":"IQHMJQDL","created_at":"2026-07-05T09:16:14.586655+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04184","citing_title":"GroupToM-Bench: Benchmarking Group Theory of Mind and Nonlinear Social Emergence in MLLMs","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31916","citing_title":"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22602","citing_title":"Think Thrice Before You Speak: Dual knowledge-enhanced Theory-of-Mind Reasoning for Persuasive Agents","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IQHMJQDLMZ2IV36PEEOU37WXRM","json":"https://pith.science/pith/IQHMJQDLMZ2IV36PEEOU37WXRM.json","graph_json":"https://pith.science/api/pith-number/IQHMJQDLMZ2IV36PEEOU37WXRM/graph.json","events_json":"https://pith.science/api/pith-number/IQHMJQDLMZ2IV36PEEOU37WXRM/events.json","paper":"https://pith.science/paper/IQHMJQDL"},"agent_actions":{"view_html":"https://pith.science/pith/IQHMJQDLMZ2IV36PEEOU37WXRM","download_json":"https://pith.science/pith/IQHMJQDLMZ2IV36PEEOU37WXRM.json","view_paper":"https://pith.science/paper/IQHMJQDL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.13627&json=true","fetch_graph":"https://pith.science/api/pith-number/IQHMJQDLMZ2IV36PEEOU37WXRM/graph.json","fetch_events":"https://pith.science/api/pith-number/IQHMJQDLMZ2IV36PEEOU37WXRM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IQHMJQDLMZ2IV36PEEOU37WXRM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IQHMJQDLMZ2IV36PEEOU37WXRM/action/storage_attestation","attest_author":"https://pith.science/pith/IQHMJQDLMZ2IV36PEEOU37WXRM/action/author_attestation","sign_citation":"https://pith.science/pith/IQHMJQDLMZ2IV36PEEOU37WXRM/action/citation_signature","submit_replication":"https://pith.science/pith/IQHMJQDLMZ2IV36PEEOU37WXRM/action/replication_record"}},"created_at":"2026-07-05T09:16:14.586655+00:00","updated_at":"2026-07-05T09:16:14.586655+00:00"}