{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RJZZKEUBX47JLRCWLA6OKNXJFN","short_pith_number":"pith:RJZZKEUB","schema_version":"1.0","canonical_sha256":"8a73951281bf3e95c456583ce536e92b5af34884c992b2771af9780f8e6e3d9c","source":{"kind":"arxiv","id":"2401.08743","version":2},"attestation_state":"computed","paper":{"title":"MMToM-QA: Multimodal Theory of Mind Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.AI","authors_text":"Antonio Torralba, Chuanyang Jin, Jiannan Xiang, Jing Cao, Joshua B. Tenenbaum, Tianmin Shu, Tomer Ullman, Yen-Ling Kuo, Yutong Wu, Zhiting Hu","submitted_at":"2024-01-16T18:59:24Z","abstract_excerpt":"Theory of Mind (ToM), the ability to understand people's mental states, is an essential ingredient for developing machines with human-level social intelligence. Recent machine learning models, particularly large language models, seem to show some aspects of ToM understanding. However, existing ToM benchmarks use unimodal datasets - either video or text. Human ToM, on the other hand, is more than video or text understanding. People can flexibly reason about another person's mind based on conceptual representations (e.g., goals, beliefs, plans) extracted from any available data. To address this,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.08743","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-01-16T18:59:24Z","cross_cats_sorted":["cs.CL","cs.CV","cs.LG"],"title_canon_sha256":"039f72a2377b58ac8ad0448e279d8a0a995e1a1e08cd75266e1fc48be463131c","abstract_canon_sha256":"12346fe14dbcc77fd1fb9031b8585188ab677a8c4b70d1c11671589605d12678"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:10.588512Z","signature_b64":"F5QYeqrs1qcyHLbKl77PNZpsxhW/Ajyl/UM60lHdPUkGPAo10ncu30GV+W+jNS9eYDd9zk+SzC6vpOkNS7I4Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a73951281bf3e95c456583ce536e92b5af34884c992b2771af9780f8e6e3d9c","last_reissued_at":"2026-07-05T08:32:10.587980Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:10.587980Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMToM-QA: Multimodal Theory of Mind Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.AI","authors_text":"Antonio Torralba, Chuanyang Jin, Jiannan Xiang, Jing Cao, Joshua B. Tenenbaum, Tianmin Shu, Tomer Ullman, Yen-Ling Kuo, Yutong Wu, Zhiting Hu","submitted_at":"2024-01-16T18:59:24Z","abstract_excerpt":"Theory of Mind (ToM), the ability to understand people's mental states, is an essential ingredient for developing machines with human-level social intelligence. Recent machine learning models, particularly large language models, seem to show some aspects of ToM understanding. However, existing ToM benchmarks use unimodal datasets - either video or text. Human ToM, on the other hand, is more than video or text understanding. People can flexibly reason about another person's mind based on conceptual representations (e.g., goals, beliefs, plans) extracted from any available data. To address this,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.08743","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.08743/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.08743","created_at":"2026-07-05T08:32:10.588038+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.08743v2","created_at":"2026-07-05T08:32:10.588038+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.08743","created_at":"2026-07-05T08:32:10.588038+00:00"},{"alias_kind":"pith_short_12","alias_value":"RJZZKEUBX47J","created_at":"2026-07-05T08:32:10.588038+00:00"},{"alias_kind":"pith_short_16","alias_value":"RJZZKEUBX47JLRCW","created_at":"2026-07-05T08:32:10.588038+00:00"},{"alias_kind":"pith_short_8","alias_value":"RJZZKEUB","created_at":"2026-07-05T08:32:10.588038+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.15887","citing_title":"Mind the Motions: Benchmarking Theory-of-Mind in Everyday Body Language","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20506","citing_title":"Reinforcing Human Behavior Simulation via Verbal Feedback","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17819","citing_title":"PDDL-Mind: Large Language Models are Capable on Belief Reasoning with Reliable State Tracking","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RJZZKEUBX47JLRCWLA6OKNXJFN","json":"https://pith.science/pith/RJZZKEUBX47JLRCWLA6OKNXJFN.json","graph_json":"https://pith.science/api/pith-number/RJZZKEUBX47JLRCWLA6OKNXJFN/graph.json","events_json":"https://pith.science/api/pith-number/RJZZKEUBX47JLRCWLA6OKNXJFN/events.json","paper":"https://pith.science/paper/RJZZKEUB"},"agent_actions":{"view_html":"https://pith.science/pith/RJZZKEUBX47JLRCWLA6OKNXJFN","download_json":"https://pith.science/pith/RJZZKEUBX47JLRCWLA6OKNXJFN.json","view_paper":"https://pith.science/paper/RJZZKEUB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.08743&json=true","fetch_graph":"https://pith.science/api/pith-number/RJZZKEUBX47JLRCWLA6OKNXJFN/graph.json","fetch_events":"https://pith.science/api/pith-number/RJZZKEUBX47JLRCWLA6OKNXJFN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RJZZKEUBX47JLRCWLA6OKNXJFN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RJZZKEUBX47JLRCWLA6OKNXJFN/action/storage_attestation","attest_author":"https://pith.science/pith/RJZZKEUBX47JLRCWLA6OKNXJFN/action/author_attestation","sign_citation":"https://pith.science/pith/RJZZKEUBX47JLRCWLA6OKNXJFN/action/citation_signature","submit_replication":"https://pith.science/pith/RJZZKEUBX47JLRCWLA6OKNXJFN/action/replication_record"}},"created_at":"2026-07-05T08:32:10.588038+00:00","updated_at":"2026-07-05T08:32:10.588038+00:00"}