{"as_of":"2026-08-09T10:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5f4e390e6918d089e194a3984184f279fad32a8c2d4938081c1afae81ac776a7","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:00:46.176692Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T16:39:57.697910Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-08-07T12:00:46.176692Z","title":"Yun-Shiuan Chuang, Siddharth Suresh, Nikunj Harlalka, Agam Goyal, Robert Hawkins, Sijia Yang, Dhavan Shah, Junjie Hu, and Timothy T","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00900","last_updated":"2025-06-01T08:36:51Z","snapshot_observed_at":"2026-08-07T11:53:02.038417Z","submitted_at":"2025-06-01T08:36:51Z","title":"SocialEval: Evaluating Social Intelligence of Large Language Models","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T12:00:46.176692Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2506.00900"},"observation_digest":"sha256:c8b91a5b54515aeb195dff635c1830b2780da17f016b7f30d67c2c858c990d29","observation_id":"096fb5c0-c6ed-41d8-b1f3-d3a05a82b008","resolution":{"observed_at":"2026-08-07T12:00:46.176692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-08-07T11:27:24.242068Z","title":"Jiayang Cheng, Lin Qiu, Tsz Ho Chan, Tianqing Fang, Weiqi Wang, Chunkit Chan, Dongyu Ru, Qipeng Guo, Hongming Zhang, Yangqiu Song, Yue Zhang, and Zheng Zhang","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02461","last_updated":"2025-06-03T05:23:25Z","snapshot_observed_at":"2026-08-09T06:40:39.583356Z","submitted_at":"2025-06-03T05:23:25Z","title":"XToM: Exploring the Multilingual Theory of Mind for Large Language Models","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T11:27:24.242068Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2506.02461"},"observation_digest":"sha256:cda1adbb81c5b2f674079c69266bda3bbed8680cebb2fe07a72120d74e8cbc72","observation_id":"70729dab-1315-4317-ac4d-b0a6e649ebde","resolution":{"observed_at":"2026-08-07T11:27:24.242068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-08-07T04:51:29.167252Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09450","last_updated":"2025-06-11T06:55:40Z","snapshot_observed_at":"2026-08-09T03:53:25.809004Z","submitted_at":"2025-06-11T06:55:40Z","title":"UniToMBench: Integrating Perspective-Taking to Improve Theory of Mind in LLMs","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T04:51:29.167252Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2506.09450"},"observation_digest":"sha256:75a886f2b156aca16e9bef464f2b59c2ddc6dfc2a5695af6f582f80e41a5fa0e","observation_id":"6c4fd5b4-864e-404f-9bd9-99e845e896e4","resolution":{"observed_at":"2026-08-07T04:51:29.167252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-08-06T22:49:37.997496Z","title":"Tombench: Benchmarking theory of mind in large language models.arXiv preprint arXiv:2402.15052,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:37.997496Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:1ff838ac8ad0a76d3284898b73795c4b59a2b3ebf189fa186351293abadf3574","observation_id":"4e7260fd-7c34-4e43-9e8a-116036605cd2","resolution":{"observed_at":"2026-08-06T22:49:37.997496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-08-06T20:13:25.488814Z","title":"arXiv preprint arXiv:2402.15052 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.03543","last_updated":"2025-07-04T12:50:43Z","snapshot_observed_at":"2026-08-09T03:48:55.347454Z","submitted_at":"2025-07-04T12:50:43Z","title":"H2HTalk: Evaluating Large Language Models as Emotional Companion","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T20:13:25.488814Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2507.03543"},"observation_digest":"sha256:3a533e24dacb0dafb63d0952517ca4b09aa4c44c31e2ec8eb796bf10a85e3f3d","observation_id":"aa3e38bd-9013-4341-a854-eefd00f70df9","resolution":{"observed_at":"2026-08-06T20:13:25.488814Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-08-05T11:46:53.197894Z","title":"Preprint, arXiv:2402.15052","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.02292","last_updated":"2026-06-28T02:36:12Z","snapshot_observed_at":"2026-08-05T11:46:52.768597Z","submitted_at":"2025-09-02T13:11:24Z","title":"LLMs and their Limited Theory of Mind: Evaluating Mental State Annotations in Situated Dialogue","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-05T11:46:53.197894Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2509.02292"},"observation_digest":"sha256:fc92134327ed0897fb84f4cc49eba5587ba38d1cf7e8b3adf5697651ce6d888d","observation_id":"225b74d8-5bc3-4b11-9035-eabe25557369","resolution":{"observed_at":"2026-08-05T11:46:53.197894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-08-04T12:42:02.985944Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.02660","last_updated":"2026-06-09T21:44:21Z","snapshot_observed_at":"2026-08-08T15:08:35.616665Z","submitted_at":"2025-10-03T01:37:32Z","title":"When Researchers Say Mental Model/Theory of Mind of AI, What Are They Really Talking About?","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T12:42:02.985944Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2510.02660"},"observation_digest":"sha256:a0cc89f77f3f90f0ef215dba119d202d1fdb0874f4f4b0fa2243fcc4bf3ab0c0","observation_id":"de63b124-760f-4767-aaef-880f32043c7c","resolution":{"observed_at":"2026-08-04T12:42:02.985944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":"2402.15052","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-07-04T16:39:57.697910Z","title":"arXiv preprint arXiv:2402.15052 , year=","venue":null,"work_id":"9a7c86bb-c7c7-48d1-b2e7-f3b4e5fce5d9","year":2026},"citing_paper":{"arxiv_id":"2604.17989","last_updated":"2026-04-20T09:12:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-20T09:12:47Z","title":"AIT Academy: Cultivating the Complete Agent with a Confucian Three-Domain Curriculum","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T05:08:54.560648Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2604.17989"},"observation_digest":"sha256:ef32606cd8791856e14e41cad2125ca684bae53ddb252c06bbe816c32a5060c7","observation_id":"017fe9d4-f3df-4f66-9bb6-2c19ab37ac0e","resolution":{"observed_at":"2026-05-10T09:43:49.848790Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":"2402.15052","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-07-04T16:39:57.697910Z","title":"arXiv preprint arXiv:2402.15052 , year=","venue":null,"work_id":"9a7c86bb-c7c7-48d1-b2e7-f3b4e5fce5d9","year":2026},"citing_paper":{"arxiv_id":"2604.20749","last_updated":"2026-04-22T16:39:52Z","snapshot_observed_at":"2026-07-06T23:07:25.185196Z","submitted_at":"2026-04-22T16:39:52Z","title":"Where and What: Reasoning Dynamic and Implicit Preferences in Situated Conversational Recommendation","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-05-09T23:52:22.319870Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2604.20749"},"observation_digest":"sha256:5a79585aa972e897f10333397286de773dc78b550336b46682c621e1dd12c773","observation_id":"cd73f55d-325c-43f4-8121-7eaab50b55a5","resolution":{"observed_at":"2026-05-11T13:56:05.796036Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":"2402.15052","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-07-04T16:39:57.697910Z","title":"arXiv preprint arXiv:2402.15052 , year=","venue":null,"work_id":"9a7c86bb-c7c7-48d1-b2e7-f3b4e5fce5d9","year":2026},"citing_paper":{"arxiv_id":"2605.03149","last_updated":"2026-05-04T20:42:16Z","snapshot_observed_at":"2026-07-06T23:16:05.372336Z","submitted_at":"2026-05-04T20:42:16Z","title":"Are you with me? A Framework for Detecting Mental Model Discrepancies in Task-Based Team Dialogues","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-08T18:19:57.347494Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2605.03149"},"observation_digest":"sha256:0e6d65b6d3e1dfe251a8fda2f50c9dcbbd56527b46f2d37eb0c8b412a8792d27","observation_id":"9bc8158f-526b-4759-9fc6-9785713b76ab","resolution":{"observed_at":"2026-05-09T06:35:38.897667Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":"2402.15052","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-07-04T16:39:57.697910Z","title":"arXiv preprint arXiv:2402.15052 , year=","venue":null,"work_id":"9a7c86bb-c7c7-48d1-b2e7-f3b4e5fce5d9","year":2026},"citing_paper":{"arxiv_id":"2605.20423","last_updated":"2026-05-19T19:19:26Z","snapshot_observed_at":"2026-08-02T08:56:18.389445Z","submitted_at":"2026-05-19T19:19:26Z","title":"OSCToM: RL-Guided Adversarial Generation for High-Order Theory of Mind","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-21T07:09:37.399954Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2605.20423"},"observation_digest":"sha256:8c9ed193fd2620c2c228c5c5f5c0d3987f11fa637bf500defdc9cc139ad7f943","observation_id":"5ee50c9b-2199-4444-828f-204af964f158","resolution":{"observed_at":"2026-05-21T07:09:46.252198Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":"2402.15052","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-07-04T16:39:57.697910Z","title":"arXiv preprint arXiv:2402.15052 , year=","venue":null,"work_id":"9a7c86bb-c7c7-48d1-b2e7-f3b4e5fce5d9","year":2026},"citing_paper":{"arxiv_id":"2605.20506","last_updated":"2026-05-19T21:23:24Z","snapshot_observed_at":"2026-08-03T04:27:59.284440Z","submitted_at":"2026-05-19T21:23:24Z","title":"Reinforcing Human Behavior Simulation via Verbal Feedback","version":1},"reference_index":156,"source":"arxiv_source","source_observed_at":"2026-05-21T07:21:48.649289Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2605.20506"},"observation_digest":"sha256:473d7db047ccd3695a32e58bf7fa215f8d8549601ca1c3c484450b1ada519e6e","observation_id":"e999a52a-b47f-41d9-9018-4f6ad92da217","resolution":{"observed_at":"2026-05-21T07:24:02.458142Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":"2402.15052","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-07-04T16:39:57.697910Z","title":"arXiv preprint arXiv:2402.15052 , year=","venue":null,"work_id":"9a7c86bb-c7c7-48d1-b2e7-f3b4e5fce5d9","year":2026},"citing_paper":{"arxiv_id":"2606.21315","last_updated":"2026-06-19T10:55:30Z","snapshot_observed_at":"2026-08-06T19:59:05.825917Z","submitted_at":"2026-06-19T10:55:30Z","title":"Social World Model for Lifelong Social Intelligence","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-26T14:29:11.743627Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2606.21315"},"observation_digest":"sha256:be52181d8225e242a5729d84872387252a85f4ba1cb01eabfe5aaeccb21e627b","observation_id":"a09f3e32-ca1f-419c-bb34-e7844e7e3107","resolution":{"observed_at":"2026-07-04T06:29:37.152307Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":"2402.15052","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-07-04T16:39:57.697910Z","title":"arXiv preprint arXiv:2402.15052 , year=","venue":null,"work_id":"9a7c86bb-c7c7-48d1-b2e7-f3b4e5fce5d9","year":2026},"citing_paper":{"arxiv_id":"2606.24267","last_updated":"2026-07-14T05:22:07Z","snapshot_observed_at":"2026-08-02T08:24:58.918165Z","submitted_at":"2026-06-23T07:52:22Z","title":"Pigeonholing: how bad prompts hurt models, causing collapse and mistakes","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-26T00:22:34.367119Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2606.24267"},"observation_digest":"sha256:d4df3d8bd8234658ecc1313504e588586e0035479c0f1892cb22560f8e8e38b2","observation_id":"a5d78b53-6416-4a3c-bd8b-7996d8f73c68","resolution":{"observed_at":"2026-07-04T16:39:57.699869Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-07-15T10:35:09.399899Z","title":"Myra Cheng, Cinoo Lee, Pranav Khadpe, Sunny Yu, Dyllan Han, and Dan Jurafsky","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.24267","last_updated":"2026-07-14T05:22:07Z","snapshot_observed_at":"2026-08-02T08:24:58.918165Z","submitted_at":"2026-06-23T07:52:22Z","title":"Pigeonholing: how bad prompts hurt models, causing collapse and mistakes","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-15T10:35:09.399899Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2606.24267"},"observation_digest":"sha256:855cff6845982ad8a8148ba346b689e39a64a05fd7320a94dba54d311f566a90","observation_id":"5c4dded9-82ea-4717-99f6-3212b66dfe39","resolution":{"observed_at":"2026-07-15T10:35:09.399899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-07-30T11:07:38.423077Z","title":"ToMBench : Benchmarking theory of mind in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.27201","last_updated":"2026-07-29T17:59:39Z","snapshot_observed_at":"2026-08-07T07:01:04.657816Z","submitted_at":"2026-07-29T17:59:39Z","title":"Mental World Modeling","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-07-30T11:07:38.423077Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2607.27201"},"observation_digest":"sha256:8c1e5cb09d2ca5b57a5c0b0f81a2073b0776c837758e13db8616040401f30548","observation_id":"29f06851-5dc7-4c70-875c-5fbd2e7f7f8c","resolution":{"observed_at":"2026-07-30T11:07:38.423077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-08-01T08:29:29.096107Z","title":"arXiv preprint arXiv:2402.15052 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27379","last_updated":"2026-07-29T18:37:14Z","snapshot_observed_at":"2026-08-06T22:55:57.490924Z","submitted_at":"2026-07-29T18:37:14Z","title":"HSS-Synth: Humanities and Social Sciences Data Synthesis for LLMs","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-01T08:29:29.096107Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2607.27379"},"observation_digest":"sha256:d9dc3051537fed9e7382002448dcccf1b1d79836cf91a8b9e0278056e3c2bf16","observation_id":"1d3a5ffb-3472-4559-8c2a-53f637b92169","resolution":{"observed_at":"2026-08-01T08:29:29.096107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.15052/citation-record","integrity":"/paper/2402.15052/integrity","json":"/paper/2402.15052/citation-record.json","paper":"/paper/2402.15052"},"outbound":[],"paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 17 inbound Pith citation observations for arXiv:2402.15052."}