{"as_of":"2026-08-08T02:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cd3158fa27cd3516a0a2ba762a6071fbb6c67068b0fd06d12462d3248e345c3b","coverage":[{"denominator":59,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":59,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T01:28:59.297634Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T13:53:46.937166Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.17574","snapshot_observed_at":"2026-08-01T13:53:46.937166Z","title":"Deepinsight: A unified evaluation infrastructure across the physical ai stack, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.18985","last_updated":"2026-07-25T09:51:58Z","snapshot_observed_at":"2026-08-05T19:56:11.121302Z","submitted_at":"2026-07-21T11:19:21Z","title":"Athena-Brain Technical Report: An Efficient Robot Brain for General Intelligence and Embodied Interaction","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-01T13:53:46.937166Z"},"links":{"cited_paper":"/paper/2606.17574","citing_paper":"/paper/2607.18985"},"observation_digest":"sha256:cefeae24cc680ce5cd52bb4cac9207c2fd25a12c7ce686bcafb3d64595eda6fd","observation_id":"cf181efd-5be5-4ada-8ec4-6ade15169a2e","resolution":{"observed_at":"2026-08-01T13:53:46.937166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2606.17574/citation-record","integrity":"/paper/2606.17574/integrity","json":"/paper/2606.17574/citation-record.json","paper":"/paper/2606.17574"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Introducing helix 02: Full-body autonomy.https://www.figure.ai/news/helix-02, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:5aabe5e8d0b449b1ec5d53fe93f58651a3d3c001a9e1b17abb6dfac5c8c1f543","observation_id":"9d0c4706-c772-476e-9fd2-7021a3ae3beb","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:68258da5488dce58cebcf2767d90651c4725ce1463ca214c165463b5fc657b4a","observation_id":"3a4d40f1-8eaf-4dba-8d94-5aae873f49ca","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":"2110.14168","doi":"10.1002/j.1545-","metadata_source":"pith","pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training Verifiers to Solve Math Word Problems","venue":"cs.LG","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","year":2021},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:4a7fc93a983905ca4e7abf2922fa8bafb0435566c19b6f8b5d831af6ccc184bb","observation_id":"48078aec-2fea-4d24-aff6-87aebdebd11f","resolution":{"observed_at":"2026-07-03T20:18:56.519889Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":"2107.03374","doi":"10.48550/arxiv.2107.03374","metadata_source":"pith","pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating Large Language Models Trained on Code","venue":"cs.LG","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:61bf4ae7891b50d96f30941ab4abbf2df3a65fbb86f25a2ee0c9b57d7c0664c3","observation_id":"c84b1117-b8b6-49c7-8ee3-83dc731d9542","resolution":{"observed_at":"2026-07-03T20:18:56.535162Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Jimenez, John Yang, Alexander Wettig, Shunyu Yao, Kexin Pei, Ofir Press, and Karthik Narasimhan","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:01093de4d9f3346739c582c887a4661a0d183199d27b6375e30b1a31ff64777b","observation_id":"fe194f51-4a96-4743-bdf1-2b51635f6ba6","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"GAIA: A benchmark for general ai assistants","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:6bc85a1cf786fbdd065a473379a106cf7f3bedd292b8300c1930d3e003b507e7","observation_id":"62ea7505-cd8e-45bb-a08b-a7a49c4c33f9","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"OSWorld: Benchmarking multimodal agents for open-ended tasks in real computer environments","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:9f7f9c6e01ce685891ac02eb35e69cd4de738591be9842bd9e103235d2f01874","observation_id":"b3cb0d74-26c9-41e2-be37-23f4c6ebe0f3","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-02T22:19:29.043854Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":"2406.12045","doi":"10.48550/arxiv.2406.12045","metadata_source":"pith","pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","venue":"cs.AI","work_id":"6a8d8dc4-0cc0-4052-8109-abbcdcd4a962","year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:95c038d085d80d5ba4308d61fc7a742cc8a8f7341ab869da1aa767d8689b92ea","observation_id":"7c39ab67-c49a-4b2d-93c0-5a9e8e4f8e5d","resolution":{"observed_at":"2026-07-03T20:18:56.541207Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Xu, Hao Zhu, Xuhui Zhou, Robert Lo, Abishek Sridhar, Xianyi Cheng, Tianyue Ou, Yonatan Bisk, Daniel Fried, Uri Alon, and Graham Neubig","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:feed2ef11d9b131d650f89287fda3f0ef224b9d4fb3e01bf79c7edfddc974e60","observation_id":"000c0c81-3f88-4ea4-843e-be737591c6bc","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"CALVIN: A benchmark for language-conditioned policy learning for long-horizon robot manipulation tasks.IEEE Robotics and Automation Letters, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:d3ac8bb3830f1764788e676a3b7a33a5415494bd95e1d5f404429a2f7c299079","observation_id":"2c8fdb92-4fcb-4e3d-84ea-6387fc01c00a","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"LIBERO: Benchmarking knowledge transfer for lifelong robot learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:1b72d2c207b152b8f2491cceeee0fbf7962b9364b79dab22ca170785dd2b19c9","observation_id":"f023eaf6-12c2-44f8-af86-73a8fc803482","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Meta-world: A benchmark and evaluation for multi-task and meta reinforcement learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:9ef6338d75d111553b8e659a08df3650581c487d02e31a618c5d48fc3ec51d5b","observation_id":"ffb73923-4378-425c-b7fa-3740155cf60d","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:14d2d97b9b447645fef4f390334bd26bf090babd3fc4f77672909bd9727a31a9","observation_id":"803db83d-5b69-4758-9fb2-a62484252f7e","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08864","last_updated":"2025-05-14T15:22:36Z","snapshot_observed_at":"2026-08-08T01:53:43.555672Z","submitted_at":"2023-10-13T05:20:40Z","title":"Open X-Embodiment: Robotic Learning Datasets and RT-X Models","version":9},"cited_work":{"arxiv_id":"2310.08864","doi":"10.48550/arxiv.2310.08864","metadata_source":"pith","pith_arxiv_id":"2310.08864","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open X-Embodiment: Robotic Learning Datasets and RT-X Models","venue":"cs.RO","work_id":"62f0fb6c-e6ae-4dc4-95a4-d9dd64b240e8","year":2023},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2310.08864","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:c30272e9742b98a16f823f41878d2ab7c75d7ca47e0a12c6e174f0e2bbde7592","observation_id":"9e197a75-b186-4970-92c4-6063ff075bd2","resolution":{"observed_at":"2026-07-03T20:18:56.572379Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.05941","last_updated":"2024-05-09T17:30:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-09T17:30:16Z","title":"Evaluating Real-World Robot Manipulation Policies in Simulation","version":1},"cited_work":{"arxiv_id":"2405.05941","doi":"10.48550/arxiv.2405.05941","metadata_source":"pith","pith_arxiv_id":"2405.05941","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating Real-World Robot Manipulation Policies in Simulation","venue":"cs.RO","work_id":"7f4ca6cb-1b94-454c-9623-b52441b74b61","year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2405.05941","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:c355e9846225ee3bf55d9e27d1dfad58c1e20db3dd29609e4ae191241a8dbffa","observation_id":"2842f995-f892-42d8-a907-f08737ddf8d1","resolution":{"observed_at":"2026-07-03T20:18:56.513442Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.10506","last_updated":"2024-06-18T18:11:07Z","snapshot_observed_at":"2026-08-04T05:16:12.953751Z","submitted_at":"2024-03-15T17:45:44Z","title":"HumanoidBench: Simulated Humanoid Benchmark for Whole-Body Locomotion and Manipulation","version":2},"cited_work":{"arxiv_id":"2403.10506","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.10506","snapshot_observed_at":"2026-07-04T19:40:06.916232Z","title":"Humanoid- bench: Simulated humanoid benchmark for whole-body locomotion and manipulation.arXiv preprint arXiv:2403.10506","venue":null,"work_id":"8e42d104-ddc6-4291-b477-f09257aa5c80","year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2403.10506","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:1daa2864cfb6248a4ac996ed7531cfda09f1e0e4af6b25ada5e7d83ba22ac34f","observation_id":"0c305888-fd76-44de-a7fa-b8843614417f","resolution":{"observed_at":"2026-07-03T20:18:56.580913Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"RoboHive: A unified framework for robot learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:4ea2a847bc679a6aa8fcabbb81ce44883bab4f806836ff671bb2902cd2f41bb8","observation_id":"3cbf843e-fd60-4ace-9808-6ce7c2b1d27f","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Orbit: A unified simulation framework for interactive robot learning environments.IEEE Robotics and Automation Letters, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:0308d4348f12302d55fba2939770c56b3d37343c93d2e56f83e7cb83dfa604e2","observation_id":"f1fc877c-7310-4b1b-ba20-2cfd3cdde422","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Michaelov, Hanwool Albert Lee, Janna, Leonid Sinev, Khalid, Kiersten Stokes, Zden ˇek Kasner, and KonradSzafer","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:cd036143321394fe47a86e1562aa4020a6dcae7ea0f804435b9014e2144f2d38","observation_id":"fc8c09d1-53bb-4cfa-b45e-63f95be25cd9","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"OpenCompass: A universal evaluation platform for foundation models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:7946cd7d7b395c05e37a1cc7b98c45ea6eafde92e970829caef2ec074bd5f34f","observation_id":"3c36b662-255b-401e-84b3-a806ef671913","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09110","last_updated":"2023-10-01T21:44:23Z","snapshot_observed_at":"2026-08-01T19:14:56.803459Z","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models","version":2},"cited_work":{"arxiv_id":"2211.09110","doi":"10.1007/bf01194075","metadata_source":"pith","pith_arxiv_id":"2211.09110","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Holistic Evaluation of Language Models","venue":"cs.CL","work_id":"cc02a01e-7218-47dc-8e66-3333e7e4adec","year":2022},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2211.09110","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:5b8ee38d7918eac2e1f5701f59329142503fc284b513bf9146d276daeb9a4b4a","observation_id":"e9222d0e-af71-4f2e-a0cd-697abb9fe1f0","resolution":{"observed_at":"2026-07-03T20:18:56.587390Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11691","last_updated":"2025-08-28T09:40:49Z","snapshot_observed_at":"2026-07-06T18:47:08.782799Z","submitted_at":"2024-07-16T13:06:15Z","title":"VLMEvalKit: An Open-Source Toolkit for Evaluating Large Multi-Modality Models","version":4},"cited_work":{"arxiv_id":"2407.11691","doi":"10.48550/arxiv.2407.11691","metadata_source":"pith","pith_arxiv_id":"2407.11691","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vlmevalkit: An open-source toolkit for evaluating large multi-modality models","venue":"cs.CV","work_id":"c639e7fb-11f5-459a-8942-659daa5ac1d8","year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2407.11691","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:6f748c4beef6a7aed127a992ebf5efb34c53f7bc3bc2c838e740243d699f2993","observation_id":"14906927-c213-4dfe-a9f6-9bdad2c30e1f","resolution":{"observed_at":"2026-07-07T02:19:36.652705Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12772","last_updated":"2025-05-05T04:48:45Z","snapshot_observed_at":"2026-07-06T18:47:56.109836Z","submitted_at":"2024-07-17T17:51:53Z","title":"LMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models","version":2},"cited_work":{"arxiv_id":"2407.12772","doi":"10.48550/arxiv.2407.12772","metadata_source":"pith","pith_arxiv_id":"2407.12772","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models","venue":"cs.CL","work_id":"257da118-790b-4686-87d7-92321d7e1ae0","year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2407.12772","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:4dc030ca39407cf6f0c5d4d716e3ba80716053d7285ba0fb944d3237d173b56c","observation_id":"d84ed359-c4f1-4ecf-b189-b9f9f51441fe","resolution":{"observed_at":"2026-07-03T20:18:56.562509Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.5281/zenodo.18434279","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Inspect AI: Framework for large language model evaluations, 2024","venue":"Zenodo (CERN European Organization for Nuclear Research)","work_id":"30ab7db1-8a3e-4c0b-859a-cd913f7aa670","year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:0d66a9bba17f684928a3ed9863eb830a944b030e464669f7de0f50f43bd9c07c","observation_id":"33ddd818-f224-4ec7-82b0-6b68f15064bd","resolution":{"observed_at":"2026-06-27T01:30:20.408273Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"DeepSeek-V4: Technical report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:bd44077917f42d4c032e4468be30aa24aea53697b33f4d14e9ac8db44c887360","observation_id":"19739858-a943-4b72-a280-d472425b5bc0","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04368","last_updated":"2024-11-07T01:58:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-07T01:58:42Z","title":"Measuring short-form factuality in large language models","version":1},"cited_work":{"arxiv_id":"2411.04368","doi":"10.48550/arxiv.2411.04368","metadata_source":"pith","pith_arxiv_id":"2411.04368","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring short-form factuality in large language models","venue":"cs.CL","work_id":"f8e490ab-7057-43fb-8c6d-06fc603836c7","year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2411.04368","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:a3b68edfc0c3e65bb2e3c31dcd243bee3021d063fbfa2ade1f84153bc3a8c819","observation_id":"eb147307-4954-4cb6-90e1-cad360ba02e4","resolution":{"observed_at":"2026-07-03T20:18:56.565461Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"MMLU-Pro: A more robust and challenging multi-task language understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:83da1d04640487f1a7b8c92e5b68409f815f73ff2ba9a801e5067939d585ef21","observation_id":"7ca212be-2a01-4a30-893e-a4e4eee31410","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:fdf0b84733eed6f7b1b3abb0d27c2d841e11f7044eba17e728cf4cbc1703299c","observation_id":"b268ac5f-8580-4dba-978a-667f7cf58e88","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14739","last_updated":"2025-03-28T15:21:44Z","snapshot_observed_at":"2026-07-29T21:41:16.650188Z","submitted_at":"2025-02-20T17:05:58Z","title":"SuperGPQA: Scaling LLM Evaluation across 285 Graduate Disciplines","version":4},"cited_work":{"arxiv_id":"2502.14739","doi":"10.48550/arxiv.2502.14739","metadata_source":"pith","pith_arxiv_id":"2502.14739","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SuperGPQA: Scaling LLM Evaluation across 285 Graduate Disciplines","venue":"cs.CL","work_id":"58a97b2d-494b-4c1a-bd65-13fa6bb7f8f6","year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2502.14739","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:74f91592cbf7ec2d88d2c153688f0ce4c27fded99fb244a32dbc133f3fcf64a9","observation_id":"d1384d45-0785-4ce6-828d-26a9db2da0f9","resolution":{"observed_at":"2026-07-03T20:18:56.568783Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-22T13:52:42.853111+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-22T13:52:42.853111+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:adf7cf2a5379c5297612608d548254d6ce4578b07e4bcdacca5a4911bddd8811","observation_id":"06ce4879-bd0f-49c3-9665-58c2006139cb","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"C-Eval: A multi-level multi-discipline chinese evaluation suite for foundation models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:e817353c1b24b754cc78bc201240d20cea29ec309d5b661260a3948ef65ca9c1","observation_id":"8ea4d919-05d7-4791-ae17-fea528ed8d31","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"cited_work":{"arxiv_id":"2501.14249","doi":"10.1038/s41586-025-09962-4","metadata_source":"pith","pith_arxiv_id":"2501.14249","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Humanity's Last Exam","venue":"cs.LG","work_id":"59ea00d4-16a8-45e1-aafc-290a6f91d9f4","year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2501.14249","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:2af472f269c1016e25a93e2b00c6d60f78e76ad7aaac36622e918f2763ba0f35","observation_id":"e40060e2-602d-4874-b0f8-1c3eab0efcc8","resolution":{"observed_at":"2026-07-03T20:18:56.582548Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-03T22:08:32.844211+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T22:08:32.844211+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"MathArena: Evaluating LLMs on uncontaminated math competitions.Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:04cfbac635a4f85b4e100fd230a2d7c07a9a7865f06d7814784692d2c5b0e066","observation_id":"b5333460-ca37-4550-bf93-a25d36ddb22b","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"LiveCodeBench: Holistic and contamination free eval- uation of large language models for code","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:b040c115d689322c526630e520f87984609562d6e96157c8f25e6096d9599bae","observation_id":"34b96082-ade2-47b2-83a2-89d1eda7741e","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"LiveBench: A challenging, contamination-limited LLM benchmark","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:cdbfb60ccc6f5f6273b6c9c789e2de83348abb2a38be426284da7655f7b7ccb9","observation_id":"adc0dac5-123a-431a-9e1c-be3542abaa94","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Measuring mathematical problem solving with the MATH dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:bb90f7cad9c93f03fbfd818b1d69e6631cd46a7a748de0c8a7bdbc142d2d48de","observation_id":"2da44b45-b6d3-4e67-8e9e-7cc4af0ef013","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Let’s verify step by step","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:d8009869fe2ea80d7cabd647eae25772017cbd7f120db804d419c9dc3e7d73d4","observation_id":"00533017-745f-46bd-a3f9-134c08684039","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":"2311.07911","doi":"10.48550/arxiv.2311.07911","metadata_source":"pith","pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Instruction-Following Evaluation for Large Language Models","venue":"cs.CL","work_id":"3aa06177-125a-4f5a-8f4a-8070c5986c26","year":2023},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:58899e104c18a6df79d67f268f8dd5735e726d85d15db3b10e0c3d43f2013d83","observation_id":"6b97a20f-30f4-4dee-92ad-a318ec69de01","resolution":{"observed_at":"2026-07-03T20:18:56.577524Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Patil, Huanzhi Mao, Fanjia Yan, Charlie Cheng-Jie Ji, Vishnu Suresh, Ion Stoica, and Joseph E","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:2c52dd38a94160cede91650a3c96b642931248f40cd04722222632abece12415","observation_id":"0fdf88e8-565c-4fc6-8074-0e48289f69ff","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"MMMU: A massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:fa8bafe2ecb8020f947f9eb991262d93d41875a2ab6f0bfa2303aeede4a2ab08","observation_id":"9dda837a-7c18-48f1-9ae9-a24761cdf027","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"MMMU-Pro: A more robust multi-discipline multimodal understanding benchmark","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:8925d60211612ab075915e0e2fa86b8d75e618f557c2be97d90e9b4daf8ebd9f","observation_id":"5c6ac217-6049-465c-9e9c-f844fb638155","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"MathVista: Evaluating mathematical reasoning of foundation models in visual contexts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:36ae2159b9920350e7bbf83a44e7e6b070c1255dc5690780be0e18b446646970","observation_id":"77dee267-f976-4c4c-82e2-b7d670d75d5a","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"DynaMath: A dynamic visual benchmark for evaluating mathematical reasoning robustness of vision language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:31f11d854a97f537305e99a45a45c5f479fe40a7493d6d9d264c4a2f8f156d78","observation_id":"fa191aea-507c-4f38-9585-1e15a12dc8c2","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Vision language models are blind: Failing to translate detailed visual features into words","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:babcc1bcd042e890324b83e2aaaa1e09c4f0b966ed45e490e41a687795dfc167","observation_id":"f744dd39-d822-4954-8706-62829f968395","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"MMBench: Is your multi-modal model an all-around player? InEuropean Conference on Computer Vision, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:cf3409599986a350c3aafd00a7c1c9044dfab058a1198493e9ad1dca131530a6","observation_id":"a028ef52-8ac0-4a21-978a-d948f3d36d48","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Are we on the right way for evaluating large vision-language models? InAdvances in Neural Information Processing Systems, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:d0492610718f2e59b72b04d340b833c1cd60419eb41c95b7e8f9a1244be29bf2","observation_id":"ce73596b-7664-4865-8a9c-87b449fd91d7","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"RealWorldQA.https://huggingface.co/datasets/xai-org/RealworldQA, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:2ef12a5b728de850af5c70ed054e7c4f8b061707954b5f186837103e732deece","observation_id":"e326c79d-3a34-4c16-b58a-46a89197fb72","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"SimpleVQA: Multimodal factuality evaluation for multimodal large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:4d34cb03ba5c7675953bc78851858635aba5a6c3a3b98c10c28de0bbfadeeb59","observation_id":"eb5b73ff-e5c9-4798-b510-ab292685f955","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.18521","last_updated":"2024-06-26T17:50:11Z","snapshot_observed_at":"2026-08-06T23:04:27.981034Z","submitted_at":"2024-06-26T17:50:11Z","title":"CharXiv: Charting Gaps in Realistic Chart Understanding in Multimodal LLMs","version":1},"cited_work":{"arxiv_id":"2406.18521","doi":"10.48550/arxiv.2406.18521","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.18521","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Charxiv: Charting gaps in realistic chart understanding in multimodal llms","venue":"arXiv (Cornell University)","work_id":"c9b18bcb-4462-48c9-be76-491ad33d4809","year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2406.18521","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:835e167bfb614f66b13fabf1a0679572b9807c2472e6382c83f3862d4d56f0b6","observation_id":"01ce1a4f-893e-42d7-89e5-2674561c10c6","resolution":{"observed_at":"2026-07-03T20:18:56.584395Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"OCRBench: On the hidden mystery of OCR in large multimodal models.Science China Information Sciences, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:49918fecdb5529c64f915fd0beb4bd2b117a65d857b55f2235d15efaf0a36a8f","observation_id":"3837a3ae-13d8-4197-8b4f-b6f25aa63e6b","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Teaching CLIP to count to ten","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:860cbf080eee1bd7f50aaab8d3de93477400cafa80c540e3dc3347e61df590b4","observation_id":"b22ae11c-b951-4a68-b116-772869b8094b","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Berg, and Tamara L","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:3ea536eb753d050778b997260c64b339bee231b3d06a13b90068203f7f08a127","observation_id":"6f38a86a-af7b-468d-a387-eda6450f3ce1","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20020","last_updated":"2025-03-25T19:02:56Z","snapshot_observed_at":"2026-08-02T07:15:58.798604Z","submitted_at":"2025-03-25T19:02:56Z","title":"Gemini Robotics: Bringing AI into the Physical World","version":1},"cited_work":{"arxiv_id":"2503.20020","doi":"10.48550/arxiv.2503.20020","metadata_source":"pith","pith_arxiv_id":"2503.20020","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini Robotics: Bringing AI into the Physical World","venue":"cs.RO","work_id":"f7c5ce10-8364-4fbe-964f-2802b81c3a98","year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2503.20020","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:651c7beb3d950983c3e4d57d13defdcc00d9b573d49c8bd417ae6d8c29c9ec4d","observation_id":"f232f156-15f7-4dec-aafb-c003e7d3e8cc","resolution":{"observed_at":"2026-07-03T20:18:56.554037Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Video-MME: The first-ever comprehensive evaluation benchmark of multi-modal LLMs in video analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:dd7228273668325f8c02970f4656a639547c070e19afbae05a8528b6313041fe","observation_id":"641c6212-c570-4e80-a20f-e66f1911b946","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"LibriSpeech: An ASR corpus based on public domain audio books","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:865b1937368e6273071aa03adc9f21dd5d11d2f81479532e8199b6e202c40657","observation_id":"36e951a3-edad-45d2-a32d-6f16d9758464","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"WenetSpeech: A 10000+ hours multi-domain mandarin corpus for speech recognition","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:046ec518f17e71c4afcd18d37433173fd6baa783bde2855676c3e006b4bacfd9","observation_id":"10da4ae3-4e52-4445-927a-f27bd83a8724","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04326","last_updated":"2026-03-01T04:35:41Z","snapshot_observed_at":"2026-07-06T20:32:23.502480Z","submitted_at":"2025-02-06T18:59:40Z","title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","version":3},"cited_work":{"arxiv_id":"2502.04326","doi":"10.48550/arxiv.2502.04326","metadata_source":"pith","pith_arxiv_id":"2502.04326","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","venue":"cs.CV","work_id":"9fb20f56-0773-406f-93c4-122a3e2ab9e4","year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2502.04326","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:24ced8107195f71f876618a20c3446751bb467c7255553a3f03e873f7d6de236","observation_id":"fba57328-79e0-4014-b496-74d7a04f73e5","resolution":{"observed_at":"2026-07-03T20:18:56.555124Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.17765","last_updated":"2025-09-22T13:26:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-22T13:26:24Z","title":"Qwen3-Omni Technical Report","version":1},"cited_work":{"arxiv_id":"2509.17765","doi":"10.48550/arxiv.2509.17765","metadata_source":"pith","pith_arxiv_id":"2509.17765","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-Omni Technical Report","venue":"cs.CL","work_id":"ae43e594-8bab-4471-b6af-92a300f6a048","year":2025},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"cited_paper":"/paper/2509.17765","citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:b67650fbc093411d3e7f1ecc38b34146b78cb51f725f310af1f1b45a408b1147","observation_id":"724b29f9-f116-4780-9c31-e3cbe9a5e483","resolution":{"observed_at":"2026-07-03T20:18:56.558413Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T01:28:59.297634Z","title":"Beyond the nav-graph: Vision and language navigation in continuous environments","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:59.297634Z"},"links":{"citing_paper":"/paper/2606.17574"},"observation_digest":"sha256:9681fef8b6f9d43d41b23872244cbefe86666bda650fccec93830e3e69198ed8","observation_id":"4e90127e-528e-49bb-8855-90707b08cc2f","resolution":{"observed_at":"2026-06-27T01:28:59.297634Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.17574","last_updated":"2026-06-16T06:22:09Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-02T05:34:30.444420Z","submitted_at":"2026-06-16T06:22:09Z","title":"DeepInsight: A Unified Evaluation Infrastructure Across the Physical AI Stack"},"reference_resolution":{"displayed":59,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":40,"verified_exact":16,"verified_fuzzy":0},"total_outbound_references":59},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 59 of 59 outbound references and 1 inbound Pith citation observation for arXiv:2606.17574."}