{"as_of":"2026-08-19T00:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:186e2f764eb39bad7dc14095dac0dfef6061232721bfc485305954271c35b4c3","coverage":[{"denominator":95,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":95,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T18:20:58.963152Z","state":"measured"},{"denominator":110,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":110,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":15,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T15:32:27.422997Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T00:49:17.558383Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-08-04T07:23:04.393109Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.26113","last_updated":"2026-06-18T20:06:47Z","snapshot_observed_at":"2026-08-18T01:19:54.035244Z","submitted_at":"2025-10-30T03:53:22Z","title":"EgoExo-Con: Exploring View-Invariant Video Temporal Understanding","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-04T07:23:04.393109Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2510.26113"},"observation_digest":"sha256:00bf7dc5e61daaab8909ea719d80cf5f870dfb79afa044429214fc6f063c5bb5","observation_id":"ded4086a-74cc-4ade-a563-ab2d0cdba6ca","resolution":{"observed_at":"2026-08-04T07:23:04.393109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":"2507.18342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-04T00:49:17.558383Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":"221f55d6-58ce-4326-88b3-b2d7ea6f3a53","year":2025},"citing_paper":{"arxiv_id":"2511.18127","last_updated":"2026-05-15T06:06:57Z","snapshot_observed_at":"2026-08-17T15:14:40.490371Z","submitted_at":"2025-11-22T17:22:24Z","title":"SFHand: Learning Embodied Manipulation by Streaming Egocentric 3D Hand Forecasting","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-21T17:53:19.002603Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2511.18127"},"observation_digest":"sha256:2844439cc287dfd791c8eb3346135e62da96fbcb2c88b5ccec52fd1de55ea2af","observation_id":"e0c2ee20-3038-46d8-8b9a-1409b44e840b","resolution":{"observed_at":"2026-05-21T17:54:18.229368Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-08-03T16:55:20.757154Z","title":"EgoExoBench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.11393","last_updated":"2026-08-13T11:40:44Z","snapshot_observed_at":"2026-08-16T23:11:50.762075Z","submitted_at":"2025-12-12T09:07:21Z","title":"The N-Body Problem: Parallel Execution from Single-Person Egocentric Video","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-03T16:55:20.757154Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2512.11393"},"observation_digest":"sha256:b1112bba86b1778b9fd2a12d3d9bf46a412c7d7636cae3b72d26bb409c3ae4db","observation_id":"2ecf6caa-271e-43b3-bc76-5a0fff9478e5","resolution":{"observed_at":"2026-08-03T16:55:20.757154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-08-03T03:29:06.350552Z","title":"Accessed: 2026-01-10","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2602.07864","last_updated":"2026-05-29T03:38:34Z","snapshot_observed_at":"2026-08-15T13:28:03.530712Z","submitted_at":"2026-02-08T08:29:38Z","title":"Thinking in Structures: Evaluating Spatial Intelligence in Constraint-Governed Spaces","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-03T03:29:06.350552Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2602.07864"},"observation_digest":"sha256:999a7b2af26cf437c5f1367c520659244396ed139149409d3e25d1cdd378707e","observation_id":"bb7a0976-f2e6-4f81-9200-b8ed1f5c95a0","resolution":{"observed_at":"2026-08-03T03:29:06.350552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":"2507.18342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-04T00:49:17.558383Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":"221f55d6-58ce-4326-88b3-b2d7ea6f3a53","year":2025},"citing_paper":{"arxiv_id":"2604.13793","last_updated":"2026-07-01T12:10:20Z","snapshot_observed_at":"2026-08-13T13:28:39.975215Z","submitted_at":"2026-04-15T12:32:25Z","title":"From Synchrony to Sequence: Exo-to-Ego Generation via Interpolation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:35.738229Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2604.13793"},"observation_digest":"sha256:428a8cfd2f8aa5f69495f1ed738eb5b7196ba6b68dfb0190f78d181c83efff22","observation_id":"13a31d8e-665d-4506-b2d4-d4238d16157d","resolution":{"observed_at":"2026-05-10T13:25:26.186466Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-12T20:34:35.048469Z","title":"arXiv preprint arXiv:2507.18342 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.13793","last_updated":"2026-07-01T12:10:20Z","snapshot_observed_at":"2026-08-13T13:28:39.975215Z","submitted_at":"2026-04-15T12:32:25Z","title":"From Synchrony to Sequence: Exo-to-Ego Generation via Interpolation","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-12T20:34:35.048469Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2604.13793"},"observation_digest":"sha256:26ecbdc3aa3f1a0cced2b515dcc0cfcb41dd17aa34232936aa52f16c50af2fab","observation_id":"46bb81c8-d476-42cf-a242-af7a1fed4cd8","resolution":{"observed_at":"2026-07-12T20:34:35.048469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":"2507.18342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-04T00:49:17.558383Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":"221f55d6-58ce-4326-88b3-b2d7ea6f3a53","year":2025},"citing_paper":{"arxiv_id":"2604.26565","last_updated":"2026-04-29T11:51:35Z","snapshot_observed_at":"2026-08-10T21:10:06.962801Z","submitted_at":"2026-04-29T11:51:35Z","title":"DenseStep2M: A Scalable, Training-Free Pipeline for Dense Instructional Video Annotation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-07T13:50:23.835653Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2604.26565"},"observation_digest":"sha256:a1a34295639d6ca1290dd41953e2e68172eb2c389b59fd5a1d2c2e6bc5d9e482","observation_id":"c59f0476-3ed6-4de5-bfbe-dea78629240a","resolution":{"observed_at":"2026-05-12T08:46:26.137872Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":"2507.18342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-04T00:49:17.558383Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":"221f55d6-58ce-4326-88b3-b2d7ea6f3a53","year":2025},"citing_paper":{"arxiv_id":"2605.18431","last_updated":"2026-05-19T05:12:41Z","snapshot_observed_at":"2026-08-14T15:01:13.021738Z","submitted_at":"2026-05-18T14:04:26Z","title":"Seeing Together: Multi-Robot Cooperative Egocentric Spatial Reasoning with Multimodal Large Language Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-20T10:36:20.388169Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2605.18431"},"observation_digest":"sha256:8966c01e6d3cc55a223360e670471c685fc820de88a73c94fbd2e81650da91e6","observation_id":"37c82ded-8913-4a0d-9f9a-8e0e4ce53947","resolution":{"observed_at":"2026-05-20T10:38:12.458834Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":"2507.18342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-04T00:49:17.558383Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":"221f55d6-58ce-4326-88b3-b2d7ea6f3a53","year":2025},"citing_paper":{"arxiv_id":"2605.19559","last_updated":"2026-05-19T09:02:20Z","snapshot_observed_at":"2026-08-16T01:19:28.511820Z","submitted_at":"2026-05-19T09:02:20Z","title":"EgoCoT-Bench: Benchmarking Grounded and Verifiable Operation-Centric Chain of Thought Reasoning for MLLMs","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-20T05:53:05.450946Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2605.19559"},"observation_digest":"sha256:854a7e2424d09d56983e58f58774d2c6c2b74d004eb5999af9551daecdc0da3a","observation_id":"a2990647-57cd-42d9-985a-36ff731ae53d","resolution":{"observed_at":"2026-05-20T05:53:22.284466Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":"2507.18342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-04T00:49:17.558383Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":"221f55d6-58ce-4326-88b3-b2d7ea6f3a53","year":2025},"citing_paper":{"arxiv_id":"2605.22570","last_updated":"2026-05-21T14:48:35Z","snapshot_observed_at":"2026-08-18T06:27:43.173870Z","submitted_at":"2026-05-21T14:48:35Z","title":"VGenST-Bench: A Benchmark for Spatio-Temporal Reasoning via Active Video Synthesis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T07:12:02.612292Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2605.22570"},"observation_digest":"sha256:74c1d06a11056952a62dc25a228e84782b74b133263fcb9306032780dc6afb28","observation_id":"dc39be0c-1d2b-44a0-be0c-473551593e09","resolution":{"observed_at":"2026-05-22T07:14:42.751993Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":"2507.18342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-04T00:49:17.558383Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":"221f55d6-58ce-4326-88b3-b2d7ea6f3a53","year":2025},"citing_paper":{"arxiv_id":"2605.23216","last_updated":"2026-07-02T06:54:56Z","snapshot_observed_at":"2026-08-03T00:30:19.785022Z","submitted_at":"2026-05-22T04:19:29Z","title":"CaST-Bench: Benchmarking Causal Chain-Grounded Spatio-Temporal Reasoning for Video Question Answering","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-25T04:54:23.077914Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2605.23216"},"observation_digest":"sha256:f7a41ffc0c16587b5ead83ebeb824b8b0a6d813589e596dafdc5847421eb34cd","observation_id":"4ed66bc6-0995-4ad6-b27a-fd3fa3cd3cc3","resolution":{"observed_at":"2026-05-25T04:55:23.228583Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":"2507.18342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-04T00:49:17.558383Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":"221f55d6-58ce-4326-88b3-b2d7ea6f3a53","year":2025},"citing_paper":{"arxiv_id":"2605.23216","last_updated":"2026-07-02T06:54:56Z","snapshot_observed_at":"2026-08-03T00:30:19.785022Z","submitted_at":"2026-05-22T04:19:29Z","title":"CaST-Bench: Benchmarking Causal Chain-Grounded Spatio-Temporal Reasoning for Video Question Answering","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-04T00:41:02.284215Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2605.23216"},"observation_digest":"sha256:e69bf7b4f212a11177d0553f0ea79d893897a1f5c61863cc8007b841e184038b","observation_id":"7761571f-da78-4229-90ec-ab3281be825a","resolution":{"observed_at":"2026-07-04T00:49:17.561558Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":"2507.18342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-04T00:49:17.558383Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms","venue":null,"work_id":"221f55d6-58ce-4326-88b3-b2d7ea6f3a53","year":2025},"citing_paper":{"arxiv_id":"2606.05677","last_updated":"2026-06-04T04:00:12Z","snapshot_observed_at":"2026-08-13T16:51:39.502297Z","submitted_at":"2026-06-04T04:00:12Z","title":"LongSpace: Exploring Long-Horizon Spatial Memory from Perception to Recall in Video","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-06-28T02:16:25.730555Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2606.05677"},"observation_digest":"sha256:1cbda7805cdcc40bd13879dec6207fc62c2976502d40f02ea027b49554767836","observation_id":"aaba518a-16eb-410b-a63c-70c166379fb3","resolution":{"observed_at":"2026-07-02T12:16:57.252484Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-07-14T05:01:06.200663Z","title":"arXiv preprint arXiv:2507.18342 (2025) 3","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11523","last_updated":"2026-07-13T13:09:45Z","snapshot_observed_at":"2026-08-15T01:01:51.688250Z","submitted_at":"2026-07-13T13:09:45Z","title":"Vinci2: Providing Proactive Assistance in Continuous Egocentric Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-14T05:01:06.200663Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2607.11523"},"observation_digest":"sha256:3e6904878ccf6749de6a9d9fc60aa123daed91e4c5b49a0f823c97a64a38b386","observation_id":"1ec72eaf-36e5-4e72-9cb2-369079c40204","resolution":{"observed_at":"2026-07-14T05:01:06.200663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.18342","snapshot_observed_at":"2026-08-15T15:32:27.422997Z","title":"Egoexobench: A benchmark for first-and third-person view video understanding in mllms.arXiv preprint arXiv:2507.18342, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22864","last_updated":"2026-07-24T19:09:25Z","snapshot_observed_at":"2026-08-17T12:59:56.492496Z","submitted_at":"2026-07-24T19:09:25Z","title":"Spatial-IQ: Deconstructing Spatial Intelligence via Hierarchical Capability Tests","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T15:32:27.422997Z"},"links":{"cited_paper":"/paper/2507.18342","citing_paper":"/paper/2607.22864"},"observation_digest":"sha256:25b8f8d33c1766bbb7d9c52a2ac6e2707d8c4c8b4139d0f86adfcf78731939c9","observation_id":"4cc2130b-ef76-41f9-831b-afa575490aff","resolution":{"observed_at":"2026-08-15T15:32:27.422997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.18342/citation-record","integrity":"/paper/2507.18342/integrity","json":"/paper/2507.18342/citation-record.json","paper":"/paper/2507.18342"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.617371Z","title":"Visual-policy learning through multi-camera view to single-camera view knowledge distillation for robot manipulation tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.617371Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:9a0cbbb26dcf87f44c12f3c07df4f55e4ceab8894f1b9093caa8ee85f55d0dd3","observation_id":"f2310575-32fd-4f90-bef6-c88311fa63f3","resolution":{"observed_at":"2026-08-15T18:20:58.617371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.621745Z","title":"Claude 3.7 sonnet and claude code, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.621745Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:e4dd378d6ea3c622b316ae8b070983797a98d4da67855b94b7bf4d9eec04fb86","observation_id":"32b58db8-8323-46c7-8617-6ad42750aa93","resolution":{"observed_at":"2026-08-15T18:20:58.621745Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.625637Z","title":"Observational learning","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.625637Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:ebd17d12d4fea1ca6baa61c469f31249b3cf2da5efdcfd80ef4787de5e58f35c","observation_id":"67f8bc59-1544-40c0-82e4-0bc7bdb8fd29","resolution":{"observed_at":"2026-08-15T18:20:58.625637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.629556Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.629556Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:a0801962f6238f7d464f23a2ae67c8a5706bd8f65047ed053048f9116a4d0fb5","observation_id":"a3a41bf0-8eaa-4dc2-baec-da47b7e8841a","resolution":{"observed_at":"2026-08-15T18:20:58.629556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.633385Z","title":"In your place: neuropsychological evidence for altercentric remapping in embodied perspective taking","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.633385Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:b9130c4d7f3acf7c7aee23e74e4c60ad1e63a2452e88af6e892f641f1a6ddfe6","observation_id":"195e7901-9056-485b-bf87-80485589499a","resolution":{"observed_at":"2026-08-15T18:20:58.633385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.637270Z","title":"Spatial memory: how egocentric and allocentric combine","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.637270Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:f000eea3a9aef1354ee55c052d9d0b04d23cf153fe568459edb8e7dd6b66cb45","observation_id":"a56965a1-0cd4-41bb-b21e-754d1fe5c59f","resolution":{"observed_at":"2026-08-15T18:20:58.637270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.641150Z","title":"Hourvideo: 1-hour video-language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.641150Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:f18fb4ffd9069ee69c507d02cd00db7973668e79292efaa260c84a6422be2ce5","observation_id":"500c462b-1311-4611-b691-1b571ffc462d","resolution":{"observed_at":"2026-08-15T18:20:58.641150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-08-17T01:38:26.232918Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12075","snapshot_observed_at":"2026-08-15T18:20:58.644807Z","title":"Cg-bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.644807Z"},"links":{"cited_paper":"/paper/2412.12075","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:f234229ff9242654a709e6e42916c774f4c7a1074bd08177b5d0a8923880b66c","observation_id":"40bfb0cd-576b-4e7e-be9a-a50316f37e40","resolution":{"observed_at":"2026-08-15T18:20:58.644807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.648836Z","title":"put myself into your place","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.648836Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:4d15c9a60e5c3146890a6673f65d7e6192e8d6bc7a588deac3e91410cbdd9f4b","observation_id":"7c8cf13b-8852-4646-bd10-07e2c70b9d5b","resolution":{"observed_at":"2026-08-15T18:20:58.648836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.652640Z","title":"Scaling egocentric vision: The epic-kitchens dataset","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.652640Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:0b885306b87db0f35938d235931287989f9ac114bafaddffcd7879a7b4a8ec4e","observation_id":"1ffa20f7-275f-40f7-9773-25e0efae1da7","resolution":{"observed_at":"2026-08-15T18:20:58.652640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.656210Z","title":"Epic-kitchens visor benchmark: Video segmentations and object relations","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.656210Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:827c2d93777f5b2620596bd9211b3acb88708409ad415d653168c2fd417f9f93","observation_id":"343622d4-2d36-4e23-bbfa-a25025e05a77","resolution":{"observed_at":"2026-08-15T18:20:58.656210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.659999Z","title":"Gemini 2.5 pro, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.659999Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:4c99660f08576a0e646e8568b0f4c0608dde90535421c6ab911338209dd71bc6","observation_id":"1e2b9dac-14ca-4ba8-88f9-950b36361009","resolution":{"observed_at":"2026-08-15T18:20:58.659999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.663511Z","title":"Interacting networks of brain regions underlie human spatial navigation: a review and novel synthesis of the literature","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.663511Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:7fc6cfbe55f646c15fa92fa2891bec3dfe22a23fda255e497ca616e8ad5d7b66","observation_id":"fd5530cb-e27c-4997-a7b7-8aa2d3169a73","resolution":{"observed_at":"2026-08-15T18:20:58.663511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-15T18:20:58.667394Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.667394Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:6c6da8c9ea42698155088cdbab039a437f636fdbbabd4b4be2f1280bea021a9b","observation_id":"798a96e5-3f76-4954-8086-ca2d90d20441","resolution":{"observed_at":"2026-08-15T18:20:58.667394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.00720","last_updated":"2023-01-30T09:15:49Z","snapshot_observed_at":"2026-08-16T16:27:25.222553Z","submitted_at":"2022-10-03T05:33:27Z","title":"Complexity-Based Prompting for Multi-Step Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.00720","snapshot_observed_at":"2026-08-15T18:20:58.671208Z","title":"Complexity-based prompting for multi-step reasoning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.671208Z"},"links":{"cited_paper":"/paper/2210.00720","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:cd816d283e86df30ce6644b6f031337fff2fbc59d0b80891bf96bf858d3eff4a","observation_id":"69c1d61f-42a0-4578-9dea-3a8343feaf39","resolution":{"observed_at":"2026-08-15T18:20:58.671208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.674985Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.674985Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:1008307dff103b6cdf101529c63f14a05d5bb1535eaad3bafb2379a5dee5cd43","observation_id":"3f37f27f-952d-4e41-a810-a0d090650b14","resolution":{"observed_at":"2026-08-15T18:20:58.674985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.678833Z","title":"Ego-exo4d: Understanding skilled human activity from first-and third-person perspectives","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.678833Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:87260b0549720178c90b5b734c7ce28635744b69cff96db66e38a8c1ac3c9e0a","observation_id":"34a89421-6e29-4280-aacb-d753faa57df3","resolution":{"observed_at":"2026-08-15T18:20:58.678833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.682780Z","title":"Ava: A video dataset of spatio-temporally localized atomic visual actions","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.682780Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:b263df75b317e01513412097c3470d00a0ee81fdcd918ca4d4d77092375f28cd","observation_id":"8ea24c61-3eb8-4d72-a07a-cae17bb42941","resolution":{"observed_at":"2026-08-15T18:20:58.682780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.686233Z","title":"Multiple human association and tracking from egocentric and complementary top views","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.686233Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:e851586c865d1722f26ac2980fe613780987582be0c62b5c6777ca238aca0b4e","observation_id":"ec7a0169-2b96-4690-ab8c-ea79d5195bde","resolution":{"observed_at":"2026-08-15T18:20:58.686233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.691028Z","title":"What is modelled during observational learning? Journal of sports sciences, 25(5):531–545, 2007","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.691028Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:4eea4b1cd0830e60c2639b12bba2be41b5e67227858d7588d9d4ba6810b007fb","observation_id":"4ce2ff04-61e6-4fb5-b252-e953c23626aa","resolution":{"observed_at":"2026-08-15T18:20:58.691028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:21:00.160872Z","title":"An ego-vision system for discovering human joint attention","venue":null,"work_id":"37b3dac8-e821-420d-9cd3-e3257128e345","year":2020},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.694818Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:fdb721f11a7cc9cd7f478f9fe8a827f4dd1aaefda8ee1850424a58c09885be84","observation_id":"6c900cf2-3c50-4a7f-a6a1-e3dbd1d9f79d","resolution":{"observed_at":"2026-08-15T18:21:00.165437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:21:00.144704Z","title":"Improving action segmentation via graph-based temporal reasoning","venue":null,"work_id":"281933f8-0e85-4249-be9e-d79a3353f6fe","year":2020},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.698351Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:b0532a228b408a60111b74df7a39629cb873356d202b05d4dcde244da72e0719","observation_id":"0bf67345-564d-4955-b7fb-7278172bdb99","resolution":{"observed_at":"2026-08-15T18:21:00.151308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:21:00.126335Z","title":"Egoexolearn: A dataset for bridging asynchronous ego-and exo-centric view of procedural activities in real world","venue":null,"work_id":"2342a1c5-646f-4772-a635-ab3b29389520","year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.702040Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:21cafea110eb540dd7693d93ec5d462afb483aeac0f41cb2c0678ea191d9f563","observation_id":"5a595925-4655-4dc4-9c46-8cf167c170e7","resolution":{"observed_at":"2026-08-15T18:21:00.132433Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.21080","last_updated":"2024-12-30T16:57:05Z","snapshot_observed_at":"2026-08-16T22:23:33.828463Z","submitted_at":"2024-12-30T16:57:05Z","title":"Vinci: A Real-time Embodied Smart Assistant based on Egocentric Vision-Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.21080","snapshot_observed_at":"2026-08-15T18:20:58.705374Z","title":"Vinci: A real-time embodied smart assistant based on egocentric vision-language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.705374Z"},"links":{"cited_paper":"/paper/2412.21080","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:b4e58893a87f19b09eac934f598363ee82dc8763f4653e380c95e9d74555cfa2","observation_id":"b5ad38a0-fd46-4150-b409-72d8b5758dce","resolution":{"observed_at":"2026-08-15T18:20:58.705374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T14:02:47.366139Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-15T18:20:58.710025Z","title":"Gpt-4o system card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.710025Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:a3bd27bcbe2f01c274e3d5cc15bb963df7797add636f757a84e381060b459c7e","observation_id":"55a6e0b9-07ba-42fc-8449-267a40846e03","resolution":{"observed_at":"2026-08-15T18:20:58.710025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:21:00.110007Z","title":"Lemma: A multi-view dataset for le arning m ulti-agent m ulti-task a ctivities","venue":null,"work_id":"27497a48-4959-455e-b458-f2ecf90f841f","year":2020},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.714181Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:d1292d4c4a3db61f1e2a99068e1453828731dc8c2e763052a1683ad47f6cefe4","observation_id":"ac2851b2-345a-4d95-b3d8-a5dc82ee5977","resolution":{"observed_at":"2026-08-15T18:21:00.114741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:21:00.093820Z","title":"Egotaskqa: Understanding human tasks in egocentric videos","venue":null,"work_id":"28409b2b-bdbb-4883-9b33-7787157c4ee2","year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.717632Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:e3d71517d0c2be412f14162bed07e765a7b4bd7bdc2465df208c5fb8d2e5fc2f","observation_id":"f1e05fce-ca9f-4c0a-a436-620e78c46e79","resolution":{"observed_at":"2026-08-15T18:21:00.100849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.722270Z","title":"Large language models are zero-shot reasoners","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.722270Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:a176d4e162e938ff8c7ddf678f818ac57fca762d9608aa85a3022e6d2ffded9f","observation_id":"fa0dee72-f454-4fa2-9618-c7dced4d2c91","resolution":{"observed_at":"2026-08-15T18:20:58.722270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:21:00.065529Z","title":"A neural code for egocentric spatial maps in the human medial temporal lobe","venue":null,"work_id":"663d4993-6b03-4cde-8d6c-5a55898714ec","year":2021},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.725834Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:fbb133ab246dd2455cf55bc8948bc5deb1dd3d5a514e1ddc76c92efe06e6ec0e","observation_id":"322f785f-d3a1-4aa0-b614-7d8e03ee7c5b","resolution":{"observed_at":"2026-08-15T18:21:00.070602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:21:00.039366Z","title":"H2o: Two hands manipulating objects for first person interaction recognition","venue":null,"work_id":"44221f35-b35c-42c7-b2d4-eba070f057b6","year":2021},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.729432Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:e1531e5a3be11ac09447abb661c09511502a239d6e224943e85ff00428addd8f","observation_id":"a6d62541-ba34-49f7-8966-6f94c836dac0","resolution":{"observed_at":"2026-08-15T18:21:00.047449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-08-18T11:56:50.710310Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-15T18:20:58.732844Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.732844Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:865102e14fb4d2868ed6cf5327eb9ec2348f404337a34c1d42e59b05bf5ef823","observation_id":"3e8c75a4-a3f7-4ae7-8709-f41783d74906","resolution":{"observed_at":"2026-08-15T18:20:58.732844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.03272","last_updated":"2021-11-03T18:51:07Z","snapshot_observed_at":"2026-08-16T18:04:44.524003Z","submitted_at":"2021-08-06T18:41:39Z","title":"iGibson 2.0: Object-Centric Simulation for Robot Learning of Everyday Household Tasks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.03272","snapshot_observed_at":"2026-08-15T18:20:58.736826Z","title":"igibson 2.0: Object-centric simulation for robot learning of everyday household tasks","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.736826Z"},"links":{"cited_paper":"/paper/2108.03272","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:4110525aeb8d41c0470bef1edd4112665c65b38693a438a1162b74f59ac5c895","observation_id":"5cd4631f-db5c-43c6-b8be-e66c8270d8b0","resolution":{"observed_at":"2026-08-15T18:20:58.736826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-15T18:20:58.740653Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.740653Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:54130d58125fd744b8e1cdd0d67438b4247073d9f3d324e4a11ddb0c7dbae748","observation_id":"1cf0da7c-2ee8-49dc-960f-bb2c2f3371d5","resolution":{"observed_at":"2026-08-15T18:20:58.740653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.744663Z","title":"Mvbench: A comprehensive multi-modal video understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.744663Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:f8aec91d00d6a6af43f566377586db178f58d6bca6122916a61ff6fe0edcc894","observation_id":"abe12be3-7929-4c38-aff4-45b63c0ef68d","resolution":{"observed_at":"2026-08-15T18:20:58.744663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.996710Z","title":"Ego-exo: Transferring visual representa- tions from third-person to first-person videos","venue":null,"work_id":"a19ee8f5-7dc8-4856-9ce6-df4ddbd4e858","year":2021},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.747904Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:a649d9711627733d4a112dd4e7700a6615b9b2f4703f6696a49ee46ddda68cc6","observation_id":"b84f0f9b-786a-4894-b00b-49e17b95a2e0","resolution":{"observed_at":"2026-08-15T18:21:00.002322Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-18T18:18:37.449517Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-15T18:20:58.751334Z","title":"Deepseek-v3 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.751334Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:b8ed9473e3dc819233f3c67d4ffa5c4d40fd920bbdbcad9acc4f8f99c3abd2a0","observation_id":"40fbdaf5-3353-4bcf-a1f1-e526057788ce","resolution":{"observed_at":"2026-08-15T18:20:58.751334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.984576Z","title":"Skill transfer learning for autonomous robots and human–robot cooperation: A survey","venue":null,"work_id":"bb37efa0-6af4-4439-b843-ab9e652dd9de","year":2020},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.755144Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:ec3a709f4444a5474d2d00cbab443697c6db921fe07e22196046b793fb6e7628","observation_id":"b762a704-7475-4755-a799-80d8f08b9f74","resolution":{"observed_at":"2026-08-15T18:20:59.989510Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.758843Z","title":"Hoi4d: A 4d egocentric dataset for category-level human-object interaction","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.758843Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:e7fb0a3210d1af3571fb0c844b2fdb726e61dd7124b9ad9d2bd6598d09b93656","observation_id":"24ef29ac-de58-4737-831d-e7a86a718373","resolution":{"observed_at":"2026-08-15T18:20:58.758843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04468","last_updated":"2026-04-25T07:16:42Z","snapshot_observed_at":"2026-08-03T08:48:57.969106Z","submitted_at":"2024-12-05T18:59:55Z","title":"NVILA: Efficient Frontier Visual Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04468","snapshot_observed_at":"2026-08-15T18:20:58.762270Z","title":"Nvila: Efficient frontier visual language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.762270Z"},"links":{"cited_paper":"/paper/2412.04468","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:f3069fc60fdad747f90ba2cc39695a1372d47687e1a1c93001e1f8fb258c5495","observation_id":"b2c29678-f4a0-4ad3-b74e-81883267bf42","resolution":{"observed_at":"2026-08-15T18:20:58.762270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.966156Z","title":"Egoschema: A diagnostic benchmark for very long-form video language understanding","venue":null,"work_id":"ddaae5a4-bc1b-4b8d-8508-65b110be04e4","year":2023},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.766048Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:6cfd419a8d53d6ac6376afde2b64a5abee87c2e8669dac08017ec7d76567cd91","observation_id":"e7d8ac96-6b1e-4ad1-a8f5-806e8f3421ca","resolution":{"observed_at":"2026-08-15T18:20:59.970058Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.769555Z","title":"Howto100m: Learning a text-video embedding by watching hundred million narrated video clips","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.769555Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:4b0cf88c1bdbece9ea030ae1c23508ca6868e192ca878a986d386e69ae611c81","observation_id":"a5af1db1-350c-4fce-a7fb-e54fb3c7546b","resolution":{"observed_at":"2026-08-15T18:20:58.769555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.948030Z","title":"Gpt-4o mini: advancing cost-efficient intelligence, 07 2024","venue":null,"work_id":"357829d7-5346-4fab-8031-e8d9d8981da8","year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.773316Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:43d5304eda484c0ddb35e4c512e920e64df849b019a93ccf6573d9090a569f5f","observation_id":"45887d3f-a001-41de-9ec1-2a5e5fea8579","resolution":{"observed_at":"2026-08-15T18:20:59.952062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.776896Z","title":"Egovlpv2: Egocentric video-language pre-training with fusion in the backbone","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.776896Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:9034683807dff32ecb8518db3fe9980d5e16548fd03f6e6e33552537d685974a","observation_id":"d4856754-4a7b-4506-bd2e-896f5c28f8d7","resolution":{"observed_at":"2026-08-15T18:20:58.776896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19061","last_updated":"2025-03-30T02:44:43Z","snapshot_observed_at":"2026-08-10T20:51:25.702157Z","submitted_at":"2025-01-31T11:48:22Z","title":"EgoMe: A New Dataset and Challenge for Following Me via Egocentric View in Real World","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.19061","snapshot_observed_at":"2026-08-15T18:20:58.780213Z","title":"Egome: Follow me via egocentric view in real world","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.780213Z"},"links":{"cited_paper":"/paper/2501.19061","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:9ff28464bb944c0cb90dcc0a2fbae04721c7ad5601f72d1a9887505cf307c722","observation_id":"7a573c7e-9f12-4203-97d3-0fdadb4bdb56","resolution":{"observed_at":"2026-08-15T18:20:58.780213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02638","last_updated":"2024-07-16T11:27:04Z","snapshot_observed_at":"2026-08-16T14:37:33.248044Z","submitted_at":"2023-12-05T10:24:43Z","title":"Synchronization is All You Need: Exocentric-to-Egocentric Transfer for Temporal Action Segmentation with Unlabeled Synchronized Video Pairs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.02638","snapshot_observed_at":"2026-08-15T18:20:58.784500Z","title":"Synchronization is all you need: Exocentric-to-egocentric transfer for temporal action segmenta- tion with unlabeled synchronized video pairs, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.784500Z"},"links":{"cited_paper":"/paper/2312.02638","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:34c975959229b861503de3465dbbbe7d5da9fccb6111c00449c44d93f2785100","observation_id":"922753bc-70f8-40d7-b25d-958dfa7bb363","resolution":{"observed_at":"2026-08-15T18:20:58.784500Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.929631Z","title":"The meccano dataset: Understanding human-object interactions from egocentric videos in an industrial-like domain","venue":null,"work_id":"151195c3-8bdb-45d6-92ce-9f3abad10bfa","year":2021},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.788336Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:70c7789252ebd4e6506618ac321c794ca3f18bf2c1bdf76bbb6f3895fcaa292c","observation_id":"42fb4af6-ce29-4ddd-97e6-b9505cb88f72","resolution":{"observed_at":"2026-08-15T18:20:59.933386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.918194Z","title":"Home action genome: Cooperative compositional action understanding","venue":null,"work_id":"362b1db0-fc00-4af5-aea8-2b2bd742927a","year":2021},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.791786Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:ba3a4020ce302c959a4adf415a93f4d2c68e4d8929eeaa71bb7cd1a6914480a2","observation_id":"33f12641-5c0c-4e36-b1c9-49807e2031d9","resolution":{"observed_at":"2026-08-15T18:20:59.922384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.906211Z","title":"Watch and learn: the cognitive neuroscience of learning from others’ actions","venue":null,"work_id":"d9bbee8c-e1c6-4fb7-bcb7-10dd6ed4d0eb","year":2021},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.795433Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:dbb3e8e5ee7f3e31be01c0aa9fbc15c42ebb200902b6d8bf8d7b05cf7fc5ec6d","observation_id":"2d56ccb1-da42-45fb-8f52-bcac6016e0d7","resolution":{"observed_at":"2026-08-15T18:20:59.910231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.10084","last_updated":"2019-08-27T08:50:17Z","snapshot_observed_at":"2026-08-14T05:02:11.716316Z","submitted_at":"2019-08-27T08:50:17Z","title":"Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.10084","snapshot_observed_at":"2026-08-15T18:20:58.798955Z","title":"Sentence-bert: Sentence embeddings using siamese bert-networks","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.798955Z"},"links":{"cited_paper":"/paper/1908.10084","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:b1c0c7947d390c57cc748a6535dabbccefb7a0c1b4b5b5dd5b2cce5722ba558c","observation_id":"f1b7abe3-f17f-4e4d-a059-ab720e673389","resolution":{"observed_at":"2026-08-15T18:20:58.798955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07927","last_updated":"2025-03-16T06:23:34Z","snapshot_observed_at":"2026-08-16T14:09:52.485275Z","submitted_at":"2024-02-05T19:49:13Z","title":"A Systematic Survey of Prompt Engineering in Large Language Models: Techniques and Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07927","snapshot_observed_at":"2026-08-15T18:20:58.802920Z","title":"A systematic survey of prompt engineering in large language models: Techniques and applications","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.802920Z"},"links":{"cited_paper":"/paper/2402.07927","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:fbd9632ebb07fed8b3d3e58fe4fc8b3ce20713ee3a8fecb364cd2814836316a7","observation_id":"1f4cd9b7-d2ee-4c50-bc57-68db6b5e666b","resolution":{"observed_at":"2026-08-15T18:20:58.802920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.894688Z","title":"Sener, D","venue":null,"work_id":"5f0513c6-2a6c-41dc-897c-09faffb3dbc0","year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.806519Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:3e6a092b21d3d60f6ae1347a449f4989c3dc82961608f24d208a2d894984bcbb","observation_id":"3a7c4021-baec-4106-a456-4cbce44a88a1","resolution":{"observed_at":"2026-08-15T18:20:59.898336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.882634Z","title":"Self-supervised disentangled representation learning for third-person imitation learning","venue":null,"work_id":"224cb476-0be4-43fc-b9f1-81d15e4222b2","year":2021},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.809903Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:f158dbe4b32c989ddd50bd819ea94d78b7227bad8d7993bd6d9c3bce937f757a","observation_id":"acded3b6-d552-48e6-ae4b-96d140bfdd78","resolution":{"observed_at":"2026-08-15T18:20:59.886385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.871150Z","title":"Third-person visual imitation learning via decoupled hierarchical controller","venue":null,"work_id":"920101ca-a650-46b1-b6d4-da94db81766c","year":2019},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.813364Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:bc079d565cad0af1086390107a015bf16209f492276bd52a5ca466c3c6c8b47a","observation_id":"6c33268e-7a7f-4cba-86ea-05c5b4051135","resolution":{"observed_at":"2026-08-15T18:20:59.875278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.859618Z","title":"Actor and observer: Joint modeling of first and third-person videos","venue":null,"work_id":"59bc9f12-d83c-43ba-a838-a637bf8002c1","year":2018},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.816505Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:bd04eb64cd85b227ec9fa3efc130db628c9eacba8c14f88657d968d162d25bd5","observation_id":"defd28b4-0e73-4694-a699-09591d2c2629","resolution":{"observed_at":"2026-08-15T18:20:59.863436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.847783Z","title":"Ego4d goal-step: Toward hierarchical understanding of procedural activities","venue":null,"work_id":"ecd3d1cf-0279-4560-9f9a-5c04b2304d66","year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.819680Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:1832ce76b97808194033538f3464822f7de958f41da90f4ea569cde7fb650a94","observation_id":"08cbaa41-583a-40de-8add-546444c82e9a","resolution":{"observed_at":"2026-08-15T18:20:59.851917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1212.0402","last_updated":"2012-12-03T14:45:31Z","snapshot_observed_at":"2026-08-16T21:21:44.768787Z","submitted_at":"2012-12-03T14:45:31Z","title":"UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1212.0402","snapshot_observed_at":"2026-08-15T18:20:58.822961Z","title":"Ucf101: A dataset of 101 human actions classes from videos in the wild, 2012","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.822961Z"},"links":{"cited_paper":"/paper/1212.0402","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:08aa71e57ca71dc189e3de9e674fb60dd92c138061f238ae7c837e543a7865ee","observation_id":"3199f91d-1a2c-4add-8a29-6bf69c572b99","resolution":{"observed_at":"2026-08-15T18:20:58.822961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.826248Z","title":"Qwen2.5-vl, January 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.826248Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:5b941d17d17e77e6da4e14a853941cc7fe908a368653aa44f43dbb1d39f87d76","observation_id":"4a022816-f474-4dee-a8d4-91f07793b44c","resolution":{"observed_at":"2026-08-15T18:20:58.826248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.829108Z","title":"Learning from semantic alignment between unpaired multiviews for egocentric video recognition","venue":null,"work_id":"8365b062-2c3c-4b04-93d8-f185e1849edf","year":2023},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.830251Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:50eace55a29f6f436de374ce35063884b92ce737657e767fb019d87c1164215e","observation_id":"7df283ff-6895-4807-82ea-4fc136f625cf","resolution":{"observed_at":"2026-08-15T18:20:59.833739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-08T19:38:26.415599Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08035","snapshot_observed_at":"2026-08-15T18:20:58.833497Z","title":"Lvbench: An extreme long video understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.833497Z"},"links":{"cited_paper":"/paper/2406.08035","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:e625cc4b0d3552b871412c28ee1fd7e17c358dcec22469f6143c28e6758271be","observation_id":"cb910283-684b-4147-a1b4-b580779b192c","resolution":{"observed_at":"2026-08-15T18:20:58.833497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-08-16T01:49:22.176843Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-15T18:20:58.837266Z","title":"Self-consistency improves chain of thought reasoning in language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.837266Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:3cca2a0b5b934f96e58a18f1c7f9e5b7c877bd268a363951b823208879f6b248","observation_id":"df3eec7f-ade8-4dd1-aa0c-724fef2db1ab","resolution":{"observed_at":"2026-08-15T18:20:58.837266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.817757Z","title":"See what i see: Enabling user-centric robotic assistance using first-person demonstrations","venue":null,"work_id":"68932edf-f70d-45d5-99c7-139a231e244a","year":2020},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.840921Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:94f131b7a0f829917d0f9bcffbb4d5b8724ded49ac513e2a261904a119d509cc","observation_id":"d85953e3-5241-408d-bb23-e2caedd1daad","resolution":{"observed_at":"2026-08-15T18:20:59.821670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.844180Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.844180Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:f9f62e2e4d037be14033d56a16c08ebf914a465b367c487d2c1fc1f612bc97a3","observation_id":"64bbbf1c-abf3-445b-b7c0-9aa7b7079950","resolution":{"observed_at":"2026-08-15T18:20:58.844180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.798934Z","title":"Incomplete multi-view domain adaptation via channel enhancement and knowledge transfer","venue":null,"work_id":"edff4184-3dd9-41ed-b976-60736ceecef9","year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.847409Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:59ba0151a729bf27ccdb2b4c2e436823b0ef17e01be4ded9050c740b0853093a","observation_id":"8b429db5-0f38-4f96-b01b-9955e9e23fba","resolution":{"observed_at":"2026-08-15T18:20:59.802996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.850945Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.850945Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:40654b7efac8808043e6bc0cb61bf2f12bdfb579346c8e81d7e5d0efd634ba84","observation_id":"b44d3de0-c7ab-4f5a-a924-ee2341e72d14","resolution":{"observed_at":"2026-08-15T18:20:58.850945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.854230Z","title":"Can i trust your answer? visually grounded video question answering","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.854230Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:66b1e9bae1d5bbbd27d31e13cff93420434c1a958d9d3063570d1d41b7763068","observation_id":"518387af-c61e-4d7e-9ad7-7eadf9dcc0c3","resolution":{"observed_at":"2026-08-15T18:20:58.854230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.773692Z","title":"Pov: Prompt-oriented view-agnostic learning for egocentric hand-object interaction in the multi-view world","venue":null,"work_id":"691cdc8c-a2b6-4968-be93-26cc15d2008a","year":2023},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.857657Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:0700522787bcd83f4a4b942c1bd14103fa6a9fa50a07cc16024212156ac770e7","observation_id":"f7a65284-674c-459a-819d-2171ddf567cc","resolution":{"observed_at":"2026-08-15T18:20:59.777635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.762799Z","title":"Retrieval-augmented egocentric video captioning","venue":null,"work_id":"24e3b135-1412-4e7f-8252-66a39014aa8a","year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.860995Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:bb12fe204e831ce102673fb1201029c9134aed635d4969d2f6b8d3610a20aac2","observation_id":"8e7d8b54-8e8c-480d-8c44-f16eef383f26","resolution":{"observed_at":"2026-08-15T18:20:59.766885Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11732","last_updated":"2025-04-16T03:12:39Z","snapshot_observed_at":"2026-08-16T12:40:38.319816Z","submitted_at":"2025-04-16T03:12:39Z","title":"EgoExo-Gen: Ego-centric Video Prediction by Watching Exo-centric Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.11732","snapshot_observed_at":"2026-08-15T18:20:58.864319Z","title":"Egoexo-gen: Ego-centric video prediction by watching exo-centric videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.864319Z"},"links":{"cited_paper":"/paper/2504.11732","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:f361f38e0f97c25a4a5712b8537973d141e9a2372f9fd271a094e0f4de882d8d","observation_id":"5dcc23a0-8e38-46b0-88af-9278962a70a8","resolution":{"observed_at":"2026-08-15T18:20:58.864319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.751896Z","title":"Advancing high-resolution video-language representation with large-scale video transcriptions","venue":null,"work_id":"adf18ed5-3d38-4254-91db-d62ac5a2cac1","year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.867879Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:67d148c0a27f926261d4efd205d23d2e53947d50ab37f0cef39ab0d68821cb92","observation_id":"834eaabb-6377-4be5-a98a-e4bd841d5949","resolution":{"observed_at":"2026-08-15T18:20:59.756120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-15T18:20:58.871155Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.871155Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:f9bd39449954542261c0dc9eedb4e375fec5dd67b58894daa763fcc2aff30c6c","observation_id":"596d892d-73ae-473a-b003-c984dd3d0763","resolution":{"observed_at":"2026-08-15T18:20:58.871155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-08-18T15:30:47.457779Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14171","snapshot_observed_at":"2026-08-15T18:20:58.874554Z","title":"Thinking in space: How multimodal large language models see, remember, and recall spaces.arXiv preprint arXiv:2412.14171, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.874554Z"},"links":{"cited_paper":"/paper/2412.14171","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:570c298e0544577f9afec41d8b8b8ab63d7696166bd80e2d81cc4c08fc823aeb","observation_id":"c9136d0f-360b-4248-943e-1fa6c86beb79","resolution":{"observed_at":"2026-08-15T18:20:58.874554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:58.878139Z","title":"Egolife: Towards egocentric life assistant","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.878139Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:ee52f26591da8a136f3a4fda89779981d02cbf70735cc161330564d7024c9b24","observation_id":"dc6b34ac-2dc9-48c3-b89d-d1a83c4a3c1a","resolution":{"observed_at":"2026-08-15T18:20:58.878139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.740863Z","title":"Interact before align: Leveraging cross-modal knowledge for domain adaptive action recognition","venue":null,"work_id":"a81fdeb3-6877-4d8f-996a-43ba6fe90866","year":2022},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.881438Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:c7224c10f81e01a24be57c3de3c5607e50d4c9d11728b2408ca9dd31c4c75a31","observation_id":"c8b282ef-3762-4dd1-8294-00a2c0a0d065","resolution":{"observed_at":"2026-08-15T18:20:59.744952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.729997Z","title":"Fine-grained affordance annotation for egocentric hand-object interaction videos","venue":null,"work_id":"eb5cd421-f741-43e6-90f7-d5aede8981a4","year":2023},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.884973Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:af8fcd970bd5ab43eabdb514d631af9921490fa493666bf14d0407bcb7a03e72","observation_id":"0580ab10-c00b-4d0d-a4f4-59aefd1d49d3","resolution":{"observed_at":"2026-08-15T18:20:59.734002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-15T18:20:58.888448Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.888448Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:cafee2166716f7cfab1574b776227f0477874a701afd92281c7eb13187b94b81","observation_id":"7a9b7ffe-15ef-43b2-8076-e826bd277000","resolution":{"observed_at":"2026-08-15T18:20:58.888448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.717968Z","title":"Fusing personal and environmental cues for identification and segmentation of first-person camera wearers in third-person views","venue":null,"work_id":"a1c94816-34fd-47f4-8587-8255526f27a4","year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.892194Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:4653be2cae87503cfa34fc610a2662e9fc7842a5458c746ae7c31e05c08ea10d","observation_id":"375f7e57-2fd9-4c80-ac91-7aa5e1ac6ea9","resolution":{"observed_at":"2026-08-15T18:20:59.722825Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09797","last_updated":"2024-10-07T04:28:04Z","snapshot_observed_at":"2026-08-16T15:39:13.160910Z","submitted_at":"2023-04-19T16:29:48Z","title":"Progressive-Hint Prompting Improves Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09797","snapshot_observed_at":"2026-08-15T18:20:58.895550Z","title":"Progressive-hint prompting improves reasoning in large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.895550Z"},"links":{"cited_paper":"/paper/2304.09797","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:3390bf0c2fd2826f3dc835dc248472fa87bcf303a06624aa43ff611597cb089e","observation_id":"c3c402f3-cf2f-4f73-ae74-400c88736e56","resolution":{"observed_at":"2026-08-15T18:20:58.895550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-15T18:20:58.899257Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.899257Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:8c4adfc976a802e54ae1301540572930ee8f82a9d11cc37bee6406a466886e4c","observation_id":"c27682d3-9ae7-456c-a81e-129857e924a0","resolution":{"observed_at":"2026-08-15T18:20:58.899257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-17T09:56:52.502317Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-15T18:20:58.902748Z","title":"Internvl3: Exploring advanced training and test-time recipes for open-source multimodal models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.902748Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:37c0599078e7d985a7f795265939a47a32faa4062696946002b1aee36f90b28b","observation_id":"808b2c60-d78f-4e2e-9c40-7b72bb381f05","resolution":{"observed_at":"2026-08-15T18:20:58.902748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.707298Z","title":null,"venue":null,"work_id":"1545d8af-e8c9-4e03-83f4-6314ad730985","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.906977Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:0a0f9bc2a1b1323751b2be14a43d7067bd867ac3db50bab47280a8b03138e2b6","observation_id":"f24dfd04-ffee-4dec-9b00-3bb9237752ab","resolution":{"observed_at":"2026-08-15T18:20:59.711139Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.697018Z","title":null,"venue":null,"work_id":"c4625658-8e5c-48ae-9801-5fe3741afd57","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.910865Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:689f8d6234c8798a40f7e3f4d5ec23a8ccf24c5c5cbbf25c64cd61535e8d80c9","observation_id":"d183422f-1521-4f58-b83f-5aced6b6d0e7","resolution":{"observed_at":"2026-08-15T18:20:59.700559Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.686537Z","title":null,"venue":null,"work_id":"19ec4cf3-0776-4cc8-a8e2-5124851f28cc","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.914567Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:bd47124d39b6e1553baf20dce71d794551c6d1368a71cedec913e2eef4f955b5","observation_id":"275dc58b-0209-4a90-9ede-77d618066f60","resolution":{"observed_at":"2026-08-15T18:20:59.689958Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.675955Z","title":null,"venue":null,"work_id":"67a2e80c-b3c7-4f91-99a6-475c0962f6a9","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.917998Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:5125be05d84d8733e59de9b5ef395a1df9c6e0986e0e95e3f9fac777346434b8","observation_id":"991598cc-7fe2-4422-956b-b2630b39cd97","resolution":{"observed_at":"2026-08-15T18:20:59.679413Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.664836Z","title":"Question","venue":null,"work_id":"3afb8b90-a31a-4e58-9514-73853387f32d","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.921699Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:0fa1496bd283c14f523438cc26d6f4943f67708587dc22eaa1b52337306b48dc","observation_id":"3b0e048e-607e-41fe-8c3c-3f58574c13de","resolution":{"observed_at":"2026-08-15T18:20:59.668778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.654733Z","title":null,"venue":null,"work_id":"ed603bdc-c50c-46e3-9083-86286554b840","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.925362Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:ce76ca7aaacd7cb89fc5f3435e5b936b7b4540a74db311826516f53bc064c977","observation_id":"0f43294f-02f9-478d-831f-db6855b19560","resolution":{"observed_at":"2026-08-15T18:20:59.658026Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.644401Z","title":"Video 1\" and the second video as","venue":null,"work_id":"46339bbb-3bc7-40ed-84e1-111e095b0b1d","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.929948Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:78d8c146418019a94fb12013403914e4aa61aa1ce56271be36a4378223590fbf","observation_id":"282679b6-7fd9-4da3-9f4a-9086d0ab3dfb","resolution":{"observed_at":"2026-08-15T18:20:59.647911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.633246Z","title":null,"venue":null,"work_id":"02274a9f-be13-4e1a-be27-e7eee0765763","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.933851Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:3d247c429004b1b95d12c4601a31141149879bf2286a3fb07429312d3b2f3596","observation_id":"a3489ba2-d2e2-4206-affa-6062b7194dbf","resolution":{"observed_at":"2026-08-15T18:20:59.636852Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.622028Z","title":null,"venue":null,"work_id":"5f3383e0-e768-49bb-88e3-ca0ca472418d","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.937450Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:97101f1b2c2f51a59cf906b79afbb95dea0fa11bf7882880b44e300619b0fa6e","observation_id":"785ac986-5495-4259-bf75-900708b3c800","resolution":{"observed_at":"2026-08-15T18:20:59.625591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.612044Z","title":null,"venue":null,"work_id":"10c87b50-c8e5-4779-b5e1-facc19f54eed","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.940976Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:86f64a7225ff53e2f6472616c604d5d6521e126c077faca1263aa24029097874","observation_id":"b4d7efa8-aef5-4ec6-9178-0a70a44bceb6","resolution":{"observed_at":"2026-08-15T18:20:59.615422Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.599856Z","title":"Video 2:","venue":null,"work_id":"28b0f890-120a-45e3-9b10-93c9616fb11d","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.944844Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:653f3207ba3e755b15a741550e889c6ead3f09b9bf22b421af45e1ab0d22041d","observation_id":"4e6a24d5-3896-44e0-99fc-5e66bbd0fca1","resolution":{"observed_at":"2026-08-15T18:20:59.603633Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.588197Z","title":null,"venue":null,"work_id":"12b3297e-6edc-4b73-a393-edbb5aa640bd","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.948326Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:f993639887367342f3d3c25b5bf2dd9cc53b2bdc1c6f7a6d97296d2e86f1e63c","observation_id":"d323f144-c266-40cf-9e2e-cdd767555048","resolution":{"observed_at":"2026-08-15T18:20:59.591877Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.576789Z","title":null,"venue":null,"work_id":"0d5169ac-b5fa-4da4-8262-3494885cc9f3","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.951798Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:caf3f69076ef784ff544c0c061809122766418913390f712072d7e4bbfad990c","observation_id":"5ff29efd-15c9-4534-afab-5b658022d0fb","resolution":{"observed_at":"2026-08-15T18:20:59.580983Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.564726Z","title":"Conclusion: After reviewing the sequences, Option B correctly describes the actions","venue":null,"work_id":"0d8fd24a-cb1e-46ad-9247-c01ee623cb96","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.955466Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:45c3dd0551a3af6ac0c760527814838c030368c2fd727a640cf603afb52a69b2","observation_id":"828da7f9-eb37-450d-bc93-7f7af754d9d5","resolution":{"observed_at":"2026-08-15T18:20:59.568475Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.553284Z","title":null,"venue":null,"work_id":"a2f1dbce-ab44-4dc4-a1ee-9ee5ee2a8acb","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.959906Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:3cdad2629565bfda69524917a7ca0badf729d67959cd1931279b471eec1b42b4","observation_id":"d9295f79-8092-4f58-9914-c2b31e807622","resolution":{"observed_at":"2026-08-15T18:20:59.557191Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:20:59.539112Z","title":"Video 2: The person within the bounding box is positioned at the bottom of the scene, facing upwards, and is in a different area relative to others","venue":null,"work_id":"485f5846-e665-42f1-bb06-5886031757c8","year":null},"citing_paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:58.963152Z"},"links":{"citing_paper":"/paper/2507.18342"},"observation_digest":"sha256:d2cd4d94eef6c67dbc23bc57c0e7b0189f94ced74300f8c51f714f19ddf4261a","observation_id":"a60e31f7-1397-42b0-b532-4d555af401ab","resolution":{"observed_at":"2026-08-15T18:20:59.545118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.18342","last_updated":"2025-07-24T12:14:49Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T08:41:11.392803Z","submitted_at":"2025-07-24T12:14:49Z","title":"EgoExoBench: A Benchmark for First- and Third-person View Video Understanding in MLLMs"},"reference_resolution":{"displayed":95,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":62,"verified_exact":0,"verified_fuzzy":33},"total_outbound_references":95},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 95 of 95 outbound references and 15 inbound Pith citation observations for arXiv:2507.18342."}