{"as_of":"2026-07-25T02:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f6b7da3a98407ac2d643aab801d22c9ec4629aad508ebb0831ab359e00cf273a","coverage":[{"denominator":50,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-17T02:49:12.987772Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-07-24T06:31:00.690269+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T19:07:48.425181Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-04T20:40:07.858196Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"cited_work":{"arxiv_id":"2512.01707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.01707","snapshot_observed_at":"2026-07-04T20:40:07.858196Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","venue":"cs.CV","work_id":"3565c50f-f747-43f6-922f-cc27e8bbd235","year":2025},"citing_paper":{"arxiv_id":"2604.20361","last_updated":"2026-04-22T09:00:51Z","snapshot_observed_at":"2026-07-06T23:06:49.210284Z","submitted_at":"2026-04-22T09:00:51Z","title":"Object Referring-Guided Scanpath Prediction with Perception-Enhanced Vision-Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T00:42:32.031949Z"},"links":{"cited_paper":"/paper/2512.01707","citing_paper":"/paper/2604.20361"},"observation_digest":"sha256:6992d5a3277096a8f5c58c6876dcc2420394b89970ddf36d819bc274d591a872","observation_id":"35429678-97a8-451e-8ca8-2ebca5e0d55f","resolution":{"observed_at":"2026-05-14T01:55:05.995540Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"cited_work":{"arxiv_id":"2512.01707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.01707","snapshot_observed_at":"2026-07-04T20:40:07.858196Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","venue":"cs.CV","work_id":"3565c50f-f747-43f6-922f-cc27e8bbd235","year":2025},"citing_paper":{"arxiv_id":"2606.00523","last_updated":"2026-05-30T04:31:29Z","snapshot_observed_at":"2026-07-06T23:41:15.410587Z","submitted_at":"2026-05-30T04:31:29Z","title":"ProactiveLLM: Learning Active Interaction for Streaming Large Language Models","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-06-28T19:07:48.425181Z"},"links":{"cited_paper":"/paper/2512.01707","citing_paper":"/paper/2606.00523"},"observation_digest":"sha256:b7701a6b31fb558f136bcc19958e1e2f0d6ff079cb152a662e0192404c0f746e","observation_id":"1730a785-f3a9-4f85-984f-17cebf551712","resolution":{"observed_at":"2026-06-28T19:12:34.881476Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"cited_work":{"arxiv_id":"2512.01707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.01707","snapshot_observed_at":"2026-07-04T20:40:07.858196Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","venue":"cs.CV","work_id":"3565c50f-f747-43f6-922f-cc27e8bbd235","year":2025},"citing_paper":{"arxiv_id":"2606.26080","last_updated":"2026-06-24T17:54:08Z","snapshot_observed_at":"2026-07-07T00:00:26.827253Z","submitted_at":"2026-06-24T17:54:08Z","title":"Neglected Free Lunch from Post-training: Progress Advantage for LLM Agents","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-06-25T19:55:51.114244Z"},"links":{"cited_paper":"/paper/2512.01707","citing_paper":"/paper/2606.26080"},"observation_digest":"sha256:70918b6246b0a4f62707564f17ebdae68e2179d46197796038b5c82a7b1e36b5","observation_id":"387777e5-f6f8-4103-afe0-44235cae8f26","resolution":{"observed_at":"2026-07-04T20:40:07.859463Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2512.01707/citation-record","integrity":"/paper/2512.01707/integrity","json":"/paper/2512.01707/citation-record.json","paper":"/paper/2512.01707"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-07-11T01:17:45.013705Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:1e994a15245f961d3d1f5c94c95fc85e4e9962c0ce620996746f00a17053bd18","observation_id":"de7e54bc-e1b7-409a-a4cd-ba871d6ef8ba","resolution":{"observed_at":"2026-05-17T02:51:28.123876Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videollm-online: Online video large language model for streaming video","venue":null,"work_id":"885d8c04-c026-483a-82e5-a62573ca4f94","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:ca59f69023228568406e30498d84dc3ba56b9389cb9532712f5d5614b4837d25","observation_id":"b7b5dcb8-e364-48b8-acb4-f86011bd7f29","resolution":{"observed_at":"2026-05-17T02:51:28.854110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01957","last_updated":"2025-10-24T02:32:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-03T18:59:52Z","title":"VITA-1.5: Towards GPT-4o Level Real-Time Vision and Speech Interaction","version":4},"cited_work":{"arxiv_id":"2501.01957","doi":"10.48550/arxiv.2501.01957","metadata_source":"pith","pith_arxiv_id":"2501.01957","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"VITA-1.5: Towards GPT-4o Level Real-Time Vision and Speech Interaction","venue":"cs.CV","work_id":"510354f3-222a-48da-bda9-96a2df30bb68","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2501.01957","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:79dd785569267811480d6e0bff1a61833d74d547915fd259374370f0e2309c41","observation_id":"522c58d7-b179-40e6-af9a-f950cac52035","resolution":{"observed_at":"2026-05-17T21:08:19.955297Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:18:46.657095+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:18:46.657095+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12769","last_updated":"2025-03-17T03:05:31Z","snapshot_observed_at":"2026-07-06T20:53:41.124209Z","submitted_at":"2025-03-17T03:05:31Z","title":"ViSpeak: Visual Instruction Feedback in Streaming Videos","version":1},"cited_work":{"arxiv_id":"2503.12769","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.12769","snapshot_observed_at":"2026-07-02T19:07:17.481649Z","title":"Vispeak: Visual instruction feedback in streaming videos","venue":null,"work_id":"ff02f96e-6220-439c-a2d7-ebe7d4b79876","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2503.12769","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:d9d35b631952f49e0f209a6a1b3f9cf14e6b47ab0e5c3c4c5ccf354eefa99e46","observation_id":"b09d52e4-25f3-43c1-b130-33c9e61191ff","resolution":{"observed_at":"2026-05-17T02:51:28.106200Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":"51dabc78-b062-4e25-a2e5-1c7b1410cb2d","year":2022},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:00aa9deec0533be47a2b83003f81f98a59229de98f326ccd8c4c0145feb89423","observation_id":"ce23f90e-b794-477e-b061-4cced6bee74c","resolution":{"observed_at":"2026-05-17T02:51:28.840730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Egoexolearn: A dataset for bridging asyn- chronous ego-and exo-centric view of procedural activities in real world","venue":null,"work_id":"04a9a1d9-7119-4e23-aef2-a3e92733fc20","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:16eac7b67e68d3217b59619e324b69c68f99df50ada7da760443482b9dd2788a","observation_id":"39995b52-d539-49a3-9365-06c485055cf3","resolution":{"observed_at":"2026-05-17T02:51:28.842740Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gazevqa: a video question answering dataset for multiview eye-gaze task- oriented collaborations","venue":null,"work_id":"f976568c-dc7b-458b-8cd6-0138888f793c","year":2023},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:1f3824235280336b47d1db1152d53f7a038356075d5de4ff59452c9af6474992","observation_id":"cf772ea9-246a-46bb-9df0-f435514e81fe","resolution":{"observed_at":"2026-05-17T02:51:28.838499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17545","last_updated":"2024-03-26T09:49:35Z","snapshot_observed_at":"2026-07-06T17:50:54.472429Z","submitted_at":"2024-03-26T09:49:35Z","title":"A Gaze-grounded Visual Question Answering Dataset for Clarifying Ambiguous Japanese Questions","version":1},"cited_work":{"arxiv_id":"2403.17545","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17545","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A gaze-grounded visual question answering dataset for clarifying ambiguous japanese questions","venue":null,"work_id":"5631ec6f-4935-4ce7-87de-dc41d16cdbf6","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2403.17545","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:e9ccbc8bedb2829a7405bbd85f5a4f0a734e44dec0ed7520adec0403e1f371ef","observation_id":"e4bda8a4-bc75-4d72-8acf-a24e8e24d414","resolution":{"observed_at":"2026-05-17T02:51:28.080643Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Foveal vision anticipates defining features of eye movement targets.Elife, 11:e78106","venue":null,"work_id":"74090ae2-11a6-4083-82e9-a1f9e2c6d070","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:2ae0de17da1f112e529244d9d094df3d6b3c97bd03e0c6d110a550f94ae53db7","observation_id":"5cd37902-e0e8-47ef-8d21-70ce00d02de0","resolution":{"observed_at":"2026-05-17T02:51:28.846959Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Listen to look into the future: Audio-visual egocentric gaze anticipation","venue":null,"work_id":"a7935255-ca2c-4a72-a19e-2979f8263504","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:c4acfb17d9c5e71fa2564fb631b6eb478cdf4f9aebf7f75893167116bc2ee767","observation_id":"e14583e9-027b-4f10-9e51-912c074c1efc","resolution":{"observed_at":"2026-05-17T02:51:28.849015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In the eye of the beholder: Gaze and actions in first person video.IEEE trans- actions on pattern analysis and machine intelligence, 45(6): 6731–6747","venue":null,"work_id":"4f873767-488d-4f3e-a229-83879f0bfc2d","year":2021},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:1e3ff49ef8f44d4d4eff80ba2020f9ede9ef3985545a2b2c3963049f726add83","observation_id":"7f8536b5-e76f-49ca-945c-f38c4ede5493","resolution":{"observed_at":"2026-05-17T02:51:28.834238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":"2311.10122","doi":"10.48550/arxiv.2311.10122","metadata_source":"pith","pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","venue":"cs.CV","work_id":"e2121c51-a55e-476a-af81-7ba6970fe6cf","year":2023},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:124001a8729da9a36de8d3df508b7120008dd40c0af68827d9f550abbae06f2f","observation_id":"f9defde7-f4b3-4667-873d-bbe294f07dbd","resolution":{"observed_at":"2026-05-17T02:51:28.090819Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-07-06T19:45:57.808548Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":null,"work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:1e707fa45c206ff89cc317378fa30e7cb0bbd87859f8ca71f47626dc99750056","observation_id":"9b5af3be-a259-4fad-ac21-759bdd354340","resolution":{"observed_at":"2026-05-17T02:51:28.116389Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:03ccc3f88d118b9cb9292a15edac8bd28971ca3d18804168351f10233d7fba0a","observation_id":"b5c68ae3-7955-4fa6-b102-f464b8aa660a","resolution":{"observed_at":"2026-05-17T02:51:28.132505Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ovo-bench: How far is your video-llms from real-world online video understanding? InProceedings of the Computer Vision and Pattern Recognition Conference, pages 18902–18913","venue":null,"work_id":"4dea8b04-a109-4e62-a349-ba4490102e85","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:a5cf41811d06ba697f925c5d8f5b8b153415ed3042287f60b60b85c234bad1bd","observation_id":"3b2d9316-097f-40eb-b4fe-611753a5b0e7","resolution":{"observed_at":"2026-05-17T02:51:28.875244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual search in naturalistic scenes from foveal to peripheral vision: A compar- ison between dynamic and static displays.Journal of Vision, 22(1):10–10","venue":null,"work_id":"4dab8687-e168-4cd7-8933-fbd64296b7de","year":2022},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:aa2249fe1f176f52693b5dece3baeb0d0d4fdd256eb3f1e4005710c7d47d55af","observation_id":"0c99c62b-87ba-4de8-aa66-1ea80a3504b7","resolution":{"observed_at":"2026-05-17T02:51:28.877965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-4 technical report","venue":null,"work_id":"08358fa8-1a20-4d81-8535-793a27e4836b","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:e921117829990055e3206d57b169c50a53417430b5965b2c8c9c0e0fe4c2b952","observation_id":"af7829c2-81c4-4d95-a89e-5bd08fd9e95e","resolution":{"observed_at":"2026-05-17T02:51:28.863112Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing claude 4","venue":null,"work_id":"ce27c950-61e5-4ead-b236-dd4ae9820686","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:9f4fa0a70d7fb8497b7a08ab7fb0095b93283a7908181e92a4283dd28d150f2d","observation_id":"7e4eb0a7-85e6-4388-a299-628bd3e1da32","resolution":{"observed_at":"2026-05-17T02:51:28.867794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.07447","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T06:46:52.305785Z","title":"In the eye of mllm: Benchmarking egocentric video intent understanding with gaze-guided prompting","venue":null,"work_id":"04326be3-9128-4727-845c-6fb15a1bc3c4","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:8591656c2a787f46b61e0b6415524042071f9fc172d75db9413c32649d95fe66","observation_id":"38e710d8-6a02-4221-a290-5b54982e586a","resolution":{"observed_at":"2026-05-17T02:51:28.102625Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hd-epic: A highly-detailed egocentric video dataset","venue":null,"work_id":"5cd48bc8-e0e4-4888-847b-d62d5b4a7335","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:425010f974e0f08459c2740d45bac1e53013861ad7bb74e003608cb5b08e716e","observation_id":"31f15217-dee6-4c68-8b39-b33407f96e7e","resolution":{"observed_at":"2026-05-17T02:51:28.861263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dispider: Enabling video llms with active real-time interaction via dis- entangled perception, decision, and reaction","venue":null,"work_id":"7f191253-9c99-4c21-8d6a-0b82893dbbf1","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:b1d870d57d15845a30a0cd98c0010e2cc75da689699df85271b163968cb3e034","observation_id":"ab3b5bb1-498b-420a-86e5-fcc54f97e142","resolution":{"observed_at":"2026-05-17T02:51:28.865394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Radwan, Nancy M","venue":null,"work_id":"afeb378e-26ba-4772-a6e4-b0b335726fd7","year":2012},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:5754d309b89e8efc90a398a99a987fcc4e6eed1a258af3d994f1488ee4027e90","observation_id":"a1874621-9fe7-49fe-9f44-cebd0c41fe81","resolution":{"observed_at":"2026-05-17T02:51:28.870171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gazellm: Multimodal llms incorporating human visual attention","venue":null,"work_id":"aa0b180f-f514-42e0-99e4-29750203ec9e","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:8fb0e26ffdd4f8868a7efedb1ab4784749cc1ef98cf5f45458fc32e592143631","observation_id":"0ac20ab5-544e-4569-bd1d-49902cc80d4b","resolution":{"observed_at":"2026-05-17T02:51:28.872637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A review of interactions between peripheral and foveal vision.Journal of vision, 20(12):2–2","venue":null,"work_id":"47cd94c7-abea-449f-831e-16c65c5999b9","year":2020},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:4602ffe03cccbb3298163d6062d381106dfa1b85212c032dc2f2396b453338ea","observation_id":"ed598490-c879-4bc1-a6f2-52887244dc30","resolution":{"observed_at":"2026-05-17T02:51:28.859025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":"9c9b5e6b-2450-4039-af23-f608b1a41764","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:db13a0ad9dbbd62ee792ad327da11e8c2f99708f1e2e0ed3f3b719c7f0705418","observation_id":"c9e80567-9b6c-41ff-97a9-01068526cbd1","resolution":{"observed_at":"2026-05-17T02:51:28.851994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"cited_work":{"arxiv_id":"2508.18265","doi":"10.48550/arxiv.2508.18265","metadata_source":"pith","pith_arxiv_id":"2508.18265","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","venue":"cs.CV","work_id":"b8f5e260-fff5-444e-bcf5-2c42cfefd83d","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2508.18265","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:30b6ba8e9606c3685cf515f7109cd431f0936419fe0612aef10fc871b46e2159","observation_id":"7ca2feb5-cb30-439b-91b8-c9690251c1aa","resolution":{"observed_at":"2026-05-17T02:51:28.120260Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Holoassist: an egocen- tric human interaction dataset for interactive ai assistants in the real world","venue":null,"work_id":"c50662ee-9833-4246-b7f1-61af41ab2d52","year":2023},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:240006bd8ec4263f40312aaf5bbb764f0bb2176c80ed6c5734d67716169d8d57","observation_id":"8b19bf8e-08fd-4867-888c-4f646b3b41e1","resolution":{"observed_at":"2026-05-17T02:51:28.856648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Omnimmi: A comprehensive multi- modal interaction benchmark in streaming video contexts","venue":null,"work_id":"0a8605e7-0963-424a-8b5f-95b62a5e2aca","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:07376ccf2cee0441b800dad71cb17e0b10589d1f55f2c13f95ff7e745283b733","observation_id":"a0a17614-505e-4eee-9d87-bb709e06b54c","resolution":{"observed_at":"2026-05-17T02:51:28.808755Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13468","last_updated":"2025-01-23T08:33:10Z","snapshot_observed_at":"2026-07-06T20:24:50.262766Z","submitted_at":"2025-01-23T08:33:10Z","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","version":1},"cited_work":{"arxiv_id":"2501.13468","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13468","snapshot_observed_at":"2026-07-04T03:19:29.878882Z","title":"Streaming video under- standing and multi-round interaction with memory-enhanced knowledge","venue":null,"work_id":"05bede93-84f8-4b7b-b6e5-10f6ff02394b","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2501.13468","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:ddd05047eeade15c7dd74d26532db80758c0ce3fefc3f2dbb84612edbf702328","observation_id":"96708282-2c6f-4961-b694-275db5656e7e","resolution":{"observed_at":"2026-05-17T02:51:28.098757Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:977fdd3358e9812ca348b09b53daca76a7d1b26c111fa98af1d4ed06ae5ff92a","observation_id":"2b475592-85df-4976-87b3-562c894d502c","resolution":{"observed_at":"2026-05-17T02:51:28.112365Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.10810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T12:41:04.328023Z","title":"Svbench: A benchmark with temporal multi-turn dialogues for streaming video understanding","venue":null,"work_id":"a96af222-b449-4d3e-b31d-c2fae15db9fd","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:40506ef1af139b69f270389992c9371515ca89a82c7c470fd77cbd9f84f9fb74","observation_id":"c13d360d-ccc9-4d89-a6ab-b0d0404bd68f","resolution":{"observed_at":"2026-05-17T02:51:28.094925Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"cited_work":{"arxiv_id":"2408.01800","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.01800","snapshot_observed_at":"2026-07-10T11:37:03.161139Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","venue":"cs.CV","work_id":"0f06e436-0c76-4e3c-be5e-6168f6bc4336","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2408.01800","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:3ca6e801605ac3d44fefa10d5ab0dd95914263498a509183c9ab903ade6fecc5","observation_id":"1373f7a8-472b-4681-8417-8840cf26a498","resolution":{"observed_at":"2026-05-17T02:51:28.087775Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":"2306.02858","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-07-04T16:29:57.761121Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","venue":"cs.CL","work_id":"555cf04a-49a7-44b8-9019-a83ce85ace95","year":2023},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:0b797c17904e29a7b31efeb634652bcd85b1eeafc7b94fdb266f1364c2887d42","observation_id":"c38753e3-9afb-4875-b0e9-2ce8143c3b8c","resolution":{"observed_at":"2026-05-17T02:51:28.109262Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Flash-vstream: Memory- based real-time understanding for long video streams","venue":null,"work_id":"bd7f5b25-e587-4b62-8464-6deb268766bf","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:b750c9a24efb3330ccadb1c21dd94871486ec1b36fb489b5c922ca94b29ba3d5","observation_id":"b5c7d10b-2d00-462b-ad53-709a6a8ab4d1","resolution":{"observed_at":"2026-05-17T02:51:28.818293Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deep future gaze: Gaze anticipation on egocen- tric videos using adversarial networks","venue":null,"work_id":"fb2a32f7-567b-43ab-b8dd-a4c032a5c863","year":2017},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:11f2760a4187c7b200ce77a2ee72bc5cf3531defe5c751c5fdaf3b7a01ef7ebd","observation_id":"ff2ce6cf-8626-4475-b0ba-4e0385481f91","resolution":{"observed_at":"2026-05-17T02:51:28.822798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05904","last_updated":"2025-06-06T09:23:29Z","snapshot_observed_at":"2026-07-06T21:37:51.892023Z","submitted_at":"2025-06-06T09:23:29Z","title":"Proactive Assistant Dialogue Generation from Streaming Egocentric Videos","version":1},"cited_work":{"arxiv_id":"2506.05904","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05904","snapshot_observed_at":"2026-07-02T17:07:12.870465Z","title":"person wearing [clothing description]","venue":null,"work_id":"cf176532-0efd-44ab-8327-d512bb5d992c","year":2025},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2506.05904","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:7b7973d597fa346287a068793e423e5b5086760c39769c167efecefb5db3302a","observation_id":"479f8fc6-af92-475f-9443-3c600a42c8b6","resolution":{"observed_at":"2026-05-17T02:51:28.084265Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2f9836e9-c601-4fa2-b875-9ef09f2f3d5e","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:7bde4b4c492856ee34d660343cc582a615162d47dfd77a36e0d5640871fd306b","observation_id":"2233fc82-07cd-4f39-805a-7de1bf8145de","resolution":{"observed_at":"2026-05-17T02:51:28.813368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"{Question} Figure 19.Prompt for text-based reasoning","venue":null,"work_id":"953cb875-20f9-4890-84f0-83d805530dc5","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:95deb71922beb544e6f7be44089ee4c2f931808e545c3f4812c841e10176d9eb","observation_id":"b36064d9-93bc-4064-89a7-e12332db228a","resolution":{"observed_at":"2026-05-17T02:51:28.811247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"fc108665-9e42-4832-b8f5-d04c28ff6862","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:c471874b1fa42e5655dd00958f09c95fad544317ecf4d717cb897282b80f9f62","observation_id":"3c12305e-a042-4afd-89ef-dbadca736d4b","resolution":{"observed_at":"2026-05-17T02:51:28.803634Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b8da63c4-c52a-4500-9060-8363307deee4","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:55de88fcc9dd18b1da2eb6b6ebb4a84f8199be657e342dabdebb8db6db988de9","observation_id":"d6429ed6-6012-4559-b343-f649e48a5432","resolution":{"observed_at":"2026-05-17T02:51:28.827321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1482f9d5-6965-484a-85a7-2bf0eb0c78ff","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:8b1ff9778c20e1de283f944e8a14d63c5ea060308031f9ad25401fb6f568597d","observation_id":"76d77461-b7c4-4b26-a058-2a20accf8956","resolution":{"observed_at":"2026-05-17T02:51:28.820324Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"998e9288-0d32-4a9e-bf7a-19a9676549ab","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:98e3cf3cb50c57dcc3a688626da7ec47f5f74a9226b14c2eb7c038d4246d2ed5","observation_id":"f9544432-1873-4ec5-b882-d06d918cbf64","resolution":{"observed_at":"2026-05-17T02:51:28.806163Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"{Question} Figure 20.Prompt for gaze-based reasoning","venue":null,"work_id":"21465e47-674f-4dc6-ace1-5c375f74a1bb","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:f8c640fc8682afddfdd541cf24f66ed10c5eda18d0fd816524d04f186492cb3c","observation_id":"8b1f971c-76b1-444d-8b59-862a189dee5b","resolution":{"observed_at":"2026-05-17T02:51:28.815969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c147fe8a-bd59-4541-8891-e777e4d2eb74","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:e227133ac9943978bd86e7fdae6996b0a7fef3b91fa5c0d10e3dc2ad3895221e","observation_id":"787e565d-7b23-4be4-8d29-539ad518c458","resolution":{"observed_at":"2026-05-17T02:51:28.836323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"595b3b10-4d74-4d54-9407-109a568c824a","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:daf825691118a66f86cb1572d7126e592916d58533a023b58992b69687494a02","observation_id":"d04bca69-7d82-46ff-ac9a-40f8fb40703b","resolution":{"observed_at":"2026-05-17T02:51:28.831888Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"bfb3b994-a42e-4b1f-b9bc-1f943764444d","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:ec4d764d1d2a3178324c6e9fa71e1d65841ed1bfbc8b05d984071f2205e5db34","observation_id":"383da394-e35f-4974-a3d3-4aee9169c14c","resolution":{"observed_at":"2026-05-17T02:51:28.825221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7574a8bc-4b9b-49b1-aaaf-52c488cec04c","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:8ff01678243e8968fe3802dd45cd0c6e83884d027b6f590b38763a033eaffee5","observation_id":"015fcc90-56db-458c-8928-8d5b1581170b","resolution":{"observed_at":"2026-05-17T02:51:28.844720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7b7a0d52-d0d1-40f0-ad9b-5516439c8e19","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:658036ead5cd7948901cb723ff7a5bdcd6edc4de7e4143c3d227016674809076","observation_id":"c384c31e-19cd-4d6c-b65c-01fa6b9aa4e6","resolution":{"observed_at":"2026-05-17T02:51:28.880601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"324feb4d-d575-4cff-94d2-63a8af2ab31a","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:f74c5a53bf25a4ac164cfe762001a723b8d465bdcbf181e8555eb5308e75cc29","observation_id":"12ef5ffe-2a6b-44de-8368-4827eaeb3a0d","resolution":{"observed_at":"2026-05-17T02:51:28.883172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"{Question} Figure 21.Prompt for visual-based reasoning","venue":null,"work_id":"3995adb2-b63f-49c0-954c-e84385610b87","year":null},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:51575d884e06260382d84568a7681387d3d53d4d05c83ecaf10e73213c3c8ecb","observation_id":"0b6da574-99f2-451e-bb9f-6dab0d9685d1","resolution":{"observed_at":"2026-05-17T02:51:28.829792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-24T06:31:00.690269+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos"},"reference_resolution":{"displayed":50,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":14,"verified_fuzzy":35},"total_outbound_references":50},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-07-24T06:31:00.690269+00:00","source":"crossref"},{"observed_at":"2026-07-24T06:30:55.483554+00:00","source":"retraction_watch"}],"thesis":"As of 25 July 2026, this Paper Citation Record lists 50 of 50 outbound references and 3 inbound Pith citation observations for arXiv:2512.01707."}