{"as_of":"2026-08-08T12:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e03f8b5dd5a618ace74cd7d3c4a794616af81b1b7069ca1d10e91d273651408d","coverage":[{"denominator":43,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:55:28.601657Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-22T05:41:39.396469Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-22T05:44:38.774124Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"cited_work":{"arxiv_id":"2506.01119","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01119","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MOOSE: Pay atten- tion to temporal dynamics for video understanding via optical flows.arXiv preprint arXiv:2506.01119","venue":null,"work_id":"73be5edd-6d6a-4339-bd54-db823a83eb7d","year":null},"citing_paper":{"arxiv_id":"2605.22823","last_updated":"2026-05-21T17:59:56Z","snapshot_observed_at":"2026-07-30T22:55:34.671958Z","submitted_at":"2026-05-21T17:59:56Z","title":"Which Way Did It Move? Diagnosing and Overcoming Directional Motion Blindness in Video-LLMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-22T05:41:39.396469Z"},"links":{"cited_paper":"/paper/2506.01119","citing_paper":"/paper/2605.22823"},"observation_digest":"sha256:648cb30f80b5f8d9e73288a24f312fc2a2baa9df476066378493271db97cfc8b","observation_id":"3d7e4a8f-ac48-45cc-8d2e-b9a52f16edca","resolution":{"observed_at":"2026-05-22T05:44:38.777481Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.01119/citation-record","integrity":"/paper/2506.01119/integrity","json":"/paper/2506.01119/citation-record.json","paper":"/paper/2506.01119"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.986741Z","title":"Gundavarapu, Liangzhe Yuan, Hao Zhou, Shen Yan, Jennifer J","venue":null,"work_id":"9a7410f2-ba18-4f49-bcdb-58cc6c0731a6","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.386030Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:82b00f58d3928d4baecd92f3ca4b56b1a2fb9c71abef2fc1d8b0f784530d7f5d","observation_id":"6dbce686-65b1-4798-8744-c1c3aa65eeae","resolution":{"observed_at":"2026-08-07T11:55:28.990505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1705.06950","last_updated":"2017-05-19T12:07:01Z","snapshot_observed_at":"2026-08-08T10:06:58.769155Z","submitted_at":"2017-05-19T12:07:01Z","title":"The Kinetics Human Action Video Dataset","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1705.06950","snapshot_observed_at":"2026-08-07T11:55:28.393424Z","title":"The kinetics human action video dataset.ArXiv, abs/1705.06950, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.393424Z"},"links":{"cited_paper":"/paper/1705.06950","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:c2bae491498fbf9f85a6fbffb7cde2918f292e96fd4a43ef47e4bc93583c1898","observation_id":"31296b17-4e1d-4577-ad09-ca09a7c82ab2","resolution":{"observed_at":"2026-08-07T11:55:28.393424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.978528Z","title":"The human visual system and its role in motion perception","venue":null,"work_id":"5cea4598-f940-457b-8299-e6ddee1e5861","year":2011},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.402383Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:68629bf3a4c1866cdf0829e5d8d02c9d175cb0f03b51304bad3b8f7734ad942b","observation_id":"b7d5e1b8-23fc-4be0-917f-49c2ba3768fc","resolution":{"observed_at":"2026-08-07T11:55:28.981697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2102.05095","last_updated":"2021-06-09T14:48:13Z","snapshot_observed_at":"2026-08-08T08:05:04.282318Z","submitted_at":"2021-02-09T19:49:33Z","title":"Is Space-Time Attention All You Need for Video Understanding?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2102.05095","snapshot_observed_at":"2026-08-07T11:55:28.412053Z","title":"Is space-time attention all you need for video understanding?CoRR, abs/2102.05095, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.412053Z"},"links":{"cited_paper":"/paper/2102.05095","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:02c9ac2164629c81518352473dfd059a14915800f2dcd5ac4dc98d4fd48f6458","observation_id":"5ab418a0-6bfc-4f97-990e-a8b928f6f186","resolution":{"observed_at":"2026-08-07T11:55:28.412053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.969278Z","title":"Vivit: A video vision transformer.2021 IEEE/CVF International Conference on Computer Vision (ICCV), pages 6816–6826, 2021","venue":null,"work_id":"4fba4926-ec02-4b52-a49e-048118521b1a","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.422695Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:a1a0421c4fbaa1418ec73e9e443d849875fdbdbc08db4a82c2fb5ac6e4c4913d","observation_id":"46b2f413-8943-40e6-8cb2-27593dab6d00","resolution":{"observed_at":"2026-08-07T11:55:28.973074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.960768Z","title":"De Gruyter, Berlin, Boston, 2016","venue":null,"work_id":"3792dcdf-79e5-4f93-bf59-9d163fb2dad4","year":2016},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.431253Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:9285643fd4852f123152d98d72e333e4cab1e56c2644602f9adbde7a6250c70f","observation_id":"dfb07648-2b4f-402e-9a76-45ec7b1d00b5","resolution":{"observed_at":"2026-08-07T11:55:28.963711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.951816Z","title":"Action recognition for surveillance applications using optic flow and svm","venue":null,"work_id":"0eae86cd-8a18-49fd-ad32-eea5c36004b5","year":2007},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.438232Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:6acee03587610a002870d395c7e9398c41ced037636494582bc6d4eee83b65b6","observation_id":"a7113b0f-7476-4a94-a79a-75a6b8fdb899","resolution":{"observed_at":"2026-08-07T11:55:28.955123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.942710Z","title":"Conv3d-based video violence detection network using optical flow and rgb data.Sensors, 24(2):317, 2024","venue":null,"work_id":"4fda9b6c-6a5b-4a7c-97fa-0fc937983ec8","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.450359Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:21e9585a4d4759ef7f07974b131a90e4baab43b04f63f05c75754e7b862f7a79","observation_id":"d4797cdf-ccde-4b0e-ba25-09f695a2a5c8","resolution":{"observed_at":"2026-08-07T11:55:28.946060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.932893Z","title":"A multi-modal egocentric activity recognition approach towards video domain generalization.Sensors, 24(8):2491, 2024","venue":null,"work_id":"cba87012-5d7a-4a20-b473-0176e5ec6631","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.458569Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:ac722f441eb5767cbd8dcf62b825ea01a677938004e4ea575d3a4a54b346dada","observation_id":"247d453b-637f-4fcc-8dcb-6eb04aab0fad","resolution":{"observed_at":"2026-08-07T11:55:28.936770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.924032Z","title":"Nayak, and Shrikanth S","venue":null,"work_id":"2e4f177f-4d5a-4835-8725-e00b155ec341","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.465645Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:e8d7170b62e8ba4490e5a8414f2b1edaa71dad89c4fe4072200e45918f78465c","observation_id":"30e22078-01a4-4dd2-aa6a-52eb4cde4e96","resolution":{"observed_at":"2026-08-07T11:55:28.927244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.915121Z","title":"Kosloski, Siddhi Patel, Zeke A","venue":null,"work_id":"6bf57b0e-d105-4fc1-abf3-de9831b6eec7","year":2025},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.473893Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:3f791bdee780eecce67d53597993d0ba0a7e63e25f64aad1883ce9c74a216390","observation_id":"48e6121a-cc97-4115-b8b8-8d0048e1dc14","resolution":{"observed_at":"2026-08-07T11:55:28.918941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.905630Z","title":"Childplay: A new benchmark for understanding children’s gaze behaviour","venue":null,"work_id":"8e681077-a1ad-4b86-9873-e864016cc1b2","year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.481578Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:9e410b1da4676e95017a41370f58184db38f153fc1a472f5a87c3adddfd4b795","observation_id":"3f00602f-9e2d-4f81-8fbc-2185311414f8","resolution":{"observed_at":"2026-08-07T11:55:28.909200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.896471Z","title":"Barner, and Roghayeh Leila Barmaki","venue":null,"work_id":"18c2c5f1-eac7-42f1-887f-ec4d098670d6","year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.492403Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:70290734a85f0b68388fafe29e0ea24f88b1afefc9031e3112abd10a4f304c16","observation_id":"bb5d62d2-7286-4735-90b4-4d41ed1d31e2","resolution":{"observed_at":"2026-08-07T11:55:28.900092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.887216Z","title":"Reversible vision transformers","venue":null,"work_id":"1fa13a9a-c0aa-4883-b230-832580461d18","year":2022},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.501974Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:955b5f55debcf47f1f63340d1451fa761df595f6f9194e701638a32c0821fd80","observation_id":"dcc67fb1-afe9-4b82-9f4c-4b6c84bf1209","resolution":{"observed_at":"2026-08-07T11:55:28.890570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.878278Z","title":"Multiscale vision transformers","venue":null,"work_id":"7814bb41-e189-49be-9b01-ed7cc0ab8636","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.505343Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:f50b90b5f7118a78418c2471d5c47caed0c05dfdc261a236a4485a03b7a854b7","observation_id":"94e91b00-0004-4897-9bb8-5b2b282ea082","resolution":{"observed_at":"2026-08-07T11:55:28.881706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.869490Z","title":"X3d: Expanding architectures for efficient video recognition","venue":null,"work_id":"502ff843-c569-4859-9143-bd2c984a514f","year":2020},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.508483Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:9e537d708aa8f63121b998a065eadb6894c7eb83782c402557b0ef50dbb04f43","observation_id":"fe739488-63d3-493d-9d5e-3d0e52ce9576","resolution":{"observed_at":"2026-08-07T11:55:28.872697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.860690Z","title":"A large-scale study on unsupervised spatiotemporal representation learning","venue":null,"work_id":"a2a4f910-13ca-480c-8e9e-3d02556c342c","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.511429Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:0a9d201c5b1ffca1ec483fdc8b9bb7654c1a20aed4f238e9142ff95bda293384","observation_id":"6d16135b-8ed2-40d3-8f03-1628219ff299","resolution":{"observed_at":"2026-08-07T11:55:28.863939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.851335Z","title":"Spatio- temporal collaborative module for efficient action recognition.IEEE Transactions on Image Processing, 31:7279–7291, 2022","venue":null,"work_id":"f9a0208a-2258-4a5d-a7cb-44405a0656f3","year":2022},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.514882Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:e5d3738e55e03bc5b909d19ec13d4b4f7fd6e114850e5ec07e8f4a34e6d5b940","observation_id":"06197e95-cd25-434e-b15c-0c73abe5192e","resolution":{"observed_at":"2026-08-07T11:55:28.854792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.841988Z","title":"Quo vadis, action recognition? a new model and the kinetics dataset","venue":null,"work_id":"89901d67-0ad5-491c-8c9a-63e46bd9bf40","year":2017},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.518323Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:491d6d7abdbed6f661447767a6c1cc848a934c1087435fa779b52b8272aa2db8","observation_id":"ea230c41-9731-4304-9776-7504f3072ba9","resolution":{"observed_at":"2026-08-07T11:55:28.845441Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.832997Z","title":"Batch transformer: Look for attention in batch, 2024","venue":null,"work_id":"1ef233a5-5eb8-43cd-a424-28231e157c26","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.521658Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:72e8596c9018a9e373e3a1cdbf82f25662f81db83310019a5bec80172bf5791b","observation_id":"469ec2df-3213-41ce-8db6-06f55bcefd40","resolution":{"observed_at":"2026-08-07T11:55:28.836220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.823990Z","title":"Videomae: masked autoencoders are data-efficient learners for self-supervised video pre-training","venue":null,"work_id":"5e0c3110-037d-43d7-952c-dec33b1f3b52","year":2022},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.525238Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:2b3838bd46dc4311ca8a2e09389ab9e0927fb041a49e7664329c029886aa3642","observation_id":"3bd54dc5-556c-4af1-bb20-de4ee4845e82","resolution":{"observed_at":"2026-08-07T11:55:28.827404Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.815145Z","title":"Videomae v2: Scaling video masked autoencoders with dual masking","venue":null,"work_id":"779620ec-f7ff-4a4e-9fbb-4d3191b0a81b","year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.528471Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:44eebc98a53529f4fc94da50182aae7e3a490cf045e262bbf4cb4c47cb2255a7","observation_id":"2867165f-1f18-46b0-b603-81f640b3bc1d","resolution":{"observed_at":"2026-08-07T11:55:28.818394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.805770Z","title":"Video swin transformer","venue":null,"work_id":"4b294883-c7db-4a55-8b48-47993c2dd580","year":2022},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.531356Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:726b6bf8e8ad50436dc0527c31bbc31acce2dc4198830579da414a06974a2110","observation_id":"01f74cea-7896-458c-a9d7-63a76d406320","resolution":{"observed_at":"2026-08-07T11:55:28.809366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.534382Z","title":"Pyslowfast","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.534382Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:e88e5f6cf98761cd518e755f892c70d84d5c8fdb16eb27ecb9ae9999d6fa2a63","observation_id":"a4733bde-7530-4e67-893c-184e944c8ecb","resolution":{"observed_at":"2026-08-07T11:55:28.534382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.791312Z","title":"Jampani, Andreas Geiger, and Michael J","venue":null,"work_id":"85a33d1a-2b84-4678-b43d-da201c45c8d8","year":2017},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.537466Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:08625efc421db6860463e9f3a3dfde17f3e9270958fb3d2221a9a4178879da19","observation_id":"b58e5f88-f92c-4727-b35e-a2b9a5c8914b","resolution":{"observed_at":"2026-08-07T11:55:28.794641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.781746Z","title":"Henriques","venue":null,"work_id":"0d615f8e-4ee2-4d9d-add1-fd93a20cfdad","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.540883Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:ab0d6571dcbe76d615f0546e4e46f7c56c35edb6581e44b13d53eef6246587c3","observation_id":"d35cb0e2-e7ea-4b65-bac0-d7ecd28ba1b0","resolution":{"observed_at":"2026-08-07T11:55:28.784880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.772936Z","title":"Memflow: Optical flow estimation and prediction with memory, 2024","venue":null,"work_id":"9b4c5983-d639-4869-979a-5ffa70f8026e","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.543940Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:7790b86fdd3bfc902c9882d458598be8442bc37fc11f272568bca7d9ccaa3f2d","observation_id":"f6fd2541-73a6-4144-957a-d50c8ae420b2","resolution":{"observed_at":"2026-08-07T11:55:28.776070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.04451","last_updated":"2020-02-18T16:01:18Z","snapshot_observed_at":"2026-07-06T08:50:12.690900Z","submitted_at":"2020-01-13T18:38:28Z","title":"Reformer: The Efficient Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.04451","snapshot_observed_at":"2026-08-07T11:55:28.547233Z","title":"Reformer: The efficient transformer","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.547233Z"},"links":{"cited_paper":"/paper/2001.04451","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:96aca2be583e4dc5df65004b813f8527715decfe2f7b96e36351055b6fe146ca","observation_id":"fe05929b-d2e5-4667-8e2a-a6d2f7bc8fa8","resolution":{"observed_at":"2026-08-07T11:55:28.547233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00752","last_updated":"2024-05-31T17:55:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-01T18:01:34Z","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00752","snapshot_observed_at":"2026-08-07T11:55:28.551969Z","title":"Mamba: Linear-time sequence modeling with selective state spaces","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.551969Z"},"links":{"cited_paper":"/paper/2312.00752","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:96ad63b5ae1efdb80676d3b4b83777adc172bf31d50f978d501c4b10b1480f6c","observation_id":"fdd2c949-005b-4f33-9cbf-a3c67a4305b0","resolution":{"observed_at":"2026-08-07T11:55:28.551969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.763188Z","title":"Raft: Recurrent all-pairs field transforms for optical flow","venue":null,"work_id":"f797f9e2-5915-4692-aede-db48d81eaf74","year":2020},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.555565Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:e1c27a692069df52e7fc284c3e1fab4495f7279a6310471c69ae874962987749","observation_id":"97448051-ee0f-460c-b750-ae3501c43f0c","resolution":{"observed_at":"2026-08-07T11:55:28.766560Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.754299Z","title":null,"venue":null,"work_id":"7fa4b4f4-9a30-4368-804f-79b29416e5c5","year":2012},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.559252Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:5074f59974550a9476bbe7277a8a52dab7a2307560d365b2717fb71f90505098","observation_id":"39b7184f-ab40-4135-bc7d-b3a461216f70","resolution":{"observed_at":"2026-08-07T11:55:28.757518Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.744398Z","title":"Vision transformers need registers, 2023","venue":null,"work_id":"e56c2bef-42c0-471d-bc05-5c67680f8095","year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.562573Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:88ab15f387ea7f0937b0fa82f6de116998840531d3ef27c7c47d47c891efe7b8","observation_id":"f48e9839-fec3-489c-9e26-7f1322ae2cd6","resolution":{"observed_at":"2026-08-07T11:55:28.748304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.565743Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.565743Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:ed76d7c22394748e04f07cb1dcfad6ca312c3e1785b9e6d0e2e836f58b471912","observation_id":"48a7837c-a8b6-4d22-8428-c966ac4a442c","resolution":{"observed_at":"2026-08-07T11:55:28.565743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.730864Z","title":"something something","venue":null,"work_id":"c8d785ca-7080-4533-bd3d-5cbdd420cc61","year":2017},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.569311Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:322783083090513fb7b2d9ac7a6df3c54ddc6343bf9323f5b817e6532d2e46df","observation_id":"6f8d3a28-c38d-4d96-9cf0-917e8ebab219","resolution":{"observed_at":"2026-08-07T11:55:28.733996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.720851Z","title":"Haa500: Human-centric atomic action dataset with curated videos.2021 IEEE/CVF International Conference on Computer Vision (ICCV), pages 13445–13454, 2020","venue":null,"work_id":"83e4f5e5-d086-4165-a211-fc848959c79b","year":2021},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.572573Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:00ab0429414716826f34cd08770c21d9a0dadb20e68b97b70448b1ba93828739","observation_id":"3ff158d3-b578-4830-8336-25fd38b8a102","resolution":{"observed_at":"2026-08-07T11:55:28.724868Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1212.0402","last_updated":"2012-12-03T14:45:31Z","snapshot_observed_at":"2026-07-06T03:01:10.229407Z","submitted_at":"2012-12-03T14:45:31Z","title":"UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1212.0402","snapshot_observed_at":"2026-08-07T11:55:28.576141Z","title":"Ucf101: A dataset of 101 human actions classes from videos in the wild.ArXiv, abs/1212.0402, 2012","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.576141Z"},"links":{"cited_paper":"/paper/1212.0402","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:30d138c3bf366f58e8daf8018ee3615b541ef18c7e1e252ea5162064156059f7","observation_id":"a183b072-d9e6-47f7-b98a-ecb2db274ecf","resolution":{"observed_at":"2026-08-07T11:55:28.576141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-07T11:55:28.579305Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding.arXiv preprint arXiv:2306.02858, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.579305Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:7c3bb878f51e66db58c2b1978d3d7f32640a0fac49f6c4aa5301646782d3f4b9","observation_id":"49b04534-6599-45ae-8e45-a02de3856930","resolution":{"observed_at":"2026-08-07T11:55:28.579305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-07T11:55:28.582939Z","title":"Videollama 2: Advancing spatial- temporal modeling and audio understanding in video-llms.arXiv preprint arXiv:2406.07476, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.582939Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:c5c99e8620ed9e4da7f8435452a45b34d8391781bbfe0883816f569a956cd470","observation_id":"11b6460c-27c4-44d8-a83a-db3a97378f3b","resolution":{"observed_at":"2026-08-07T11:55:28.582939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-07T11:55:28.586764Z","title":"Boqiang Zhang","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.586764Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:680cc108ee321ec55c12c39479434047c1f90d5983962637de2cfd4b5472c88d","observation_id":"6a77fe62-b43d-4c1f-a132-461310eeec73","resolution":{"observed_at":"2026-08-07T11:55:28.586764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15841","last_updated":"2024-09-15T05:00:18Z","snapshot_observed_at":"2026-08-08T03:13:28.000968Z","submitted_at":"2024-07-22T17:58:04Z","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15841","snapshot_observed_at":"2026-08-07T11:55:28.590775Z","title":"Xu Mingze","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.590775Z"},"links":{"cited_paper":"/paper/2407.15841","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:1dcbd6ab8c9cae09dd99079ccdf61462b8ff59bb37d9baf17f8ef0a63a823423","observation_id":"2ca677bd-85df-4f23-bf31-c6e4b77df4bb","resolution":{"observed_at":"2026-08-07T11:55:28.590775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-08-07T11:55:28.593941Z","title":"Video-llava: Learning united visual representation by alignment before projection.arXiv preprint arXiv:2311.10122, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.593941Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:98655be13ac8498645380d9dfe3c8b7319fe6509fdc5d4d01fddebb26eac46fd","observation_id":"38aef75a-669e-4a59-9487-ed095bf3ccb4","resolution":{"observed_at":"2026-08-07T11:55:28.593941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01852","last_updated":"2024-01-22T03:11:15Z","snapshot_observed_at":"2026-08-07T05:10:33.059352Z","submitted_at":"2023-10-03T07:33:27Z","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01852","snapshot_observed_at":"2026-08-07T11:55:28.598004Z","title":"Languagebind: Extending video-language pretraining to n-modality by language-based semantic alignment.arXiv preprint arXiv:2310.01852, 2023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.598004Z"},"links":{"cited_paper":"/paper/2310.01852","citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:b63bc982fb8a5f057f12253e96c8fe674982a9e6b1610e9028837cbc05ac4e94","observation_id":"f8d45bca-9cee-4951-a3ea-6d6c44723162","resolution":{"observed_at":"2026-08-07T11:55:28.598004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:55:28.709577Z","title":"running” or “jumping","venue":null,"work_id":"4a617660-3970-4da8-92e4-c18f85620539","year":2024},"citing_paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T11:55:28.601657Z"},"links":{"citing_paper":"/paper/2506.01119"},"observation_digest":"sha256:25fe592682beb46328f43f09171edf6541769a7a13949144a63667c9a7638f7f","observation_id":"ba947274-ad79-41ca-93f7-98a7c4503750","resolution":{"observed_at":"2026-08-07T11:55:28.713614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.01119","last_updated":"2025-06-01T18:53:27Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T11:48:15.825538Z","submitted_at":"2025-06-01T18:53:27Z","title":"MOOSE: Pay Attention to Temporal Dynamics for Video Understanding via Optical Flows"},"reference_resolution":{"displayed":43,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":0,"verified_fuzzy":29},"total_outbound_references":43},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 43 of 43 outbound references and 1 inbound Pith citation observation for arXiv:2506.01119."}