{"as_of":"2026-08-11T02:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0ede55c6a0f10a4eb603cf10e1c5af7d84d6964f1398b0c2a577c87529e48336","coverage":[{"denominator":84,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":84,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:37:24.588694Z","state":"measured"},{"denominator":86,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":86,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T13:27:56.359929Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T03:06:18.765685Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.24158","snapshot_observed_at":"2026-08-04T13:27:56.359929Z","title":"Threading keyframe with narratives: Mllms as strong long video comprehenders.arXiv preprint arXiv:2505.24158,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.00705","last_updated":"2026-06-26T15:22:20Z","snapshot_observed_at":"2026-08-10T21:44:39.000633Z","submitted_at":"2025-10-01T09:20:51Z","title":"Training-free Uncertainty Guidance for Complex Visual Tasks with MLLMs","version":3},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-04T13:27:56.359929Z"},"links":{"cited_paper":"/paper/2505.24158","citing_paper":"/paper/2510.00705"},"observation_digest":"sha256:f94c9294fae815268c0d66399befc69630b6efa9cd3c689bbc5d270b32a26520","observation_id":"76dbb3a0-84e8-434c-9c6f-48b3029920af","resolution":{"observed_at":"2026-08-04T13:27:56.359929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"cited_work":{"arxiv_id":"2505.24158","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.24158","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b67f3704-b24e-466c-96b8-410615681160","year":null},"citing_paper":{"arxiv_id":"2605.09223","last_updated":"2026-07-30T19:00:53Z","snapshot_observed_at":"2026-08-10T22:41:15.175214Z","submitted_at":"2026-05-09T23:47:46Z","title":"CREST: Curvature-Regulated Event-Centric Sampling for Efficient Long-Video Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-12T03:06:09.753634Z"},"links":{"cited_paper":"/paper/2505.24158","citing_paper":"/paper/2605.09223"},"observation_digest":"sha256:a03cd13dd9fae843979b28e8ee33d96be1c76ad867634d9c314cffa2d5b6bd60","observation_id":"767c7c7e-b77e-4aeb-8c50-e2dfe77869ac","resolution":{"observed_at":"2026-05-12T03:06:18.772638Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.24158/citation-record","integrity":"/paper/2505.24158/integrity","json":"/paper/2505.24158/citation-record.json","paper":"/paper/2505.24158"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T12:37:16.067831Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.067831Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:bc7a212400f3656bedc91e10e64de42b5b7a3c6c5e951fbac5ffdaf53a784874","observation_id":"0b0f98ab-5ecc-4387-99bc-7583abcaa79a","resolution":{"observed_at":"2026-08-07T12:37:16.067831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:16.151840Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.151840Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:72200ca29e674d4d5925ddd1b13030816f1310db856779871493874ebfb497e4","observation_id":"2a25c717-9def-4308-8ff1-c6049a818e78","resolution":{"observed_at":"2026-08-07T12:37:16.151840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:16.236696Z","title":"Vqa: Visual question answering","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.236696Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:db589bbb243f5d337e99e18bd4e94b8bbfce6775ff5589ca0f0161100c09d7dd","observation_id":"68155e93-d9bf-49c6-b5da-6cad2496cd05","resolution":{"observed_at":"2026-08-07T12:37:16.236696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-07T12:37:16.323580Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.323580Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:fdc00bcb7779af88bab029dfccfeca1f9a018f1f2ec20aa831c7e98ac9485c82","observation_id":"5bc3e1f1-f3a4-4f92-a82e-04cd9341d2bc","resolution":{"observed_at":"2026-08-07T12:37:16.323580Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:16.421314Z","title":"Solving mixed-integer quadratic programming problems with ibm-cplex: a progress report","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.421314Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:8e70b757149ac4f103694d14049af0b04335b58afdb15de13fe1fac7bb0fb431","observation_id":"e7c4e58b-8675-472d-a7ed-ac3f59df9c45","resolution":{"observed_at":"2026-08-07T12:37:16.421314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-07T12:37:16.506733Z","title":"On the opportunities and risks of foundation models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.506733Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:d452f67523f613e891d509865f388554de4f43deea3e4d9667c59b4a631f5e76","observation_id":"02f8b0ea-6d17-4e2d-9aba-f79e9be26d52","resolution":{"observed_at":"2026-08-07T12:37:16.506733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:16.620624Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.620624Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:db0ff7aaa0e4cc1ec8f268558b8b583f2c04647013bd0d6d08de5f9e5df58cdd","observation_id":"8b1037ce-82e4-47c4-a89d-2e18f8978df0","resolution":{"observed_at":"2026-08-07T12:37:16.620624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:34.412934Z","title":"Hourvideo: 1-hour video-language understanding","venue":null,"work_id":"fe8e8eaf-216c-476b-bd20-d2412037a1da","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.681958Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:57de3ab700d3f92bf5a3dffaf8ec8734fea860da59ca56beb0e79d6d4f34b6fc","observation_id":"896ac7d1-7c57-4717-a50c-28a592b161b9","resolution":{"observed_at":"2026-08-07T12:37:34.498288Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:34.269421Z","title":"Sharegpt4video: Improving video understanding and generation with better captions","venue":null,"work_id":"ed43e894-b84d-4cb6-bf04-4beeb0c04da3","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.776741Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:0c11d5b76113540fc6b8ae387e2fa519b6bfbb535aefa6281f79df63c7b7e297","observation_id":"c3568a3a-b413-41e3-88e1-e0f954ffb8ae","resolution":{"observed_at":"2026-08-07T12:37:34.329297Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-05T14:57:53.592979Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.10188","snapshot_observed_at":"2026-08-07T12:37:16.941581Z","title":"Longvila: Scaling long-context visual language models for long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.941581Z"},"links":{"cited_paper":"/paper/2408.10188","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:c990471c6345d24255f286cd6231007d7a77bdc2c3535602183e0f57556a21c9","observation_id":"27e4ed29-6118-42b5-b76a-559978453cd2","resolution":{"observed_at":"2026-08-07T12:37:16.941581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:34.019636Z","title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","venue":null,"work_id":"6b7b65c2-4028-477d-b1d4-701b1752ba2d","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.185352Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:0d337dd9ebb21904a7888070d213ccfeafa51b7639df4756bfa676511960ded3","observation_id":"e3b44931-b510-4d07-8757-d9f2446ea122","resolution":{"observed_at":"2026-08-07T12:37:34.139239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-07T12:37:17.344219Z","title":"Videollama 2: Advancing spatial-temporal modeling and audio understanding in video-llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.344219Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:f2804f8ff0e3e1697a6571ffe5aa29b4b5786a876ef48236e972466bcecd4863","observation_id":"762e141e-19cd-44e0-bc51-3a40d50f8291","resolution":{"observed_at":"2026-08-07T12:37:17.344219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:33.841826Z","title":"Patch n’pack: Navit, a vision transformer for any aspect ratio and resolution","venue":null,"work_id":"5cef66cc-8b6d-4eec-a4a2-95bcea1d1357","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.455893Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:b32fd17f832637f6bbea868c7fdf34b6e87eb8210421b55b1ae4c4adb7932d71","observation_id":"d484031f-54be-415b-940f-bb6fb78340da","resolution":{"observed_at":"2026-08-07T12:37:33.935943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:33.677794Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":"52f61c47-fac0-4f1a-8b18-3dbec1d02387","year":2020},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.620244Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:9bac117c912b04839e64b6ca06929bf7e1385beffdb4484660ce43152407b0e0","observation_id":"376adc86-d204-4089-8f9e-cecd44d61290","resolution":{"observed_at":"2026-08-07T12:37:33.752601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:33.520806Z","title":"Vlmevalkit: An open-source toolkit for evaluating large multi- modality models","venue":null,"work_id":"9d929e34-e0d1-4fe5-be3f-04702e669ff8","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.746820Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:6fa56e20effacdb1fef0c233f13a4f72e5b3564ec7936d03568fb881910f7d38","observation_id":"2676de86-cf63-4741-8c40-e4de5cae84b2","resolution":{"observed_at":"2026-08-07T12:37:33.587582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:33.336676Z","title":"Slowfast networks for video recognition","venue":null,"work_id":"c62e5db3-5ef8-4aa7-b3ec-b5e2995772de","year":2019},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.902941Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:0db3c02250ecd9c7a4ea3081e85bbf3eaa8e7493ad9c03a15ca5c3ab02253b6e","observation_id":"7f21be05-08c7-49b4-9be9-780019e38bd2","resolution":{"observed_at":"2026-08-07T12:37:33.431518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:33.064148Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":"5ebd45e1-dc25-4149-836d-79b629b261f4","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.023177Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:8070cc4e1d3401e2fc1b310ab16eb62e511f83fa079445fc63cdec5a3a56934e","observation_id":"809b72ba-d3a4-41c5-8700-937a966fc819","resolution":{"observed_at":"2026-08-07T12:37:33.155237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-10T16:40:37.411115Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T12:37:18.121807Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.121807Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:f9b0a726f21a8e645003797b9273cc76d3f7ca70d082f41a14ab4a4ce04bc13d","observation_id":"47dd4bc0-84f5-4db0-b975-1d6a06d65372","resolution":{"observed_at":"2026-08-07T12:37:18.121807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:32.876865Z","title":"M-llm based video frame selection for efficient video understanding","venue":null,"work_id":"156e9f8c-38be-4ee8-b1af-b148f01e6fe2","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.286872Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:c7d0dcb8558b6e6b340f1a48d920412b6c8dabac6ce5183549644d577b6efccb","observation_id":"2381eef4-f583-4f76-94e7-c73d273512bf","resolution":{"observed_at":"2026-08-07T12:37:32.959706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:32.701466Z","title":"Chat-univi: Unified visual representation empowers large language models with image and video understanding","venue":null,"work_id":"250a8329-1659-42e8-bb66-095a8882e5a4","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.409318Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:7944b279653feebc65ad438ecd0e96b1a923ade69459610f2553a76df4dcbd64","observation_id":"3750f7a5-fdfc-4ed9-aafa-7789b8c5546b","resolution":{"observed_at":"2026-08-07T12:37:32.786989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:32.516316Z","title":"Language repository for long video understanding","venue":null,"work_id":"667c4a48-df45-404d-8160-96ebe5204d8b","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.555490Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:2d7a815da04b64a920531d30c1223d6df57a83415c54058d597d213cdd517477","observation_id":"fa668bc0-cee4-43e7-b805-46e964487e50","resolution":{"observed_at":"2026-08-07T12:37:32.625784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:32.329054Z","title":"An image grid can be worth a video: Zero-shot video question answering using a vlm","venue":null,"work_id":"3a02dfd1-42b5-466e-991f-532ce31977bf","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.682198Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:6cba7385d816f8adbe71292a710d7cd7aac4875007a26300265cfd397365f3a7","observation_id":"a8d39fa3-b1cf-4f27-b2a2-2587ffad92a7","resolution":{"observed_at":"2026-08-07T12:37:32.441482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:32.078456Z","title":"Lmms-eval: Accelerating the development of large multimoal models, March 2024","venue":null,"work_id":"7920c36b-12a4-4027-8985-9dc714bcc882","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.798283Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:9c7a35a523d15b2755f9acc9f07691e3437722854b74a319a35c8a1bfffd1636","observation_id":"4241b8f0-2af6-46c9-9a51-1bf2b12aaab9","resolution":{"observed_at":"2026-08-07T12:37:32.178385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-07T12:37:18.912546Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.912546Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:29c1821e5c237e3f19d8c85bac36ad03f502cd44292128b8e351dec79b3a114f","observation_id":"291baed7-2303-4cf2-977a-841f9a3c3bf3","resolution":{"observed_at":"2026-08-07T12:37:18.912546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:31.857768Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":"be444341-5ef6-4a05-ae8e-ab7522b5f8eb","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.979088Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:78c66ef951bc840f321da3e964fe26e9cdf1810319ecd01de0d1cc252e84bc20","observation_id":"5d203498-19e8-4011-bb17-7995405bec65","resolution":{"observed_at":"2026-08-07T12:37:31.953262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T12:37:19.138328Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.138328Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:b8728ecc471c52cc0193950b6d56fa7760a292308901431c3aa429af7a75e78b","observation_id":"60b66c9c-95ff-4c99-aa4f-806c7132fb93","resolution":{"observed_at":"2026-08-07T12:37:19.138328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:31.707621Z","title":"Llama-vid: An image is worth 2 tokens in large language models","venue":null,"work_id":"5c185666-bbbe-4b66-9fb2-8590fea490e5","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.222339Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:8e7be9a3978b2be7de34216dae4f9a9ab147e9f7a3310860ccd1bdd1fdfa39be","observation_id":"01bfaaa7-1cec-48f4-8706-3dc35bce83d9","resolution":{"observed_at":"2026-08-07T12:37:31.759563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:31.497843Z","title":"Video-llava: Learning united visual representation by alignment before projection","venue":null,"work_id":"0ef7b386-628b-425c-b049-b9a51695eb70","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.291562Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:021b94c5e9fc13c69547caa6732f18372354cdcfa395bdd5bbc18805e11819d3","observation_id":"5d5a836a-f36a-4515-98b8-f0983000e696","resolution":{"observed_at":"2026-08-07T12:37:31.614389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:31.334767Z","title":"Vila: On pre- training for visual language models","venue":null,"work_id":"e1848844-5ed4-4aba-bf93-5fee15f54072","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.353735Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:a67ef7e13ae422f16f018ea498a78d3b17774f09980cf5d548a8c5c34ae501a0","observation_id":"638b84e3-ccde-4770-8b05-ab1b71fd4652","resolution":{"observed_at":"2026-08-07T12:37:31.426866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:19.424939Z","title":"Llava-next: Improved reasoning, ocr, and world knowledge, January 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.424939Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:0207e2e539933456a53003438717c00985dd270f6d7119a59952bd14243c6c95","observation_id":"c00adb3d-44ed-426f-a8e7-1acd59c1535a","resolution":{"observed_at":"2026-08-07T12:37:19.424939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:31.061164Z","title":"Visual instruction tuning.Advances in neural information processing systems (NeurIPS), 36:34892–34916, 2023","venue":null,"work_id":"17d1035b-45dd-4c27-a01c-960ab99edec0","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.493444Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:550cef8aea20cec8d9148d7928bf61a19aa985f38ca33680c01b99f920b0bece","observation_id":"e16f8ea2-1456-4418-80ed-0254c24804d7","resolution":{"observed_at":"2026-08-07T12:37:31.189946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:30.822419Z","title":"Timecraft: Navigate weakly-supervised temporal grounded video question answering via bi-directional reasoning","venue":null,"work_id":"6430c44f-08e0-4e2d-9cbe-3d5444d8f9cb","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.566948Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:0550c365a0fb34fd5b63c8a8857f58f33f1595bfd16f1ef45b9c767ae61e4371","observation_id":"6954443f-d472-471a-ba74-4e194b2f62cf","resolution":{"observed_at":"2026-08-07T12:37:30.918499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:30.551095Z","title":"Lost in the middle: How language models use long contexts","venue":null,"work_id":"6c82aa6c-70a4-41be-bffa-a69607b92a06","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.644740Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:88654964406d5053a92be5d7fac78672b690d817810dc65f9f7e2354119a119b","observation_id":"b426fb8e-9c4a-44d3-9d07-e7078607e366","resolution":{"observed_at":"2026-08-07T12:37:30.659000Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:30.243332Z","title":"St-llm: Large language models are effective temporal learners","venue":null,"work_id":"6bee1e60-3096-4163-90f0-a900a342d4e8","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.721077Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:ed588af93adc96ef7e68135ef250e63f4e486a39cc1cafebbdfcee42a0d952cf","observation_id":"2f4cc9f1-282b-41ed-9b30-625553421422","resolution":{"observed_at":"2026-08-07T12:37:30.370543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:30.052866Z","title":"Bolt: Boost large vision-language model without training for long-form video understanding","venue":null,"work_id":"78ac377d-70ef-47a7-a91a-72aa6d619c28","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.791097Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:b9a3c9a6a49dfff43acd293c8154be06807e1506ebb110111219cdc95bf89ac8","observation_id":"81f55e11-6d72-4da8-a94c-73e27f7c50ec","resolution":{"observed_at":"2026-08-07T12:37:30.122315Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.895831Z","title":"Drvideo: Document retrieval based long video understanding","venue":null,"work_id":"7641425e-56db-4c8b-b561-b0293420b35d","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.859288Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:f02a33765fbb61895eeeb31a4cd1368c1701b516f2db5daffe07ca0976e4056f","observation_id":"5342c0cd-7585-46c2-a601-c2740e0a62c8","resolution":{"observed_at":"2026-08-07T12:37:29.987173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-07T12:37:19.932480Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.932480Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:a10d14bdb2a659db31516f738963408bbf78644c6be8340078ad28e00c7c605f","observation_id":"d4d72409-db44-44cc-b222-457dff76e831","resolution":{"observed_at":"2026-08-07T12:37:19.932480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.695081Z","title":"Egoschema: A diagnostic benchmark for very long-form video language understanding","venue":null,"work_id":"c88238c2-80c7-4846-9256-9932780a9ab7","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.004759Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:806bdbe5b9029b8dffbed3b77afeef8ee5b70b5fe10b0335355fb00f9ad34b0c","observation_id":"ffbf24f6-cd17-462d-8bb0-3c82bb673924","resolution":{"observed_at":"2026-08-07T12:37:29.773619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.568181Z","title":"Morevqa: Exploring modular reasoning models for video question answering","venue":null,"work_id":"7fde5000-9252-4316-ac61-6b80e4b3b49b","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.072028Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:0454be3cefd88575e509b9e616bdac66a090526ee9d95d220a1ff312cbac07f0","observation_id":"ae471a22-25b4-492d-a176-bca78cc195e3","resolution":{"observed_at":"2026-08-07T12:37:29.610912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.450488Z","title":"Branch-and-bound algorithms: A survey of recent advances in searching, branching, and pruning","venue":null,"work_id":"e07157e8-b9c6-463b-b25d-c4069160834a","year":2016},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.145930Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:78162a95fd67f75b1d946ec7605ee2b472cf571dd590647e0807c5a74fff8fce","observation_id":"0abc113c-9916-4985-825b-dc805945947b","resolution":{"observed_at":"2026-08-07T12:37:29.526632Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.333001Z","title":"Chatgpt: Optimizing language models for dialogue, 2023","venue":null,"work_id":"87c4749c-cdd7-4bbe-b611-d31ccb3afcee","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.236870Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:7c04380de3e75691f5cb871a6574910c6366616ac3152ff859d434100a864922","observation_id":"d1c22e62-5148-4f3c-bd01-6c185a24dec6","resolution":{"observed_at":"2026-08-07T12:37:29.389856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.182548Z","title":"Too many frames, not all useful: Efficient strategies for long-form video qa","venue":null,"work_id":"7691b435-ef1e-4b2c-b9a9-ab727b7df31a","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.321049Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:2679b4432216459e64fa0b8493b986127269d84bd45b3e4a3545cbc133963303","observation_id":"02da2d3a-496d-4709-bcc1-4a52e6737496","resolution":{"observed_at":"2026-08-07T12:37:29.235070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.019542Z","title":"Momentor: Advancing video large language model with fine-grained temporal reasoning","venue":null,"work_id":"d6d73ae9-bfa7-442d-befe-f5508cd53c08","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.447778Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:83bec18614b75f6c98a564d089594c4b3d6654adfa350f9e61236d245ae5a686","observation_id":"deb2fcb6-9589-4fa4-b275-b8777c51689d","resolution":{"observed_at":"2026-08-07T12:37:29.090406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.908519Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"7c53ddfe-fae8-40ed-a208-73f0f0f76692","year":2021},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.542794Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:1ac6b396d04bec1c00a36a56cdfbe7542d4e33fc4a713c9910c60c618a514723","observation_id":"9347cb9c-82d3-4a9e-bf0e-53f2dcf2f7f3","resolution":{"observed_at":"2026-08-07T12:37:28.954666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:20.635358Z","title":"The knapsack problem: a survey","venue":null,"work_id":null,"year":1975},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.635358Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:b4e495156775ece995bc5b6b1acb0f376acb78d78e31ab20a720352a9c14193f","observation_id":"3bc64dc1-212e-4d68-9cf6-0e9ba1c2a852","resolution":{"observed_at":"2026-08-07T12:37:20.635358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17434","snapshot_observed_at":"2026-08-07T12:37:20.727692Z","title":"Longvu: Spatiotemporal adaptive compression for long video-language understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.727692Z"},"links":{"cited_paper":"/paper/2410.17434","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:3b55c37e03358b64d5a8e0946a1d4a46357b4c059905172d56637f74c14e3a16","observation_id":"312e0497-9293-483e-9372-348a673c1466","resolution":{"observed_at":"2026-08-07T12:37:20.727692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14485","last_updated":"2024-12-10T12:45:31Z","snapshot_observed_at":"2026-08-10T13:02:21.561135Z","submitted_at":"2024-09-22T15:13:31Z","title":"Video-XL: Extra-Long Vision Language Model for Hour-Scale Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.14485","snapshot_observed_at":"2026-08-07T12:37:20.806587Z","title":"Video-xl: Extra-long vision language model for hour-scale video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.806587Z"},"links":{"cited_paper":"/paper/2409.14485","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:d59bdc39a51abbb004d392286157b2f022271150a1b9335d0811df2f6081b712","observation_id":"f19b5cbc-5429-425a-8010-6d84e9d4ed61","resolution":{"observed_at":"2026-08-07T12:37:20.806587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.755969Z","title":"Two-stream convolutional networks for action recognition in videos","venue":null,"work_id":"74aab278-409e-421b-ae17-7b205cf21b64","year":2014},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.873682Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:475dda42ed540146bd8e7fd0f65e4a18c57e3b6cc357b13b16de4540c307e09c","observation_id":"fbd4598c-ac5d-4e0c-aeba-4bd4cc9806cf","resolution":{"observed_at":"2026-08-07T12:37:28.843270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.636677Z","title":"Moviechat: From dense token to sparse memory for long video understanding","venue":null,"work_id":"8530be3e-3582-432a-aad0-f04e6396d29c","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.984827Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:833a04e6af167ba8ddea5255d257fc270af4f89c6d215f7dd491a4e80958e327","observation_id":"d462dcf3-d607-4bff-9bed-dd3901c0c289","resolution":{"observed_at":"2026-08-07T12:37:28.689205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:21.085807Z","title":"Mdp3: A training-free approach for list-wise frame selection in video-llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.085807Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:122452002492cbd27f416461865e7b53b8ab512851e46a4835cdc6e499cd60e8","observation_id":"7b755651-2206-48b2-9d89-86b81dba6344","resolution":{"observed_at":"2026-08-07T12:37:21.085807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.454405Z","title":"Adaptive keyframe sampling for long video understanding","venue":null,"work_id":"4df9b1cf-8b45-4483-8d66-f457e90677d7","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.159282Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:6fc6ecb641d652b1d35dcc50da807050993fd819c86b45607c871668187dd62f","observation_id":"542d090a-bde5-476e-93c8-631aaba47c51","resolution":{"observed_at":"2026-08-07T12:37:28.544301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-07T12:37:21.246351Z","title":"Gemma: Open models based on gemini research and technology","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.246351Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:855619d6cf08c7abcf44944aee2b6501fc5650a981bd988b7436000225b2ee37","observation_id":"fb5be959-6296-4046-9e8b-69de40a997a2","resolution":{"observed_at":"2026-08-07T12:37:21.246351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.331323Z","title":"Cambrian-1: A fully open, vision- centric exploration of multimodal llms","venue":null,"work_id":"849743c3-f752-406f-95c5-955031387901","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.360353Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:b3e32b89873dd2ce2123037d1d81d8709df8c213b9c82ce884eddc9435d15450","observation_id":"0bb7053e-8653-4f74-a8ed-f522d94005a6","resolution":{"observed_at":"2026-08-07T12:37:28.389747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T12:37:21.439956Z","title":"Llama 2: Open foundation and fine-tuned chat models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.439956Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:a5ae14c4a3ab5cbc6f1202ee6ec3eebc512f9580c41ed64f27ed13655306c06e","observation_id":"a09a5b92-a3c3-46d7-9c28-29122d677e3c","resolution":{"observed_at":"2026-08-07T12:37:21.439956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.133001Z","title":"Attention is all you need","venue":null,"work_id":"6a423123-6e96-44db-a556-c0e3ecb670e5","year":2017},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.559745Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:075ac7e2fb1ee224593af671036a0a7c7f333c17730baa3d9f6a412f53c4b945","observation_id":"46bcfb17-ae86-4fea-9e96-e026cddb3bfc","resolution":{"observed_at":"2026-08-07T12:37:28.221238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.013766Z","title":"Show and tell: A neural image caption generator","venue":null,"work_id":"9eba579a-55b0-4803-84df-77af6f259ab6","year":2015},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.692309Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:cdff538efc1ea850dba6bc0dfd046c3a1423c4f0c8ac3130e17534d12839d378","observation_id":"a9736535-42eb-4e5f-8817-6603449fa2e5","resolution":{"observed_at":"2026-08-07T12:37:28.070943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:27.820049Z","title":"Efficient large language models: A survey.Transactions on Machine Learning Research (TMLR), 2024","venue":null,"work_id":"45e0ece0-346a-4e4b-8036-d31040798588","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.768128Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:d4a97faba701ca3e373d5dadb1283a0f510b66fb6d9429fabc898b841ff1aaae","observation_id":"cf88f0d5-6501-4dae-b17b-c9deec7298aa","resolution":{"observed_at":"2026-08-07T12:37:27.896903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:27.627446Z","title":"Weakly supervised gaussian contrastive grounding with large multimodal models for video question answering","venue":null,"work_id":"87438bdc-bd46-49d1-bfae-205cd545fc41","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.846512Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:05a02b163d5ae95b7b203d1e4b52d26517afc370a18f2009f4934c15022938d0","observation_id":"8e98f985-ca8e-4fc2-a759-67f3c756a9be","resolution":{"observed_at":"2026-08-07T12:37:27.715659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-07T12:37:21.942911Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.942911Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:30d0065b6685eb5e74c491e3645d828cbb08e2889491b40b88aa16b48b9e3a19","observation_id":"88731b3a-d85e-4308-ab9d-076a50bdcc2a","resolution":{"observed_at":"2026-08-07T12:37:21.942911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.20504","last_updated":"2025-03-24T02:17:34Z","snapshot_observed_at":"2026-08-10T23:17:19.220083Z","submitted_at":"2024-12-29T15:42:24Z","title":"ReTaKe: Reducing Temporal and Knowledge Redundancy for Long Video Understanding","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.20504","snapshot_observed_at":"2026-08-07T12:37:22.035335Z","title":"Retake: Reducing temporal and knowledge redundancy for long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.035335Z"},"links":{"cited_paper":"/paper/2412.20504","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:7e5b8fdf354ec3c83acb9e46bf7f0cba9f41456b5f09632480369d7e1793e679","observation_id":"26926a99-6fe9-4763-ae56-44f168a64c7d","resolution":{"observed_at":"2026-08-07T12:37:22.035335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:27.433202Z","title":"Videoagent: Long-form video under- standing with large language model as agent","venue":null,"work_id":"d10c4782-c6f8-496e-9482-709c1412e708","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.139175Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:674646176400a7257562054d20a4885cce12fb1fc94f21d0aae11184a467288e","observation_id":"ae110e84-5baf-4ba8-94c9-2236d33e803d","resolution":{"observed_at":"2026-08-07T12:37:27.552815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:27.274722Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":"81974ff7-ff60-478b-9d94-0d74cfe74a72","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.236365Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:3a09ebd82cbf0c4e00418cb84fe707e09b3c5215c26ca03a26b47981fd62be2d","observation_id":"eb9742ac-5b10-4d64-a3ed-357f9bcc1d79","resolution":{"observed_at":"2026-08-07T12:37:27.342292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:27.081580Z","title":"Dibs: Enhancing dense video captioning with unlabeled videos via pseudo boundary enrichment and online refinement","venue":null,"work_id":"60b64a48-d0be-4dc4-9984-61a9ba12fb1d","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.338769Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:da819b3d656e290e3c2151108779534f358943b28b0eb024b9341fb9a4160cba","observation_id":"7cd90885-7735-4b77-8764-6ba6759e772e","resolution":{"observed_at":"2026-08-07T12:37:27.171688Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.889861Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","venue":null,"work_id":"4205c048-0e6f-499e-8e06-4e5ec5204bb5","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.450389Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:2b87a0c0035c89a59070445acca15e14f427e00f5f92a95f645c2115f2d4ed92","observation_id":"061e4745-6e27-4634-a3ca-076def2997d1","resolution":{"observed_at":"2026-08-07T12:37:26.974968Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.744859Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":"258426d4-95a0-44d6-9929-2f937df0317f","year":2021},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.581586Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:9331d1dded3c294414c7036f9aecf89e28fe032b2e95ff7001c10787bcc31b04","observation_id":"7998ee8f-53e2-406a-944f-7d9d6e446f8b","resolution":{"observed_at":"2026-08-07T12:37:26.807231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.579139Z","title":"Can i trust your answer? visually grounded video question answering","venue":null,"work_id":"bfcfb6d7-6666-4ee0-b933-71eee565a387","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.691030Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:5dcec90a2207059854aea8bbd02a3d78c100724a1605cc4441f524d30150b30c","observation_id":"519ac6c8-ad9d-415d-bb49-44f01783acce","resolution":{"observed_at":"2026-08-07T12:37:26.664010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.404911Z","title":"Effective long-context scaling of foundation models","venue":null,"work_id":"843f68be-94ac-4ab6-9cd1-aa3305a4ccd8","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.786503Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:79ef0d6404fed63642b5c90bd1e89bfddb72520b9aea8fca0cd40b19e27b9445","observation_id":"c3b083f2-ee51-4cb1-a99d-b3b5c2ef56c0","resolution":{"observed_at":"2026-08-07T12:37:26.500839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16994","snapshot_observed_at":"2026-08-07T12:37:22.900514Z","title":"Pllava: Parameter-free llava extension from images to videos for video dense captioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.900514Z"},"links":{"cited_paper":"/paper/2404.16994","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:f1af24b461e68b7ef5bc95e0a4cdefd16cf368b96b5894968edcad6b815aabad","observation_id":"a990e067-485f-4cf9-852f-ee77af800c35","resolution":{"observed_at":"2026-08-07T12:37:22.900514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15841","last_updated":"2024-09-15T05:00:18Z","snapshot_observed_at":"2026-08-10T23:43:54.740334Z","submitted_at":"2024-07-22T17:58:04Z","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15841","snapshot_observed_at":"2026-08-07T12:37:23.032653Z","title":"Slowfast-llava: A strong training-free baseline for video large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.032653Z"},"links":{"cited_paper":"/paper/2407.15841","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:b692821d9fc896f7a4893c9aacc90bb1419dff344b9f6d43f12f0b1510935986","observation_id":"cba55f33-db7f-408d-99fa-78317f5baaca","resolution":{"observed_at":"2026-08-07T12:37:23.032653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.269083Z","title":"Zero-shot video question answering via frozen bidirectional language models","venue":null,"work_id":"f01c814e-969e-4c68-9d93-936d20f0eba2","year":2022},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.130555Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:ed99c588b5bf80e7249f8e3bcc979de957c5ea20115a63659e9d6ed7bbf9e1c0","observation_id":"cb3d9dd3-06aa-45e8-bdb6-fdb75c4eda2e","resolution":{"observed_at":"2026-08-07T12:37:26.332155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.080203Z","title":"Vid2seq: Large-scale pretraining of a visual language model for dense video captioning","venue":null,"work_id":"d889a329-efd6-40b1-be78-beaf9bf792f9","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.230304Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:7fbb9c67d785a96614db189d7f5e1a8fb77516aef6ee8a0a6bfb340233a59d2c","observation_id":"51652350-edb5-4a59-8448-5a2ab454b129","resolution":{"observed_at":"2026-08-07T12:37:26.186839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:25.990715Z","title":"Dense connector for mllms","venue":null,"work_id":"69b8bb1b-0fa2-4eae-be92-9d7bf57831ba","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.342782Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:445f75eca23af334e38ac81a54ddba019e9b42ef7e1ba7fdef2241f897c9a202","observation_id":"9298836e-13e0-461f-ab6c-1abfb7ea459e","resolution":{"observed_at":"2026-08-07T12:37:26.019452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09146","last_updated":"2025-09-02T09:52:40Z","snapshot_observed_at":"2026-08-07T17:10:48.940585Z","submitted_at":"2025-03-12T08:16:39Z","title":"Generative Frame Sampler for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2503.09146","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.09146","snapshot_observed_at":"2026-08-07T12:37:24.797285Z","title":"Generative Frame Sampler for Long Video Understanding","venue":"cs.CV","work_id":"89c807d2-4027-4c52-8920-70732296db0b","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.440876Z"},"links":{"cited_paper":"/paper/2503.09146","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:9a72a760c7ff416e3ce6b10ce691ea9a3a855de2e9ae2245fd9a28d9f948a5b4","observation_id":"0ef5db10-d2dc-4306-898f-c0a74ce8d362","resolution":{"observed_at":"2026-08-07T12:37:24.889986Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.01800","snapshot_observed_at":"2026-08-07T12:37:23.584263Z","title":"Minicpm-v: A gpt-4v level mllm on your phone","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.584263Z"},"links":{"cited_paper":"/paper/2408.01800","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:1b0ddf5a5326b9692e26e3096cdf18663cdf432941b2aef8fede822fcb9d21e5","observation_id":"e9d53f51-3628-4ccd-8161-bafbb69f4d9a","resolution":{"observed_at":"2026-08-07T12:37:23.584263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:25.851332Z","title":"Self-chained image-language model for video localization and question answering","venue":null,"work_id":"a7a78d6b-5164-43c5-99d3-c2a0fbd268d6","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.714914Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:d18788f45aae869ffcf07eac2ebf70931437e064a4b3c76c24b3720390ac7645","observation_id":"1388fada-bbd9-4c81-83ce-e5b474ed503f","resolution":{"observed_at":"2026-08-07T12:37:25.913672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:25.680294Z","title":"Frame-voyager: Learning to query frames for video large language models","venue":null,"work_id":"37fd5d8f-7494-4b46-8402-35f8ec78655c","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.826563Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:17af7c0c4a85014529f7cf325ee83cde630031aeb77eae9e40b5d17e0d5d40c6","observation_id":"a2d295cf-2c3e-4d3b-864f-226d08d397da","resolution":{"observed_at":"2026-08-07T12:37:25.779824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:25.476452Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"203685c9-afbf-46c9-9579-21399d548e8f","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.882899Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:a15b1d36e22e76c5e549d6b6aafe1fd14bb79a827d622e1f69c24312129cdf48","observation_id":"49e98b79-f970-452f-9e47-f1468db5e4da","resolution":{"observed_at":"2026-08-07T12:37:25.554077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:25.330415Z","title":"A simple llm framework for long-range video question-answering","venue":null,"work_id":"ae041f53-3641-4972-9788-87f6524d1117","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.986712Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:06bd8e0cb2cfcc27cbec0bf1248f895a1a2f0687a504ce072c8517492d72703e","observation_id":"842a74e2-9b22-453c-b188-65da3f3cc9e1","resolution":{"observed_at":"2026-08-07T12:37:25.380735Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-08-07T09:52:45.942315Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-07T12:37:24.093163Z","title":"Long context transfer from language to vision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.093163Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:ff89157454276407d60c772b6693961c101a22e512f4180a86d6b41154d75c79","observation_id":"d0f16f33-24b5-4af9-ad4e-4b8e07355c6d","resolution":{"observed_at":"2026-08-07T12:37:24.093163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:24.190869Z","title":"Llava-next: A strong zero-shot video understanding model, April 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.190869Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:ce9d77f9bf75e3f7757183e5322afddb289772dc677a66597fff237f9a9f1e7d","observation_id":"79b9db04-5fc5-48b4-aadf-697adeb8f07e","resolution":{"observed_at":"2026-08-07T12:37:24.190869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T12:37:24.258273Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.258273Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:ab38168341b90a98a9fc072aeff418578827bcb17cf3621ffee92eec755bdeb8","observation_id":"b30ebedc-b77a-4463-b3a6-ef8c7a773f6d","resolution":{"observed_at":"2026-08-07T12:37:24.258273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-07T12:37:24.363162Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.363162Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:6556874c1c6064c99f1dea41b1ccf4c611635403934159ad05a28d41c3f498c5","observation_id":"b028f069-8365-4ff5-903f-0d9928d4f313","resolution":{"observed_at":"2026-08-07T12:37:24.363162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-07T12:37:24.461768Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.461768Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:416db74fdb76d7cbce8a4660ea9a1e88a1e5cfc10e6567331f0954e1e4ae9536","observation_id":"4793def1-2e8a-4082-a776-2a2c5df55d59","resolution":{"observed_at":"2026-08-07T12:37:24.461768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10360","last_updated":"2024-12-13T18:53:24Z","snapshot_observed_at":"2026-08-06T08:59:28.938249Z","submitted_at":"2024-12-13T18:53:24Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10360","snapshot_observed_at":"2026-08-07T12:37:24.588694Z","title":"A. High heels","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.588694Z"},"links":{"cited_paper":"/paper/2412.10360","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:302bcf5656cb390f4185d1c2c155353f20c4ae329acb6eac99c4782178840947","observation_id":"447bbede-b0be-4fbc-9feb-22827ba5a7bd","resolution":{"observed_at":"2026-08-07T12:37:24.588694Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T08:26:14.415807Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders"},"reference_resolution":{"displayed":84,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":30,"verified_exact":1,"verified_fuzzy":52},"total_outbound_references":84},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 84 of 84 outbound references and 2 inbound Pith citation observations for arXiv:2505.24158."}