{"as_of":"2026-08-07T05:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:74f77bb9c9ac99e1b008f8615ee6b53666dac8db080a79f770c9a129af47b43c","coverage":[{"denominator":57,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":57,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:34:44.512500Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.15569/citation-record","integrity":"/paper/2507.15569/integrity","json":"/paper/2507.15569/citation-record.json","paper":"/paper/2507.15569"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.354775Z","title":"Flamingo: a visual language model for few-shot learning.NeurIPS, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.354775Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:9fd9ece257f562cbdaae7b71fc75871a3ff13a31ca018a1d10b040142e05164a","observation_id":"f2aa759d-540a-4dd3-9b35-fcdbadc800c3","resolution":{"observed_at":"2026-08-06T15:34:44.354775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.358353Z","title":"Vivit: A video vision transformer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.358353Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:04eb98c1fb799e08acc43446150cb3dbf555b75fc9410d1edb6443f628a745c9","observation_id":"f12f1706-f49e-452d-bde1-3937039bfe13","resolution":{"observed_at":"2026-08-06T15:34:44.358353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.17274","last_updated":"2022-06-03T17:52:04Z","snapshot_observed_at":"2026-07-06T12:55:27.843060Z","submitted_at":"2022-03-31T17:59:30Z","title":"Exploring Visual Prompts for Adapting Large-Scale Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.17274","snapshot_observed_at":"2026-08-06T15:34:44.361412Z","title":"Exploring visual prompts for adapting large- scale models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.361412Z"},"links":{"cited_paper":"/paper/2203.17274","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:2402cf9ede7b4c01e901e031a49fe6828158eda22a5f9b9ef90404486cfa4119","observation_id":"a810cc3f-b520-4e8b-837c-051629199850","resolution":{"observed_at":"2026-08-06T15:34:44.361412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.182988Z","title":"Relevant intrinsic feature enhancement network for few-shot semantic segmentation","venue":null,"work_id":"0dc3ed80-a0d8-4e59-90c4-39f53536db9d","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.364929Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:c3f88ad2fc18d06cb2b09adf981e418908a6a01e3d09f8ce820e3c0442f8534f","observation_id":"9204a1de-db76-4350-988f-6541a262e604","resolution":{"observed_at":"2026-08-06T15:34:45.186177Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.173944Z","title":"Cores: Orchestrating the dance of reasoning and seg- mentation","venue":null,"work_id":"abb71ff2-73e1-4976-98b6-af5c2a387979","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.368268Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:ab79058d27f8bfa29ae54570f6040189dc769ae095d581afe4e263532dc1ec13","observation_id":"d902335b-31c9-4ce7-b95e-cd24b9314925","resolution":{"observed_at":"2026-08-06T15:34:45.177037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.371708Z","title":"Is space-time attention all you need for video understanding? In ICML, page 4, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.371708Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:c119b8cbd2fcf0cd23d1577983e0cb67095b723c65de882ccde1ca40d4abeb29","observation_id":"97a9d4ab-5612-42dc-9568-a85eafd9d064","resolution":{"observed_at":"2026-08-06T15:34:44.371708Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.158496Z","title":"Activitynet: A large-scale video benchmark for human activity understanding","venue":null,"work_id":"7a1e45b2-9cb6-494f-b4ae-7c6e030e9b05","year":2015},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.374741Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:074ec55d9c10511adaf492c8b2419c650b9fd5836ba17d6ff30d9efc0a7c9ce2","observation_id":"ed27e943-8da2-48a9-b119-bd9c8fb48b69","resolution":{"observed_at":"2026-08-06T15:34:45.161693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.148580Z","title":"Quo vadis, action recognition? a new model and the kinetics dataset","venue":null,"work_id":"5e0f7318-0226-4333-85b5-1a4762cbe712","year":2017},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.377884Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:f429561470cb501415b9bb5f0128fde89b843cfb97f5ca8e8f9d810e33691b7c","observation_id":"047a7701-6c38-4280-b268-a02799f653c6","resolution":{"observed_at":"2026-08-06T15:34:45.151912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.138965Z","title":"Deep temporal linear encoding networks","venue":null,"work_id":"07849ac5-5f1c-49ee-9a88-f914147cad22","year":2017},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.380704Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:c964acbb24757b2b41a3b6a90f0ed4a41cd7102b5273d15cc7b860860328f85c","observation_id":"31550c44-a080-4788-aebf-c472fb184c73","resolution":{"observed_at":"2026-08-06T15:34:45.142530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-06T15:34:44.383416Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.383416Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:6b24d021903e49e219a14955b33fcc1e96ae9298525378648f28e9cc7bc1f5fe","observation_id":"01a65432-7fe1-4d89-abb6-9290af5a52c7","resolution":{"observed_at":"2026-08-06T15:34:44.383416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.128691Z","title":"Convolutional two-stream network fusion for video action recognition","venue":null,"work_id":"c7d6b4bf-8d52-4775-93bf-407a71a862a8","year":1933},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.386251Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:1b72654de682fb5cca6ae604cd6178b5497f6c8dd997a484cfa7c86f9441e657","observation_id":"efca8d38-954e-475a-8066-dd34ff2a7bde","resolution":{"observed_at":"2026-08-06T15:34:45.131871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.388931Z","title":"Slowfast networks for video recognition","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.388931Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:1f2b6568c88b2319a172497f4d544879a624662df38966df36955459eb4ccfa5","observation_id":"06b216d4-bf76-415d-bfec-5b4821cd3b8e","resolution":{"observed_at":"2026-08-06T15:34:44.388931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.12980","last_updated":"2023-07-24T17:58:06Z","snapshot_observed_at":"2026-07-31T11:07:41.486136Z","submitted_at":"2023-07-24T17:58:06Z","title":"A Systematic Survey of Prompt Engineering on Vision-Language Foundation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.12980","snapshot_observed_at":"2026-08-06T15:34:44.391857Z","title":"A systematic survey of prompt engineer- ing on vision-language foundation models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.391857Z"},"links":{"cited_paper":"/paper/2307.12980","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:7f8e0143c94bbc27f8becef7e1e051028010dc3181a6cbed06db3eb69a879dda","observation_id":"fe50a836-005b-4f99-9b63-5f8df3e74e80","resolution":{"observed_at":"2026-08-06T15:34:44.391857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.112223Z","title":"Lita: Language instructed temporal-localization assistant","venue":null,"work_id":"7c1a8927-580f-4008-a8c3-a4ab4b08e2e0","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.394727Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:2093e4146ad1dbda3fc29848adade9e2672ac71bc781554ac19cd0d205755607","observation_id":"6c83cb25-e520-4df3-9bd8-c0d822c1f4a9","resolution":{"observed_at":"2026-08-06T15:34:45.115808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.397334Z","title":"Vi- sual prompt tuning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.397334Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:3d8ce498282304069c56683558bd2710256de40699aa9c8d82d89d40338fa579","observation_id":"8758f006-c3e4-4eaf-bf76-4b97f6f07d39","resolution":{"observed_at":"2026-08-06T15:34:44.397334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.096056Z","title":"Chat-univi: Unified visual representation em- powers large language models with image and video un- derstanding","venue":null,"work_id":"ba58cba5-38da-47ed-aef5-07d9d83e8905","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.400058Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:3ca50c48d670634eb6fc7641b9d6e5f88c22b5003e6fdbe12bd58060b486cd2d","observation_id":"76e16600-94a2-4da5-a431-b8651859cbbc","resolution":{"observed_at":"2026-08-06T15:34:45.099303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.402876Z","title":"Video-lavit: Unified video-language pre-training with decoupled visual-motional tokenization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.402876Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:60062acbcc366f5243b485e76f2bc4bf1447cad5dfb15aadd89275b1e7b88778","observation_id":"6b2d06ef-f871-479b-bf41-758ceadb3ae8","resolution":{"observed_at":"2026-08-06T15:34:44.402876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.083889Z","title":"Large-scale video classification with convolutional neural networks","venue":null,"work_id":"6e572fc5-40c9-41f0-93e1-54d607932090","year":2014},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.405446Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:c3fb8eba8a5de4beff87d1ced69a3e72d20f5193450f5019136d65e4642b2d0c","observation_id":"6efde5e5-e056-4173-8d90-cf61eefb4bd5","resolution":{"observed_at":"2026-08-06T15:34:45.089195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.073978Z","title":"Maple: Multi-modal prompt learning","venue":null,"work_id":"cc9e019f-4432-4d9b-892a-49d424bd645f","year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.407980Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:12bf3e8bf2c190670d6865c69f34433d186f8d9f5b060bc8b044904385a70147","observation_id":"886d5255-bf2c-432e-9ee2-bafee1ce1371","resolution":{"observed_at":"2026-08-06T15:34:45.077119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18406","last_updated":"2024-03-27T09:48:23Z","snapshot_observed_at":"2026-07-06T17:51:48.378864Z","submitted_at":"2024-03-27T09:48:23Z","title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.18406","snapshot_observed_at":"2026-08-06T15:34:44.410539Z","title":"An image grid can be worth a video: Zero- shot video question answering using a vlm","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.410539Z"},"links":{"cited_paper":"/paper/2403.18406","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:675a7de512d0599e538d197e5c60120e4404c9285463bf3f0a428238b34081da","observation_id":"86d87ae0-451a-468d-ba72-f19c6f13db1c","resolution":{"observed_at":"2026-08-06T15:34:44.410539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.00692","last_updated":"2024-05-01T05:10:13Z","snapshot_observed_at":"2026-08-06T07:43:56.889679Z","submitted_at":"2023-08-01T17:50:17Z","title":"LISA: Reasoning Segmentation via Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.00692","snapshot_observed_at":"2026-08-06T15:34:44.413798Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.413798Z"},"links":{"cited_paper":"/paper/2308.00692","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:045ad3e61281b4d5f8b8e38ebff7475516d0c658648962fb73e3735a0d05e01a","observation_id":"78686686-890a-4340-87c9-c72b035b5a40","resolution":{"observed_at":"2026-08-06T15:34:44.413798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.064498Z","title":"Deep local video feature for action recog- nition","venue":null,"work_id":"199c6a3f-e9e9-44b3-90dd-2bc7ed1e553e","year":2017},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.416913Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:7677e3708b55afbd66c2fce0b5523567c6f0cba32d7f9c5b54602ceebdcf24bc","observation_id":"e67cff02-1234-4df9-85db-504bce9b20d6","resolution":{"observed_at":"2026-08-06T15:34:45.067593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.12597","last_updated":"2023-06-15T07:57:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-01-30T00:56:51Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.12597","snapshot_observed_at":"2026-08-06T15:34:44.419656Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.419656Z"},"links":{"cited_paper":"/paper/2301.12597","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:696f31fd5f3f0a3d2abbb96d3c8cbcc34f8e69bc262901bb16caa3c40e3b300d","observation_id":"cbe3b580-50c7-4eb6-9bc5-41608f5d5610","resolution":{"observed_at":"2026-08-06T15:34:44.419656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-06T15:34:44.422636Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.422636Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:0656b4cbe8f8275cba7eeda4c465ed8c1ce6452843cb41a3dcc3b58e8c9023b1","observation_id":"1fe9950b-56f2-4ab2-b4a8-f37eaedb4cb7","resolution":{"observed_at":"2026-08-06T15:34:44.422636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.054273Z","title":"Mvbench: A comprehensive multi-modal video understand- ing benchmark","venue":null,"work_id":"0929c9d2-b2fe-4532-9709-9cbcb6a29602","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.425250Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:19fdedb2b8c1c1b6e924605005f3afc97ffc9febe73953433b09cd618bf6a564","observation_id":"3906ca21-74a7-436b-9fba-b566579c0520","resolution":{"observed_at":"2026-08-06T15:34:45.057513Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.044761Z","title":"Tgif: A new dataset and benchmark on animated gif description","venue":null,"work_id":"d958fa0b-bbd1-4167-913a-4aa41f7f1c1d","year":2016},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.427983Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:26400fee454420f388fd4aeafdf2962872b4f29e5d78eecdcc87ca50ae3ad991","observation_id":"4f78f31d-0889-4c27-ab81-189ccb23a05e","resolution":{"observed_at":"2026-08-06T15:34:45.048001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.034753Z","title":"Llama-vid: An image is worth 2 tokens in large language models","venue":null,"work_id":"810798a9-d271-4e46-b726-22b352104683","year":2025},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.430570Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:d4f88c7d31b46bce48de7c85842591f0fffcb7ac634a7e9670a61994acedca84","observation_id":"5d3997aa-1479-4679-a207-0baac91a52ba","resolution":{"observed_at":"2026-08-06T15:34:45.038073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-08-06T15:34:44.433210Z","title":"Video-llava: Learning united visual rep- resentation by alignment before projection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.433210Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:d91c913a0fbe5c9372dc0e72a13c6c7760a81f5869231dcc25d3ff40b0c91907","observation_id":"927376e1-c046-4e7c-9352-a6564ff8a61d","resolution":{"observed_at":"2026-08-06T15:34:44.433210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.436170Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.436170Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:cbcea87697534806b35cf858538bcf526736b53c5750acebb5d204af8f1e7baf","observation_id":"42037cba-cf27-4151-b814-ad705284f57f","resolution":{"observed_at":"2026-08-06T15:34:44.436170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.018683Z","title":"Pre-train, prompt, and predict: A systematic survey of prompting methods in nat- ural language processing","venue":null,"work_id":"9236db6b-9427-4c26-b154-d949392f8259","year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.438941Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:910c0db2a1336a22f4849fa792b9fee4e2bd49216b92b8336c9347dcc4a6d48f","observation_id":"f149dbf1-c940-4306-ba9b-dcb387911ee1","resolution":{"observed_at":"2026-08-06T15:34:45.022210Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.008077Z","title":"St-llm: Large language models are effective tem- poral learners","venue":null,"work_id":"235bfd06-55f5-41d4-a966-3f0c582277b3","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.441777Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:ae894c16f3aae2e5f5f821dbb0d3008b17ce97ab8a00a75b143b6332c266d558","observation_id":"f949c9ea-8c37-4d96-aab7-71637291e73b","resolution":{"observed_at":"2026-08-06T15:34:45.011156Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02327","last_updated":"2026-05-01T17:44:24Z","snapshot_observed_at":"2026-07-06T19:45:00.034364Z","submitted_at":"2024-11-04T17:50:36Z","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02327","snapshot_observed_at":"2026-08-06T15:34:44.444553Z","title":"Ppllava: Varied video se- quence understanding with prompt guidance","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.444553Z"},"links":{"cited_paper":"/paper/2411.02327","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:6f73bf79fd1f1723c4cff209f72a71db43232f0865e85297c9f19d22877dcde2","observation_id":"0a00f510-85b7-4864-9be1-0d8183088207","resolution":{"observed_at":"2026-08-06T15:34:44.444553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.998729Z","title":"Hybrid-level instruction injection for video token com- pression in multi-modal large language models","venue":null,"work_id":"1672e6fd-957b-4e2a-a522-cb55c3fd81c9","year":2025},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.447387Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:59b08fd59a6a91d922f1c46fd49dd3acc5b57278ee776dca4e9fe953a59b0491","observation_id":"25f1e70a-1b36-485c-9ad4-bcbe0a679412","resolution":{"observed_at":"2026-08-06T15:34:45.001920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08870","last_updated":"2025-03-03T17:58:12Z","snapshot_observed_at":"2026-07-06T17:01:40.744034Z","submitted_at":"2023-12-12T09:47:59Z","title":"Vista-LLaMA: Reducing Hallucination in Video Language Models via Equal Distance to Visual Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08870","snapshot_observed_at":"2026-08-06T15:34:44.450026Z","title":"Vista-llama: Reliable video narrator via equal distance to visual tokens","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.450026Z"},"links":{"cited_paper":"/paper/2312.08870","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:cc99e6a91bb47b899d8e79b91dc4742b25f62f719c0a8c8a14535bd6b7044c1e","observation_id":"22f7bdd3-7318-42a1-9806-3829e0bcfd82","resolution":{"observed_at":"2026-08-06T15:34:44.450026Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.989363Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":"d9218b6f-0ba5-46ee-a016-a1dbd887866a","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.453154Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:d4e4ca9d6aa0da9574cfa1fa3f4fd3849ca73cc1e6ab3fde3af0a89ebc4a71f9","observation_id":"b074dd2b-5d0c-4053-9b0b-bcacb9f59593","resolution":{"observed_at":"2026-08-06T15:34:44.992531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.979033Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":"ffa44f55-77e4-4413-b576-b8e1a8ef9d59","year":2021},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.456204Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:f7ec3679564507081a9325f5e5167fbc049d46b3811c88948cd3ec311104e83b","observation_id":"a26b246b-3b7a-4459-9e97-76a92169cba1","resolution":{"observed_at":"2026-08-06T15:34:44.982375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.968760Z","title":"Mul- titask vision-language prompt tuning","venue":null,"work_id":"b1182a12-8b9e-44e5-a19a-1811b096a8f1","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.458932Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:0c964e7b229b091cbdaa85e24fe16aa36bfb53239092df7a4924cff9ba81d968","observation_id":"0b4c4c9b-ef73-4f35-a271-363c8546f5fe","resolution":{"observed_at":"2026-08-06T15:34:44.971774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.958708Z","title":"What does clip know about a red circle? vi- sual prompt engineering for vlms","venue":null,"work_id":"01c37613-e36e-476a-b3ba-8763609c0a94","year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.461665Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:e879cb66e8e3d898b6548dbd6e4526c6363799523298c06a2fbdc504921b8f98","observation_id":"a007cafb-bd39-468d-850d-151dc2858a56","resolution":{"observed_at":"2026-08-06T15:34:44.962147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.949553Z","title":"Two-stream con- volutional networks for action recognition in videos","venue":null,"work_id":"965513ab-01ea-4bfd-b865-aa2587559fd3","year":2014},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.464208Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:b96af5662e15a100e2dfa65baa9e5f95790d15161e729b55416e622b57c46c04","observation_id":"0b9390b3-047e-40e2-885b-53296c9df236","resolution":{"observed_at":"2026-08-06T15:34:44.952584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.939561Z","title":"Moviechat: From dense token to sparse memory for long video understanding","venue":null,"work_id":"9e562606-8f18-46be-8233-1fd93cc030bc","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.466804Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:790ccb1848094ba83bc725f56b811d97604439a685b54c3ca25c2d560f01494e","observation_id":"8392e049-9274-45a9-a852-e6fb520459f0","resolution":{"observed_at":"2026-08-06T15:34:44.943100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.469414Z","title":"Ufo: A unified approach to fine-grained visual perception via open- ended language interface","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.469414Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:9ab2ae3a5ed9c051b604f72a2325906d87892540a0fb636fa408de9702035b10","observation_id":"5339d259-49d9-44ea-833b-97d2c51089c3","resolution":{"observed_at":"2026-08-06T15:34:44.469414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.929911Z","title":"Qwen2.5: A party of foundation models, 2024","venue":null,"work_id":"a942dd9f-5fe7-44d3-a182-40237d879652","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.472057Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:310143d44881f62196f4730fec77e0c93c63bcc2cbea24c072f9b5410348c94d","observation_id":"f116e752-67e6-49a2-b634-1073b17f5477","resolution":{"observed_at":"2026-08-06T15:34:44.932971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.474572Z","title":"Learning spatiotemporal features with 3d convolutional networks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.474572Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:92e3e776cbc9336e6c689ac42e9669e8a332662a994f950d407c6b9124116f57","observation_id":"90ac6e1f-5dd4-4f91-a9d4-2e4a9e11e204","resolution":{"observed_at":"2026-08-06T15:34:44.474572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.913882Z","title":"A closer look at spatiotemporal convolutions for action recognition","venue":null,"work_id":"fb7cb8d5-bda2-435c-8309-f202dde6e845","year":2018},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.477334Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:1ca35ef72d46a8dfba7b3c038bb22b2ec2e891cc7c626f0e7a07ac3ad82393b4","observation_id":"bf7cea43-2d1b-484f-8642-ab3d228a0e81","resolution":{"observed_at":"2026-08-06T15:34:44.917281Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.904562Z","title":"Action recogni- tion with trajectory-pooled deep-convolutional descriptors","venue":null,"work_id":"97e67244-c2fe-48a2-9466-78529e192d7e","year":2015},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.480128Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:3495fd562a2e77ab09d5762343d5958a21e02bef900f5b6b7c56947e6f045c32","observation_id":"562445aa-e6ce-488a-ac70-6da7fe1054ed","resolution":{"observed_at":"2026-08-06T15:34:44.907511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.894975Z","title":"Temporal segment net- works: Towards good practices for deep action recognition","venue":null,"work_id":"a374995b-5caf-4c34-ac11-678cac9b08d3","year":2016},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.482802Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:43113a9ae30b5c4e05b69bd7a38898e060dc9955ba9c68483247d8b0efac475f","observation_id":"1ac8883f-6190-43e9-9dd7-56414c619435","resolution":{"observed_at":"2026-08-06T15:34:44.898103Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.884657Z","title":"Deep learning for video classification and captioning","venue":null,"work_id":"ff5496f1-3599-460c-8aa4-a3ecfc274d77","year":2017},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.485388Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:3ef91f62aec45998e195719cb0b1bdd5bfaffaa563b9aff8abd51b0450fec247","observation_id":"2d08469b-7499-4da5-9a0c-291c211afaf6","resolution":{"observed_at":"2026-08-06T15:34:44.887994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.875151Z","title":"Msr-vtt: A large video description dataset for bridging video and language","venue":null,"work_id":"2f6eacee-c797-4468-9d45-cb2342897154","year":2016},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.488052Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:d7a9a53435935970578ce974c0310954c1cc79719df1d94cb8d1682315c78419","observation_id":"04648ed4-742e-4fc7-a6dc-0c4f81a1dd1e","resolution":{"observed_at":"2026-08-06T15:34:44.878270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16994","snapshot_observed_at":"2026-08-06T15:34:44.490784Z","title":"Pllava: Parameter-free llava extension from images to videos for video dense captioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.490784Z"},"links":{"cited_paper":"/paper/2404.16994","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:8e23512ea263a9b1cc5291ec84382a20cd830c8f4c688e6a5ea3bdfe55df5075","observation_id":"62968b3d-c61b-44b4-b4b3-72671e84ee31","resolution":{"observed_at":"2026-08-06T15:34:44.490784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.865283Z","title":"Beyond short snippets: Deep networks for video classification","venue":null,"work_id":"432275c7-c149-43c8-a619-23cf49d1797b","year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.493641Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:452901ec98b006f88dbb3da362641e20debf6e50190934a64fd0d0dac559d0f8","observation_id":"39de8a1a-26a0-4798-a23c-7dd48455c721","resolution":{"observed_at":"2026-08-06T15:34:44.868532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.07225","last_updated":"2022-10-13T17:50:24Z","snapshot_observed_at":"2026-07-06T14:04:55.099373Z","submitted_at":"2022-10-13T17:50:24Z","title":"Unified Vision and Language Prompt Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.07225","snapshot_observed_at":"2026-08-06T15:34:44.496621Z","title":"Unified vision and language prompt learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.496621Z"},"links":{"cited_paper":"/paper/2210.07225","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:dfe5e47eda6775e0f8f40463f7c0baafae7216f0d190ee1abfea7d1a4c8abfa1","observation_id":"624eca69-ece9-48ad-a3fe-6ee505763a9a","resolution":{"observed_at":"2026-08-06T15:34:44.496621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.855290Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"1145331e-ef4e-4cf5-87d8-68181dcdfe01","year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.499401Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:75dcba5764937db65aa9097a40febba4aad90b9f30650565a1a133369b280ed8","observation_id":"4e413645-f57d-425e-b8af-00f7ad24a065","resolution":{"observed_at":"2026-08-06T15:34:44.858675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.843464Z","title":"Real-time action recognition with enhanced motion vector cnns","venue":null,"work_id":"2dba500b-96dc-46fc-81ce-a3b5c8f73d3f","year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.502074Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:e8b4a45a89519dc4098d276346cef1506283514b9eb20745fd805adf05e1dca2","observation_id":"cadc5b69-4b8b-4b22-94c9-89979ec767a7","resolution":{"observed_at":"2026-08-06T15:34:44.848526Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-06T15:34:44.504615Z","title":"Video-llama: An instruction-tuned audio-visual language model for video un- derstanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.504615Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:3fe7f188b4f21c6b94e9adc6d25191c427fb0f0bc60253db2fde2c576dbf0efb","observation_id":"5bb3d3fb-30bd-42fc-8a91-94d8c352daf4","resolution":{"observed_at":"2026-08-06T15:34:44.504615Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.507428Z","title":"Conditional prompt learning for vision-language mod- els","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.507428Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:a8816a758fc6c48c40371555ad764ed00d7beb0a47b776ed0d20082f83091da7","observation_id":"2e549f92-9770-44e1-8e79-718f53c45890","resolution":{"observed_at":"2026-08-06T15:34:44.507428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.510044Z","title":"Learning to prompt for vision-language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.510044Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:e94a45adce9120a48f17d45c6f65f8dee431d38cb728644a653ebcb14d5338f1","observation_id":"3e308a2b-6870-4be2-8fd7-30105dfceff0","resolution":{"observed_at":"2026-08-06T15:34:44.510044Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-06T15:34:44.512500Z","title":"Minigpt-4: Enhancing vision- language understanding with advanced large language mod- els","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.512500Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:4caed25088b78c64ddbe00e99474c5e6614f5d591ca439bf1db02d68a619393b","observation_id":"35094d80-6c56-4e63-8554-ee5219e9d0da","resolution":{"observed_at":"2026-08-06T15:34:44.512500Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T15:26:30.186888Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding"},"reference_resolution":{"displayed":57,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":25,"verified_exact":0,"verified_fuzzy":32},"total_outbound_references":57},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 57 of 57 outbound references and 0 inbound Pith citation observations for arXiv:2507.15569."}