{"as_of":"2026-08-14T23:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:017ab886363d2e88af351b8db52436819b4b308069ce868766316c2c46774e0f","coverage":[{"denominator":65,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":65,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T18:04:29.255261Z","state":"measured"},{"denominator":65,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":65,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.11653/citation-record","integrity":"/paper/2501.11653/integrity","json":"/paper/2501.11653/citation-record.json","paper":"/paper/2501.11653"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.855271Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":"cbe1a22c-7dec-458a-b836-e21086d05a6d","year":null},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.058653Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:21ea3a02fe604f629c1af1daf9c9a2a98e00d47f2e73af04053ca5e5df7df34c","observation_id":"aac75950-44cf-4b22-a26d-c667546e7da8","resolution":{"observed_at":"2026-08-10T18:04:29.859186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.843320Z","title":"Learning human- human interactions in images from weak textual supervision","venue":null,"work_id":"52946e2c-d65e-49a9-a590-35883aada515","year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.062198Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:7b489113ec587b00578b75aa0b18179adcce7809f4767a93796dcc86f1b6e925","observation_id":"95eef835-4dac-40c8-ad97-aca0a8909734","resolution":{"observed_at":"2026-08-10T18:04:29.847754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.831755Z","title":"Zero-shot learning via visual abstraction","venue":null,"work_id":"a894fde6-566e-4f27-b85c-3ac3383e0611","year":2014},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.065276Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:089c3bf8f03cb01666e2eb395613f379e9943c364f4e07a82de0545aba9947e4","observation_id":"2e3a36a5-775a-497f-a6de-cf45e09c9500","resolution":{"observed_at":"2026-08-10T18:04:29.835676Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-10T18:04:29.068523Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.068523Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:8ab609d3267888aa122685fd4a8cad41b49878d57be909dc022fccbbffa27221","observation_id":"631fb98c-5f95-4ebc-a278-141f848b5558","resolution":{"observed_at":"2026-08-10T18:04:29.068523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-10T18:04:29.072486Z","title":"On the opportunities and risks of foundation models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.072486Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:4db688517f6834849cef0b94c65b9ee719c6a4d6e14108fe01a582e5acd582e7","observation_id":"01422b37-2dda-430c-9b9b-e0508b190b53","resolution":{"observed_at":"2026-08-10T18:04:29.072486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.821166Z","title":"End-to- end object detection with transformers","venue":null,"work_id":"106901a7-6048-4587-a002-3ba0682ff23d","year":2020},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.075724Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:cd6733e376d97169e09bd3ac265d522d8ab0e416bab753fb78d35504742474e0","observation_id":"f63c90eb-eab9-483f-bdf6-d8980f63670e","resolution":{"observed_at":"2026-08-10T18:04:29.824996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.811426Z","title":"Quo vadis, action recognition? a new model and the kinetics dataset","venue":null,"work_id":"cefb9479-d662-429a-b27b-d26919a93f23","year":2017},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.078754Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:b49f218b696fc4f9ca18b9848db17915c7b3c50d492927e6e4f0971b301353ae","observation_id":"bfd6cce4-ae8b-4d02-93b2-e246e6da1e8b","resolution":{"observed_at":"2026-08-10T18:04:29.814499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.801530Z","title":"A comprehensive survey of scene graphs: Generation and application","venue":null,"work_id":"bfc661a6-7191-464b-8ef2-4b429af66410","year":2021},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.081798Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:611a91e24e9cace62bb87d7ac32ab91a5dc4fae03146e5a1089f87c04cae6a63","observation_id":"57c6dedb-17bf-41ee-b038-dbacff56251d","resolution":{"observed_at":"2026-08-10T18:04:29.804672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.084523Z","title":"Hico: A benchmark for recognizing human-object interactions in images","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.084523Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:eb631dd023c68b17ed607d022824888759036c771bb6a0f8f3ae8fa3827b4e3e","observation_id":"8b25b575-0c75-45d3-af43-bd3c774dfaf2","resolution":{"observed_at":"2026-08-10T18:04:29.084523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.786055Z","title":"Learning to detect human-object interactions","venue":null,"work_id":"483a319c-6702-4dc3-a81d-63a407084251","year":2018},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.087364Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:cc7e2ccbc82f063e41e76c9ac3bf8d6922989cc06a2849d1e888848afc105c15","observation_id":"eb4750d0-d381-41ab-a303-8b1fe6e0c5a8","resolution":{"observed_at":"2026-08-10T18:04:29.789523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.776758Z","title":"ViStruct: Visual structural knowledge extrac- tion via curriculum guided code-vision representation","venue":null,"work_id":"dd5622da-ce5e-4382-aada-d1cd5709f08c","year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.090004Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:63e23330a47d9d685c0e977cf9e3e5627b256b2b69bff74618c6917d91f7d862","observation_id":"8ecd477b-6191-482f-b123-7c1ebd7e0fde","resolution":{"observed_at":"2026-08-10T18:04:29.779720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.767894Z","title":"Collab- orative transformers for grounded situation recognition","venue":null,"work_id":"b60cee3c-7bc9-4ac4-8d79-c6fcabf42153","year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.093120Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:7f4db9eead3825a0c2249aa438991b3a63e019dcde66d2974532e54657dc18c9","observation_id":"703cb924-8b57-4947-b998-0bb4fd614917","resolution":{"observed_at":"2026-08-10T18:04:29.770916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.096460Z","title":"Imagenet: A large-scale hierarchical image database","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.096460Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:18242b13403d16054b6d1395533381dca3ccf5e2ac5675970ffe06c0e418e664","observation_id":"c3e88180-8913-47f4-b6cc-6dce707d9aef","resolution":{"observed_at":"2026-08-10T18:04:29.096460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-10T18:04:29.099971Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.099971Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:215741d256ad3a67d06843333ca820778c564de6e16702a16711ccf9456bb184","observation_id":"fc8823df-6244-410f-8731-49f1861b40b0","resolution":{"observed_at":"2026-08-10T18:04:29.099971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.752548Z","title":"Background to framenet","venue":null,"work_id":"ab5a00a1-a41a-4e63-bce2-1ff27579cde9","year":2003},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.103524Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:8037d145628f30b0c68611a25617d572fa5d951d0dbeb6b4fdc831278d4d02c8","observation_id":"a2a25918-65f4-498b-aa88-02903cfe5efb","resolution":{"observed_at":"2026-08-10T18:04:29.755417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.107173Z","title":"Deep residual learning for image recognition","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.107173Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:4bf281980061f517a2d3505c2b994a1f22d14e776c14f65379d90e905f3c794c","observation_id":"c50a6fcd-c9b3-4f4a-910f-d266a34cc0dd","resolution":{"observed_at":"2026-08-10T18:04:29.107173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-11T08:20:29.798517Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-10T18:04:29.111156Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.111156Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:3403d91e8c8d9bcbb8a2de9fedf8ab8391912d020f8c80b69eef71c8fe74947d","observation_id":"d36cdb7d-015d-4418-ac90-b97eb910394b","resolution":{"observed_at":"2026-08-10T18:04:29.111156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.736470Z","title":"Language is not all you need: Aligning perception with language mod- els","venue":null,"work_id":"2bf752fc-950f-44de-9fae-e370a542506f","year":2024},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.114565Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:870d836f6e78e05cb0a3eb1cefa602b368cf50c8d4da6a280fe5b7b2a9924bfc","observation_id":"2b4871c5-eb6a-406d-bbe7-8768b5cae342","resolution":{"observed_at":"2026-08-10T18:04:29.740016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.726508Z","title":"What to look at and where: Se- mantic and spatial refined transformer for detecting human- object interactions","venue":null,"work_id":"6f6831f1-e560-49e7-96db-3c8d2f27c783","year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.117668Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:b313cdd07ef9c33d1c5e92e2d78d5cbbd5fb385da225c0680e7354eedd9e643a","observation_id":"aec5140f-b0fe-4f2f-a6f8-e9feafad167f","resolution":{"observed_at":"2026-08-10T18:04:29.729962Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.121047Z","title":"Scaling up visual and vision-language representa- tion learning with noisy text supervision","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.121047Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:4a09cf3bf0c78cb26e76da30262ddc43d6b24e143480a0560bc1c31e162c9ee3","observation_id":"98cf085f-0afa-468b-8912-494579e34914","resolution":{"observed_at":"2026-08-10T18:04:29.121047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.710480Z","title":"Detrs with hybrid matching","venue":null,"work_id":"83cbc518-89fe-48e6-869d-4b3cd291563e","year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.124643Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:fcf1151d4f83de5b7f2a07b95907af0d1377f65aefdbe3ca6187b7b67ddde421","observation_id":"9a19a440-f42a-4ffd-a385-85b6b616842c","resolution":{"observed_at":"2026-08-10T18:04:29.713768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.03726","last_updated":"2025-07-28T05:33:36Z","snapshot_observed_at":"2026-07-06T15:23:52.761222Z","submitted_at":"2023-05-05T17:59:46Z","title":"Otter: A Multi-Modal Model with In-Context Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.03726","snapshot_observed_at":"2026-08-10T18:04:29.127943Z","title":"Otter: A multi-modal model with in-context instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.127943Z"},"links":{"cited_paper":"/paper/2305.03726","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:56d60fbc5808c7188dd21da49f60e675137460f9ff3b6436d40da08ec9b185ec","observation_id":"a9f55ce5-6d81-41ca-8df5-214fd0312358","resolution":{"observed_at":"2026-08-10T18:04:29.127943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.700380Z","title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","venue":null,"work_id":"ff3816da-34b2-4bbe-9757-286f3c40744e","year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.130917Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:1eebdbea440ee6e808a12a4ee1fa1d31d8389a423d490c5b529a6b481ab202b4","observation_id":"0f0b8a0b-db11-4d83-b250-48c870406467","resolution":{"observed_at":"2026-08-10T18:04:29.704084Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.690212Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":"b3dbdffb-5d6d-4773-bd79-94236461ac3d","year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.133681Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:14de671813c616d5dc1820f9672ec92e6f626f64d35084504f06471e0e1e7bd3","observation_id":"69f40340-b426-4e15-8022-a297db112620","resolution":{"observed_at":"2026-08-10T18:04:29.693920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.680329Z","title":"Gen-vlkt: Simplify association and enhance 9 interaction understanding for hoi detection","venue":null,"work_id":"5930c7a5-69e1-44a2-961c-9ff18aaa3d09","year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.136248Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:1caf43f1ac9e085f3707efc87fc296e9081115015984e07b398ae1161b1e7222","observation_id":"9b6046be-2011-4ef4-a4bd-bc198685ad87","resolution":{"observed_at":"2026-08-10T18:04:29.683874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.671001Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"a20fea12-ca7a-4cca-805f-94fdc52bdcda","year":2014},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.139011Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:a336905fcf8b3d15a5e4814f2909e1813733e8ec24963156354eafdac90bfad4","observation_id":"c75e00cd-606a-454b-89dc-adbcbd5198a3","resolution":{"observed_at":"2026-08-10T18:04:29.674358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.141758Z","title":"Clip is also an efficient segmenter: A text-driven approach for weakly supervised semantic segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.141758Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:6096d51f0f990401191ed564d9551fdebd4d0b07723b75267f216ebaf6894d56","observation_id":"e587661f-7457-4d75-b58f-b5f7adca23f5","resolution":{"observed_at":"2026-08-10T18:04:29.141758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.144571Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.144571Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:6d48168e8f31d75ed0b2869c93b8bf1934ec38a4432601a99d7b39f130386074","observation_id":"203f3056-d11b-457e-9bc2-9ece97e445d0","resolution":{"observed_at":"2026-08-10T18:04:29.144571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.147466Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.147466Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:a3c1205bc41a19d38eaa2c04c4e32896ae9537002edd930dd9d09d2c4c5b35b6","observation_id":"173a6cd4-a4b5-47de-8e6f-d99d2fd8b58e","resolution":{"observed_at":"2026-08-10T18:04:29.147466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.149991Z","title":"Open scene understanding: Grounded situation recognition meets segment anything for helping people with visual im- pairments","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.149991Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:160d198e178d1b9a660f465bf9f7cc9044e79efad34444f39404556d16c70ece","observation_id":"28eadd23-0bcd-4653-ad1c-edd404ef1899","resolution":{"observed_at":"2026-08-10T18:04:29.149991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.637813Z","title":"Image segmenta- tion using text and image prompts","venue":null,"work_id":"4e02cc41-41d1-4f3f-8981-ff25991d00de","year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.152678Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:86571a9b60dad45680cf44832ac2ec2df5880ddbd802feda1973f4fb48f5a5ec","observation_id":"41189d75-9cf3-48b8-8ba5-e29d2e20b99e","resolution":{"observed_at":"2026-08-10T18:04:29.641316Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.628496Z","title":"Clip4hoi: Towards adapting clip for practi- cal zero-shot hoi detection","venue":null,"work_id":"9587ce78-dead-4701-a073-a13404d6e13f","year":2024},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.155309Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:b84aabf142e57d692af9405bb91d3af1eafb1f5c4b2392af396efc89ae4a2110","observation_id":"1b30bbb4-4a3a-462d-90ea-c3b706b46657","resolution":{"observed_at":"2026-08-10T18:04:29.631662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09734","last_updated":"2021-11-18T14:49:15Z","snapshot_observed_at":"2026-08-13T13:33:48.501765Z","submitted_at":"2021-11-18T14:49:15Z","title":"ClipCap: CLIP Prefix for Image Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09734","snapshot_observed_at":"2026-08-10T18:04:29.158092Z","title":"Clip- cap: Clip prefix for image captioning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.158092Z"},"links":{"cited_paper":"/paper/2111.09734","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:dc907e29778abeca34ea83f079ac970a5f431cbeb94b6af327da1cae2121f7ec","observation_id":"ff2e719c-5f1c-46f5-b7f8-5eebdb0b2c27","resolution":{"observed_at":"2026-08-10T18:04:29.158092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.619827Z","title":"Hoiclip: Efficient knowledge transfer for hoi detection with vision-language models","venue":null,"work_id":"975b8014-3858-4b21-9e3a-1bf2d0899bd6","year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.161682Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:d3f0202264b3459ccc52cf5e7bbc7de8c86be4e3ea477af1f489f0edb359e6a4","observation_id":"c2de3706-8bd1-40e1-953d-94b84f2f6800","resolution":{"observed_at":"2026-08-10T18:04:29.622766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-11T10:12:11.384939Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-10T18:04:29.164848Z","title":"Dinov2: Learning robust visual features without supervision","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.164848Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:33f1d1df6cde84263aad94023b2ec8c9cbdd56407cc3e4b015a9214b93e5c2ac","observation_id":"7827cb2b-6ab9-49b2-a6e6-498547c4e7a9","resolution":{"observed_at":"2026-08-10T18:04:29.164848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-08-12T12:24:23.815073Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14824","snapshot_observed_at":"2026-08-10T18:04:29.168429Z","title":"Kosmos-2: Ground- ing multimodal large language models to the world","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.168429Z"},"links":{"cited_paper":"/paper/2306.14824","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:089c0d40e4299b013665f76bbe8371c8795e59d546aff45973cf3ad7963059c5","observation_id":"01ecefc0-edab-4d95-bab2-38ae20cf08c2","resolution":{"observed_at":"2026-08-10T18:04:29.168429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.610959Z","title":"Grounded situation recognition","venue":null,"work_id":"1ffd6e68-ba16-4a28-a756-32ce98839e43","year":2020},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.172067Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:032b22839df8345a67130844f2f2620741bfbff16898b6d12a8c80e1d490a49e","observation_id":"c0b6bd39-9a9c-4e0f-b8e5-92328280020d","resolution":{"observed_at":"2026-08-10T18:04:29.613873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.601486Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":"b3c8d728-edb8-4518-9b79-6baa1f737285","year":2021},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.175274Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:3bf2ddd8fbeb66ff6895e6af54a12dd20e8f2f4a42cc7297975d84d1f56bec00","observation_id":"36fd30a8-89be-4a73-ac93-645cafdb9a72","resolution":{"observed_at":"2026-08-10T18:04:29.604554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.592464Z","title":"Robotic vision for human-robot interaction and collaboration: A survey and systematic re- view","venue":null,"work_id":"9323f71d-ee1a-4070-abac-f9f2463892bc","year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.178468Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:dc3978ee7a34a2248fcdcdab9316ca5c7a530f7a9d13cf463eb033d7d33df557","observation_id":"64b6a403-d32d-4f5e-8218-9aa542878d80","resolution":{"observed_at":"2026-08-10T18:04:29.595466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.181632Z","title":"High-resolution image synthesis with latent diffusion models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.181632Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:99f3b0c245988fa01663c88e296df20be5f224363afe4f478adf22373392a828","observation_id":"d350ddd5-f586-429d-8707-baa5abe9cd7d","resolution":{"observed_at":"2026-08-10T18:04:29.181632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.577262Z","title":"Clip- situ: Effectively leveraging clip for conditional predictions in situation recognition","venue":null,"work_id":"602011b4-d046-4cb3-8d32-010cc35cefc0","year":2024},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.184781Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:ef3c69b6d493ea4cadeef3e79fc38574080155744b08dd813e0a7be7e3f7bd6e","observation_id":"f45f3ff3-3435-493c-87a2-7df579bb0a82","resolution":{"observed_at":"2026-08-10T18:04:29.580244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.567579Z","title":"Ut-interaction dataset, icpr contest on semantic description of human activities (sdha)","venue":null,"work_id":"be525552-b237-400e-bb52-df9d224ad5c5","year":2010},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.188041Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:b34fab69315a39bc2c987b5273409654a840d497c1876f3cdb134eaa016ef179","observation_id":"127bb760-d0e2-4bf0-b721-374d14b86ab8","resolution":{"observed_at":"2026-08-10T18:04:29.571391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.04696","last_updated":"2020-05-21T16:53:47Z","snapshot_observed_at":"2026-08-10T19:14:01.245716Z","submitted_at":"2020-04-09T17:26:52Z","title":"BLEURT: Learning Robust Metrics for Text Generation","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.04696","snapshot_observed_at":"2026-08-10T18:04:29.191112Z","title":"Bleurt: Learning robust metrics for text generation","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.191112Z"},"links":{"cited_paper":"/paper/2004.04696","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:be0612c716a07330889b7ffd0fa92ab956a6490e6f39a416d9b92696ade915bf","observation_id":"97dbb885-b995-46bc-990c-daac2e0537cd","resolution":{"observed_at":"2026-08-10T18:04:29.191112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.557497Z","title":"Analyzing human– human interactions: A survey","venue":null,"work_id":"0d4a0aec-280a-4397-b1cc-20c26599d718","year":2019},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.194562Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:98a97736a37d0e18d7e9bdeab8de7de0ba31a95784b5195d7cf693ce80032a1a","observation_id":"3816ec7c-6c74-4cb7-b9cb-efa396178278","resolution":{"observed_at":"2026-08-10T18:04:29.561295Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.197813Z","title":"Generative multimodal mod- els are in-context learners","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.197813Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:fcf83f9627ab6416388ccda534fa9f8ee55e598163dc9e4ec3b7ba8909fe9861","observation_id":"10e15eda-365b-4947-b854-62f591b869b4","resolution":{"observed_at":"2026-08-10T18:04:29.197813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.200832Z","title":"Qpic: Query-based pairwise human-object interaction detec- tion with image-wide contextual information","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.200832Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:075542fcc84276234aaa640827d05babb4bdf3a09b4291a75d81748ba5fe5d7e","observation_id":"9f40e930-6501-479f-aff6-e125cda380c4","resolution":{"observed_at":"2026-08-10T18:04:29.200832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.535053Z","title":"Learning latent temporal structure for complex event detection","venue":null,"work_id":"c833e321-7ad6-489d-85e6-c7156d485471","year":2012},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.203325Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:1b7a778603d70745d43d05cae0e84ba734ce44085c41f25d14b6398b3e91ced4","observation_id":"87e9cac3-da5f-49a0-9d3a-05c94c91d1fc","resolution":{"observed_at":"2026-08-10T18:04:29.538545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.524855Z","title":"Learning spatiotemporal features with 3d convolutional networks","venue":null,"work_id":"ff15838f-bcd7-4ff7-8506-2aa1fcb239e9","year":2015},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.205878Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:94e5ac655ff832b328382c66ffba2a07f781f60856236ff0adf716a3ddf6115b","observation_id":"86b04a56-3d0e-4dee-bdb7-443a672c3542","resolution":{"observed_at":"2026-08-10T18:04:29.528475Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.08472","last_updated":"2021-09-17T11:21:34Z","snapshot_observed_at":"2026-08-13T18:09:42.014649Z","submitted_at":"2021-09-17T11:21:34Z","title":"ActionCLIP: A New Paradigm for Video Action Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.08472","snapshot_observed_at":"2026-08-10T18:04:29.208421Z","title":"Actionclip: A new paradigm for video action recognition","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.208421Z"},"links":{"cited_paper":"/paper/2109.08472","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:890070edf83a2054e6b5dd3f83b12291420f7efa6f306116a047271b3df799c6","observation_id":"1d9f3e91-2a1a-4368-9bce-da217fdc3560","resolution":{"observed_at":"2026-08-10T18:04:29.208421Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.515376Z","title":"Cris: Clip- driven referring image segmentation","venue":null,"work_id":"c0102cc1-a7f7-46be-bb64-8c629ac29461","year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.211084Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:f4ad20a9fc09018935708d59696ad7d0bb791af749ba91d367fabdd3ff3be517","observation_id":"d29cfbf7-2dcc-4779-aaab-4305283204c6","resolution":{"observed_at":"2026-08-10T18:04:29.518750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.505859Z","title":"Rethinking the two-stage framework for grounded sit- uation recognition","venue":null,"work_id":"4a25dfbc-71dd-4d8a-9bc1-d1025dc6bda9","year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.213705Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:f73b479fdd888c376d0e387ed68f037660b5f9d48eb45fb6ff4d2ab14f94328f","observation_id":"8357c391-7ff1-48fc-8cb8-d09e3603ec84","resolution":{"observed_at":"2026-08-10T18:04:29.509425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.496318Z","title":"Rec- ognize complex events from static images by fusing deep channels","venue":null,"work_id":"24bd879c-1178-49de-aa51-9187ce87c62f","year":null},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.216364Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:a740e0c5f61138e8d728370602c2a7f872eba0c5aae7b8c3a70eba557cc25968","observation_id":"4d4757c7-835d-4760-bafa-ec89c30a47a7","resolution":{"observed_at":"2026-08-10T18:04:29.499989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.486607Z","title":"Recognizing proxemics in personal photos","venue":null,"work_id":"d5938328-d28f-4de9-8f3d-38d00ce9c255","year":2012},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.219547Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:3d79f2222222b8dbc705c7616a67b0772068a469c8af65662d2c7421b483e563","observation_id":"40e3df64-9735-4a28-aec7-cafe846af708","resolution":{"observed_at":"2026-08-10T18:04:29.490090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.478055Z","title":"Situa- tion recognition: Visual semantic role labeling for image understanding","venue":null,"work_id":"5030fa30-c8e8-4869-9a65-56780971d225","year":null},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.222066Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:5eda86ef5d55f118170b0b497fa0a4050c23efe589038fe8fd869924ff9db348","observation_id":"d9a2302c-2e38-4d47-8166-9ab607be7c5d","resolution":{"observed_at":"2026-08-10T18:04:29.481079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.459740Z","title":"Rlip: Rela- tional language-image pre-training for human-object inter- action detection","venue":null,"work_id":"dc605551-4c19-41e2-ae26-68c74317e99b","year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.228117Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:6062fc45466a39c3f32473186c3f9ec4c509adb61108f33acceada1c017804ae","observation_id":"1fc5b6a0-4560-4c77-94cf-9e8393e3030d","resolution":{"observed_at":"2026-08-10T18:04:29.463025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.450112Z","title":"Rlipv2: Fast scaling of re- lational language-image pre-training","venue":null,"work_id":"5ce61a4e-5455-491f-80ea-0812aceca5e2","year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.230741Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:99a7030d00d1006300d3ae9ce5d39d7b8653a1a981eda590192aad84b7505f5e","observation_id":"a632b5bf-5002-4f11-9363-c366041d74ac","resolution":{"observed_at":"2026-08-10T18:04:29.453244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.439755Z","title":"Lit: Zero-shot transfer with locked-image text tuning","venue":null,"work_id":"bc6660ea-aee7-449e-af0d-60081cd11c4e","year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.233316Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:6f3e55da87711f9bd33a729247c1d37dc1d650219608b9a11f2d6bf3dc0108e5","observation_id":"a3ee7e89-a783-4f6c-9663-796d404fb7b0","resolution":{"observed_at":"2026-08-10T18:04:29.443341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.429522Z","title":"Zhang, Dylan Campbell, and Stephen Gould","venue":null,"work_id":"95788217-0eb5-4d35-87c7-e6d135d26d3b","year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.236106Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:12d8c27810bc55250ca6078e71eefac029a34b18e1e3047ca3c6b2b1472a6d4f","observation_id":"a96dc666-14e7-48e0-b7b5-af39e5851057","resolution":{"observed_at":"2026-08-10T18:04:29.432872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.419488Z","title":"Zhang, Yuhui Yuan, Dylan Campbell, Zhuoyao Zhong, and Stephen Gould","venue":null,"work_id":"7b5bbd0b-5600-4037-be78-741222bbad41","year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.238616Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:0d6b27ed1799e64ded08d0ae10273f8d40e8e7de036e6a6d19b74347eae5eff5","observation_id":"4fd9d762-a0b8-4a76-a702-5de0f1674ac7","resolution":{"observed_at":"2026-08-10T18:04:29.423024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-08-10T18:04:29.241423Z","title":"Opt: Open pre-trained trans- former language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.241423Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:7bcadd099becf48910829db398c109011e805ce927b959858fc0483ada70bad2","observation_id":"903ec1a6-83d9-435b-ae74-c067191ce6b8","resolution":{"observed_at":"2026-08-10T18:04:29.241423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.408550Z","title":"Zegclip: Towards adapting clip for zero-shot se- mantic segmentation","venue":null,"work_id":"60855165-db3b-415b-bccb-3675d012d9c3","year":2023},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.244936Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:cb36793082a3340d1543f35a828aae8845f5941ee904b3132eaed8f23730c9c6","observation_id":"92e5280f-c78a-4577-9639-a71346507a45","resolution":{"observed_at":"2026-08-10T18:04:29.412879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.00443","last_updated":"2022-06-22T09:19:40Z","snapshot_observed_at":"2026-08-14T13:00:14.886000Z","submitted_at":"2022-01-03T00:55:33Z","title":"Scene Graph Generation: A Comprehensive Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.00443","snapshot_observed_at":"2026-08-10T18:04:29.248093Z","title":"Scene graph generation: A comprehensive survey","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.248093Z"},"links":{"cited_paper":"/paper/2201.00443","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:241a91445ea25390de31dc3dbe1f006f6ff2ceadf3660f30897bbf19e3c34c2b","observation_id":"93ec1946-844a-4ce5-b39c-6318c576862c","resolution":{"observed_at":"2026-08-10T18:04:29.248093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.06567","last_updated":"2020-12-11T18:54:08Z","snapshot_observed_at":"2026-08-12T22:15:53.674168Z","submitted_at":"2020-12-11T18:54:08Z","title":"A Comprehensive Study of Deep Video Action Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.06567","snapshot_observed_at":"2026-08-10T18:04:29.251417Z","title":"A comprehensive study of deep video action recognition","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.251417Z"},"links":{"cited_paper":"/paper/2012.06567","citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:c4fd68eb267bd080d0c42acb322839ede0aeca32d02eb8b8e246084ea0b40a20","observation_id":"1dc5a034-c241-4ee0-b834-9ed607815125","resolution":{"observed_at":"2026-08-10T18:04:29.251417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.397046Z","title":null,"venue":null,"work_id":"62f02d84-5fcd-41be-abe9-ad464bf3a9d3","year":null},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.255261Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:4846e5a8434f0a73939e751632a79fb20cab42784123044796c5bc1bcd20d0be","observation_id":"6596321f-a215-413d-98bc-57a47de3d03b","resolution":{"observed_at":"2026-08-10T18:04:29.401574Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T18:04:29.469288Z","title":null,"venue":null,"work_id":"293d4152-7669-4463-b2dd-ea360742304b","year":null},"citing_paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations","version":3},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:29.224974Z"},"links":{"citing_paper":"/paper/2501.11653"},"observation_digest":"sha256:78fa19f6bafe0a108daca6aa0d1103ab7a66c5d5c5175d35fbe3ad1ccc1ba9ff","observation_id":"1356fec0-a3f6-4e6c-96b2-c9188483b343","resolution":{"observed_at":"2026-08-10T18:04:29.472198Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.11653","last_updated":"2025-05-03T18:39:13Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T18:06:47.639685Z","submitted_at":"2025-01-20T18:33:46Z","title":"Dynamic Scene Understanding from Vision-Language Representations"},"reference_resolution":{"displayed":65,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":26,"verified_exact":0,"verified_fuzzy":39},"total_outbound_references":65},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 65 of 65 outbound references and 0 inbound Pith citation observations for arXiv:2501.11653."}