{"as_of":"2026-08-16T09:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:09f6bce3e6ae2b34661161acdde8ada8e8f7ef99db0ba5f62fd79a864877fe20","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T23:50:44.081050Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.03581/citation-record","integrity":"/paper/2505.03581/integrity","json":"/paper/2505.03581/citation-record.json","paper":"/paper/2505.03581"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.706282Z","title":"Conceptgraphs: Open-vocabulary 3d scene graphs for perception and planning,","venue":null,"work_id":"7cdf7201-33a8-4a99-a6db-41ada5dfd4fc","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.879396Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:c057122cac0591a04e90ec7191f956852eee6c802abeaefb1447898a83202a6e","observation_id":"2f0771a3-f1ca-4c86-8055-a9b79bb5f917","resolution":{"observed_at":"2026-08-15T23:50:44.709445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.696442Z","title":"Beyond bare queries: Open-vocabulary object retrieval with 3d scene graph,","venue":null,"work_id":"e23eaadb-e604-4e1c-8ce8-25d7ef75c582","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.883967Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:6882f587d3f53ff51ac7e4d0de1b2d2ec3d614c99c9b7e52bdda352bf4566543","observation_id":"4c7416d5-173b-4e37-be71-32794146e0e3","resolution":{"observed_at":"2026-08-15T23:50:44.700268Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.685119Z","title":"Search3d: Hierarchical open-vocabulary 3d segmenta- tion,","venue":null,"work_id":"de22d83e-d791-400a-85b5-0b22bd15f883","year":2025},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.888014Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:04d3f610071c790d0b77ecfe172d6bea429b7028392cd6182f043c09df374515","observation_id":"84232814-3510-4da1-a4f1-dd96325f2251","resolution":{"observed_at":"2026-08-15T23:50:44.689789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.891225Z","title":"Hierarchical open-vocabulary 3d scene graphs for language-grounded robot navigation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.891225Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:966f92cd604dd07814e1e492b0878946f63706832cf6f7e0713f46af0e5bf10d","observation_id":"d7074e35-bd7b-4e54-99cb-ece5e1f239d7","resolution":{"observed_at":"2026-08-15T23:50:43.891225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.895182Z","title":"Clio: Real-time task-driven open-set 3d scene graphs,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.895182Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:8a962f6434aad18d75424f294a76b70e9b91c6fc7702f4476605608400d93b7e","observation_id":"b7d94e18-e615-42c0-b8b0-51c029add4f6","resolution":{"observed_at":"2026-08-15T23:50:43.895182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.664806Z","title":"4d panoptic scene graph generation,","venue":null,"work_id":"98b0287e-2bc6-4574-ba16-5b0f9b21a446","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.898419Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:b3e38d967ff64a3924a27c48cb0a2909564e32576bc4f5c835287ecf6aca4665","observation_id":"b9494be8-c440-42a5-b3bf-6204d5c977cd","resolution":{"observed_at":"2026-08-15T23:50:44.668082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.654878Z","title":"G-retriever: Retrieval-augmented generation for textual graph understanding and question answering,","venue":null,"work_id":"0681acce-bcba-45fe-b285-8c465ab26894","year":2025},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.902087Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:6688012da2c1df5a301f0ddbfa2aa57a2358068bfcc9e01245b0f7878eefff2e","observation_id":"46c15d20-7df7-4f50-a137-3da68dc4bdce","resolution":{"observed_at":"2026-08-15T23:50:44.658325Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05862","last_updated":"2024-02-08T17:51:44Z","snapshot_observed_at":"2026-08-15T04:29:31.008152Z","submitted_at":"2024-02-08T17:51:44Z","title":"Let Your Graph Do the Talking: Encoding Structured Data for LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05862","snapshot_observed_at":"2026-08-15T23:50:43.905097Z","title":"Let your graph do the talking: Encoding structured data for llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.905097Z"},"links":{"cited_paper":"/paper/2402.05862","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:0e7ed0af2e8c32e0c7232dde1251c684bb840a5cf6688e62fae0904ada8ad140","observation_id":"645bd11b-dc08-4e04-8761-17006d2928db","resolution":{"observed_at":"2026-08-15T23:50:43.905097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.643891Z","title":"Can llms enhance performance prediction for deep learning models?","venue":null,"work_id":"cfcbcc08-9f4c-4fab-9731-aaaaec3ffd0a","year":null},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.908611Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:9c618f4f6ebab3bab52cdef5dd94428959ac5311f5a1a0f4eed36cab3f32c85c","observation_id":"e8cc9849-0fb3-46d9-bcf3-73b8e8b9bccb","resolution":{"observed_at":"2026-08-15T23:50:44.647941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.09711","last_updated":"2024-05-15T21:53:54Z","snapshot_observed_at":"2026-08-13T05:15:32.116612Z","submitted_at":"2024-05-15T21:53:54Z","title":"STAR: A Benchmark for Situated Reasoning in Real-World Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.09711","snapshot_observed_at":"2026-08-15T23:50:43.912140Z","title":"Star: A benchmark for situated reasoning in real-world videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.912140Z"},"links":{"cited_paper":"/paper/2405.09711","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:b0de76a4713b4677b59ab9ee70574f79132ca6faa88ea2a74c1c4e08e3a3e901","observation_id":"8128cca7-1eee-44b0-a647-76d99cacdfdf","resolution":{"observed_at":"2026-08-15T23:50:43.912140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.634277Z","title":"Agqa 2.0: An updated benchmark for compositional spatio-temporal reasoning,","venue":null,"work_id":"59d061a8-a3a2-4f0a-9328-8a1aa8421ddd","year":null},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.915839Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:a22829caf30b71f99bc93b41d0ece9178f9f0a3eea3c3e5be8bc07747e75df2c","observation_id":"8c57e5aa-525e-46ac-9490-95e8720d9733","resolution":{"observed_at":"2026-08-15T23:50:44.637619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.922505Z","title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.922505Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:6ee9b4885410bb98a511ac6907a56b7ab01665e34156a4b7954526fd3bfaefdd","observation_id":"1885515b-95f2-4100-8240-6e09a118c67a","resolution":{"observed_at":"2026-08-15T23:50:43.922505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.925544Z","title":"Gqa: A new dataset for real- world visual reasoning and compositional question answering,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.925544Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:c1e6db1a869d84bf6d545c75b1affdb1e1827f697214bba21a96b240c8ad29e6","observation_id":"f758b0d9-18c6-47ca-b6e9-be1bb9780696","resolution":{"observed_at":"2026-08-15T23:50:43.925544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.615326Z","title":"Panoptic scene graph generation,","venue":null,"work_id":"18dd88f2-1a14-4876-8049-072983bc882a","year":2022},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.928530Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:9f7a2976b9c06aa1231d9b40c6980bd5e5592639991e92561fe922fed48bbfd7","observation_id":"9a767cad-a860-4189-86b5-bef289d6efa8","resolution":{"observed_at":"2026-08-15T23:50:44.618448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.606047Z","title":"Action genome: Actions as compositions of spatio-temporal scene graphs,","venue":null,"work_id":"b1cd20d0-f496-4d4b-a713-5972178e6c81","year":2020},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.931740Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:702724be2b07d9f3cbf55ffdd9dd87ee907d5a7e5c15b800942d3cc76d3c2541","observation_id":"65521ffd-71c5-42e0-b747-d37d6c798f85","resolution":{"observed_at":"2026-08-15T23:50:44.609510Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.596888Z","title":"Panoptic video scene graph generation,","venue":null,"work_id":"a98a2655-3b38-4449-9ff6-dc2be15dc72b","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.934810Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:720aecfbf4fd8f60181f11c229a208697d6163d43faa4f31753cf85e3a2089ed","observation_id":"0fed5c8e-ac80-492e-a37b-dd2684cfbfa4","resolution":{"observed_at":"2026-08-15T23:50:44.600160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.938041Z","title":"Egtr: Extracting graph from transformer for scene graph generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.938041Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:7c6c9523ceff0b889194c8c520634b760546a5fbf2fab01ba1732ef31c2fcc12","observation_id":"a090021c-ddbf-4edd-b9a9-393dcf6de988","resolution":{"observed_at":"2026-08-15T23:50:43.938041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.941400Z","title":"Reltr: Relation transformer for scene graph generation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.941400Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:6e6345ec6732b1092185bc42f827ab148ee27019aca7e66802d95eae5e327b31","observation_id":"30a67c2d-1411-459d-a581-f32df200fe56","resolution":{"observed_at":"2026-08-15T23:50:43.941400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.577043Z","title":"Oed: towards one-stage end-to- end dynamic scene graph generation,","venue":null,"work_id":"b337302e-c865-48f8-b514-401429587e0e","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.945053Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:499a29ea6c1d38190446726eb2015c905dfe7a5f2a20828d8f48c06aca900048","observation_id":"a6806e38-1f43-436e-a43a-bf1604f5a723","resolution":{"observed_at":"2026-08-15T23:50:44.580376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-15T23:50:43.948634Z","title":"Sam 2: Segment anything in images and videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.948634Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:27db8eff5cc557bef0e2f0f0732fa06d243da110480b9057320520f51ab1230c","observation_id":"1605903d-9619-4b17-b3b2-2cfbf2bf3ee1","resolution":{"observed_at":"2026-08-15T23:50:43.948634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.952811Z","title":"Llavanext: Improved reasoning, ocr, and world knowledge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.952811Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:64d55d48637f671248e4d92305007ea68cf7efe397088ea914a3453630f7c4c7","observation_id":"7e316375-8fa3-404a-83ff-1832345ffe1f","resolution":{"observed_at":"2026-08-15T23:50:43.952811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:43.956865Z","title":"Yolo-world: Real-time open-vocabulary object detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.956865Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:76e36d84cc3c06ac034a936f3d1ae126ea0539f867fd2153b679df7ba349eed4","observation_id":"22188316-e050-47d6-9f6f-be776b314683","resolution":{"observed_at":"2026-08-15T23:50:43.956865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-15T23:50:43.960371Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.960371Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:5df0114e52001a4a6ad440dc388389c6c28a575a2380d6ee6a637b6d508c6459","observation_id":"e3418715-8ace-498d-bcf7-ef9b3d605286","resolution":{"observed_at":"2026-08-15T23:50:43.960371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04652","last_updated":"2025-01-21T10:12:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-07T16:52:49Z","title":"Yi: Open Foundation Models by 01.AI","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04652","snapshot_observed_at":"2026-08-15T23:50:43.963321Z","title":"Yi: Open foundation models by 01. ai,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.963321Z"},"links":{"cited_paper":"/paper/2403.04652","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:5bf4a34cc0f82995f45a6d7349635985a7f5f601513b92d8c6617e78ea07d9ac","observation_id":"5c725269-06a3-4ead-ae5e-0a3f44765277","resolution":{"observed_at":"2026-08-15T23:50:43.963321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04468","last_updated":"2026-04-25T07:16:42Z","snapshot_observed_at":"2026-08-03T08:48:57.969106Z","submitted_at":"2024-12-05T18:59:55Z","title":"NVILA: Efficient Frontier Visual Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04468","snapshot_observed_at":"2026-08-15T23:50:43.967414Z","title":"Nvila: Efficient frontier visual language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.967414Z"},"links":{"cited_paper":"/paper/2412.04468","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:69c580a6f73de2299e5c7db9986ad2396f331fda181cc1a56cce961a099196dd","observation_id":"77f7d309-02d3-4c7c-b892-64d46f8a505c","resolution":{"observed_at":"2026-08-15T23:50:43.967414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-14T04:17:22.593941Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-15T23:50:43.970524Z","title":"Qwen2. 5-vl technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.970524Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:18f7347ce0a047c40e10bd6d6702302cae4122ff1bda16d739c262acd379cfa7","observation_id":"740d89b9-7117-4b46-a85b-ff8648391f89","resolution":{"observed_at":"2026-08-15T23:50:43.970524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06723","last_updated":"2025-02-26T22:54:53Z","snapshot_observed_at":"2026-08-12T23:25:59.697320Z","submitted_at":"2024-07-09T09:55:04Z","title":"Graph-Based Captioning: Enhancing Visual Descriptions by Interconnecting Region Captions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06723","snapshot_observed_at":"2026-08-15T23:50:43.973960Z","title":"Graph- based captioning: Enhancing visual descriptions by interconnecting region captions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.973960Z"},"links":{"cited_paper":"/paper/2407.06723","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:243f8d55feb6bce166c318d0463cd67abae61c6a5dbc680713a50a62370d7f67","observation_id":"8a8e3370-8c8f-4ecf-ad26-162d9a30876b","resolution":{"observed_at":"2026-08-15T23:50:43.973960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.07012","last_updated":"2024-12-29T03:52:23Z","snapshot_observed_at":"2026-08-14T11:59:45.077919Z","submitted_at":"2024-12-09T21:44:02Z","title":"ProVision: Programmatically Scaling Vision-centric Instruction Data for Multimodal Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.07012","snapshot_observed_at":"2026-08-15T23:50:43.977587Z","title":"Provision: Programmatically scaling vision-centric instruction data for multimodal language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.977587Z"},"links":{"cited_paper":"/paper/2412.07012","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:cb0ce11a8af8b686e5a2efbd1692fd4416a50e67fad84018f91179ba9e7e7c04","observation_id":"04a36963-6c86-4e53-87ab-7c9073ebd39a","resolution":{"observed_at":"2026-08-15T23:50:43.977587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.557312Z","title":"Llm4sgg: large language models for weakly supervised scene graph generation,","venue":null,"work_id":"658dfab4-3759-43aa-922f-47664282bb9d","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.980913Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:a7033ac27105409268bcd6600479c3083aba7d34db5f59cfa05c83dd7171876b","observation_id":"f6eeaa08-33b7-48ca-b31f-e80b1c3944c1","resolution":{"observed_at":"2026-08-15T23:50:44.560483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-08-14T22:29:28.847917Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14171","snapshot_observed_at":"2026-08-15T23:50:43.984048Z","title":"Thinking in space: How multimodal large language models see, remember, and recall spaces,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.984048Z"},"links":{"cited_paper":"/paper/2412.14171","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:367a70e9dcde3e63e92010361a050907283959424d6d96a8819868ac79ca92b5","observation_id":"93d64783-f5cc-43c8-b905-59f1660eccb1","resolution":{"observed_at":"2026-08-15T23:50:43.984048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18938","last_updated":"2024-12-03T03:56:52Z","snapshot_observed_at":"2026-08-15T00:49:52.466345Z","submitted_at":"2024-09-27T17:38:36Z","title":"From Seconds to Hours: Reviewing MultiModal Large Language Models on Comprehensive Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.18938","snapshot_observed_at":"2026-08-15T23:50:43.987304Z","title":"From seconds to hours: Reviewing multimodal large language models on comprehensive long video understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.987304Z"},"links":{"cited_paper":"/paper/2409.18938","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:b9a22317ba46f23c112cf826bad96047ef119535e952f12da6da1c4d47e34b01","observation_id":"db1551b9-ffdd-478d-b7fb-7678a1a290c6","resolution":{"observed_at":"2026-08-15T23:50:43.987304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.02765","last_updated":"2025-01-06T05:15:59Z","snapshot_observed_at":"2026-08-14T12:32:34.328935Z","submitted_at":"2025-01-06T05:15:59Z","title":"Visual Large Language Models for Generalized and Specialized Applications","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.02765","snapshot_observed_at":"2026-08-15T23:50:43.990891Z","title":"Visual large language models for generalized and specialized applications,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.990891Z"},"links":{"cited_paper":"/paper/2501.02765","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:27c373f935b9c1c2b2aeee578a8acb78dc354cc6846506dab7466eb5b8e0eb9d","observation_id":"9818144f-18fa-476e-a85b-1e19b57e3fa5","resolution":{"observed_at":"2026-08-15T23:50:43.990891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.547692Z","title":"(2.5+ 1) d spatio- temporal scene graphs for video question answering,","venue":null,"work_id":"c677c7d2-f5a9-48db-a378-62b1d7c2db90","year":2022},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.993936Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:917d347e1c76fe0312aeaab122633de6178d0bb2830d1db2ef393294118decca","observation_id":"3d0cf3e8-42d0-4d39-a3f8-79333589329e","resolution":{"observed_at":"2026-08-15T23:50:44.551272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.538485Z","title":"Action scene graphs for long-form understanding of egocentric videos,","venue":null,"work_id":"b58772ee-5ec9-413f-b2c4-9de58fe5f38f","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.997532Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:9c376baa561aa739d73c418c827e62ad826bf8c66c9f17d1ca6699c5db574561","observation_id":"4cc7d048-e096-4191-b2f3-254a8c9b05f7","resolution":{"observed_at":"2026-08-15T23:50:44.541815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.18042","last_updated":"2025-03-31T08:16:49Z","snapshot_observed_at":"2026-08-16T07:06:08.115739Z","submitted_at":"2024-11-27T04:24:39Z","title":"HyperGLM: HyperGraph for Video Scene Graph Generation and Anticipation","version":2},"cited_work":{"arxiv_id":"2411.18042","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.18042","snapshot_observed_at":"2026-08-15T23:50:44.314610Z","title":"HyperGLM: HyperGraph for Video Scene Graph Generation and Anticipation","venue":"cs.CV","work_id":"61c5b646-5432-4e8d-923b-74b8ac091cbc","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.000451Z"},"links":{"cited_paper":"/paper/2411.18042","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:adebb8d831a81752fec30de15ed49533503b57c5aabc89ebf58853ed59af8ad0","observation_id":"1b858b1f-bfeb-47f1-b0bd-5e31f8284cb9","resolution":{"observed_at":"2026-08-15T23:50:44.318336Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.00161","last_updated":"2025-03-30T14:31:41Z","snapshot_observed_at":"2026-08-16T04:20:19.899128Z","submitted_at":"2024-11-29T11:54:55Z","title":"STEP: Enhancing Video-LLMs' Compositional Reasoning by Spatio-Temporal Graph-guided Self-Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.00161","snapshot_observed_at":"2026-08-15T23:50:44.003979Z","title":"Step: Enhancing video-llms’ composi- tional reasoning by spatio-temporal graph-guided self-training,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.003979Z"},"links":{"cited_paper":"/paper/2412.00161","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:895261cd74b5c0ea03b3ebace49ed68e1fafcb56f42c28de3655915323319298","observation_id":"d0f2412a-688e-4361-b79c-5c67e17a3d8c","resolution":{"observed_at":"2026-08-15T23:50:44.003979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13663","last_updated":"2024-12-19T06:32:26Z","snapshot_observed_at":"2026-08-14T20:02:35.071920Z","submitted_at":"2024-12-18T09:39:44Z","title":"Smarter, Better, Faster, Longer: A Modern Bidirectional Encoder for Fast, Memory Efficient, and Long Context Finetuning and Inference","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.13663","snapshot_observed_at":"2026-08-15T23:50:44.007182Z","title":"Smarter, better, faster, longer: A modern bidirectional encoder for fast, memory efficient, and long context finetuning and inference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.007182Z"},"links":{"cited_paper":"/paper/2412.13663","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:92d9ef9ef842b4214521601d5d2f1dd5ace659d98ff276bf1ff3715d33d6f001","observation_id":"a1abcacc-c6b9-435a-a869-a612ea80fcd2","resolution":{"observed_at":"2026-08-15T23:50:44.007182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.010411Z","title":"Benchmarking graph neural networks,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.010411Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:81e897ed93ddd3a6165d67804c46597ac5e1736bab06639d05a625558630e607","observation_id":"d4250c15-2b0a-4a16-ae60-b28dee74ec5e","resolution":{"observed_at":"2026-08-15T23:50:44.010411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03509","last_updated":"2021-05-10T02:23:20Z","snapshot_observed_at":"2026-08-11T19:52:55.524234Z","submitted_at":"2020-09-08T04:04:04Z","title":"Masked Label Prediction: Unified Message Passing Model for Semi-Supervised Classification","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03509","snapshot_observed_at":"2026-08-15T23:50:44.013299Z","title":"Masked label prediction: Unified message passing model for semi-supervised classification,","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.013299Z"},"links":{"cited_paper":"/paper/2009.03509","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:284ca352650a821e9b5ee409a088b3de94e94161c90e0a95894fbd4148d6e86c","observation_id":"d938cc35-c39c-4172-ac51-f9f679921791","resolution":{"observed_at":"2026-08-15T23:50:44.013299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.016593Z","title":"Roformer: En- hanced transformer with rotary position embedding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.016593Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:1faea35f0eecc40919d00b562d377857c1e289d806830d6a23482e580f385b82","observation_id":"bd3c880a-c483-4c30-9183-6ac3d7a94fd0","resolution":{"observed_at":"2026-08-15T23:50:44.016593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.019556Z","title":"Blip-2: Bootstrapping language- image pre-training with frozen image encoders and large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.019556Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:5f40dc0ff67beb2a7528b45e86ca404bd1064932c8fddfa63e18670f23eac638","observation_id":"a12445ac-7fcb-4ea8-88e4-d207e0a54489","resolution":{"observed_at":"2026-08-15T23:50:44.019556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.512868Z","title":"Agqa: A benchmark for compositional spatio-temporal reasoning,","venue":null,"work_id":"b95d31ac-8819-447f-9290-db34fcdb0126","year":2021},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.022635Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:5803f0f7c98e3ea9d3e9880a0d7b06bc51caba1b4f67e20455603830d739b64d","observation_id":"8cd2c11b-fd18-4e51-806c-f2e36ee7cb5b","resolution":{"observed_at":"2026-08-15T23:50:44.516269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.025383Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.025383Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:6951fbfe4bd149b9a506d39c1f3a70f16e30c63d869655d0897b8558225bf9b3","observation_id":"c168e277-342d-4817-9a93-37fe007677cd","resolution":{"observed_at":"2026-08-15T23:50:44.025383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-15T23:50:44.029059Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.029059Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:465025f6ef080cdbaf3da2fe7e038f8e10a4948442f7828b0121d70755649fe7","observation_id":"c8f94f6c-05ae-411a-8abb-bfb10691e8ad","resolution":{"observed_at":"2026-08-15T23:50:44.029059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.11636","last_updated":"2023-02-22T20:24:35Z","snapshot_observed_at":"2026-08-13T12:39:14.838051Z","submitted_at":"2023-02-22T20:24:35Z","title":"Do We Really Need Complicated Model Architectures For Temporal Networks?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.11636","snapshot_observed_at":"2026-08-15T23:50:44.032856Z","title":"Do we really need complicated model architectures for temporal networks?","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.032856Z"},"links":{"cited_paper":"/paper/2302.11636","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:55d5463fd787e474afa806402643a3c56b7dcfd3dd4041015edf360acb819ec2","observation_id":"85b94c4e-2a7b-44a0-9418-7c43271d08d7","resolution":{"observed_at":"2026-08-15T23:50:44.032856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.036476Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.036476Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:7b17fd606fdeed22bf4d7eb061fdfdd1e26e6a741fc67b19177b232c875549c4","observation_id":"fc2ae432-b86d-49e9-82a7-11240a495aa5","resolution":{"observed_at":"2026-08-15T23:50:44.036476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10698","last_updated":"2024-07-21T00:42:15Z","snapshot_observed_at":"2026-08-13T04:17:33.815269Z","submitted_at":"2024-02-16T13:59:07Z","title":"Question-Instructed Visual Descriptions for Zero-Shot Video Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10698","snapshot_observed_at":"2026-08-15T23:50:44.040102Z","title":"Question-instructed visual descrip- tions for zero-shot video question answering,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.040102Z"},"links":{"cited_paper":"/paper/2402.10698","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:3d6f351906abc7fd5ceac16973e4ba107b669cdf885f6ffc38e4954c16fc5db4","observation_id":"ea23b79a-d636-4a5f-9585-088fc0a12db6","resolution":{"observed_at":"2026-08-15T23:50:44.040102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.491931Z","title":"Mist: Multi-modal iterative spatial-temporal transformer for long-form video question answering,","venue":null,"work_id":"59b9200d-f72e-429d-a0fe-c74849105422","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.044120Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:d36bfb70c738d322b480e872fe31485263e2417ec154c1053fab305c6dd108eb","observation_id":"f3d640f6-5953-462d-85d4-55aaab99866f","resolution":{"observed_at":"2026-08-15T23:50:44.495603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.482131Z","title":"Self-chained image-language model for video localization and question answering,","venue":null,"work_id":"91c3f047-6d9d-47ad-8570-74071b190ffb","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.047370Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:215be84895e101090b8681b3008efd6fb5f2b686f2385a583a2e1027b26e187e","observation_id":"41a135df-9a6f-469b-b794-3ed9cdc277a5","resolution":{"observed_at":"2026-08-15T23:50:44.485545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.472297Z","title":"Vila: Efficient video-language alignment for video question answering,","venue":null,"work_id":"516e9211-7cd3-4b75-adc6-84796e000e01","year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.051426Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:260f7b5bd331f8fbae5ca787941f431ae0d4f2db8935787d721254bcb169d595","observation_id":"219cc9f7-2e91-40e7-a4f4-2e2d071e4456","resolution":{"observed_at":"2026-08-15T23:50:44.476300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15047","last_updated":"2024-07-23T14:56:22Z","snapshot_observed_at":"2026-08-12T23:17:56.918683Z","submitted_at":"2024-07-21T04:09:37Z","title":"End-to-End Video Question Answering with Frame Scoring Mechanisms and Adaptive Sampling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15047","snapshot_observed_at":"2026-08-15T23:50:44.054394Z","title":"End- to-end video question answering with frame scoring mechanisms and adaptive sampling,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.054394Z"},"links":{"cited_paper":"/paper/2407.15047","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:afdaa4474ad72e10e97a7dacaffa822cd396dc0e2e64d55b76fe65c59ebb16cb","observation_id":"17ca962d-91c5-44f7-ae3e-ab8089b6aac4","resolution":{"observed_at":"2026-08-15T23:50:44.054394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.17778","last_updated":"2024-01-22T00:54:30Z","snapshot_observed_at":"2026-08-13T11:05:11.716488Z","submitted_at":"2023-06-30T16:31:14Z","title":"Look, Remember and Reason: Grounded reasoning in videos with language models","version":3},"cited_work":{"arxiv_id":"2306.17778","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.17778","snapshot_observed_at":"2026-08-15T23:50:44.154476Z","title":"Look, Remember and Reason: Grounded reasoning in videos with language models","venue":"cs.CV","work_id":"6ae677c9-b8db-427a-82b0-8df004e71e7b","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.057798Z"},"links":{"cited_paper":"/paper/2306.17778","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:1e8a22c796f5e3be512584bdf5639389012c5939b63f2f37f631399ce11999a0","observation_id":"9c57fd8c-d894-4d96-baa6-3b39df3c6587","resolution":{"observed_at":"2026-08-15T23:50:44.244898Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.462389Z","title":"Glance and focus: Memory prompt- ing for multi-event video question answering,","venue":null,"work_id":"df8b41b3-24b6-4b7a-a8d8-969015599bef","year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.060839Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:3a6f7f87f51c05a89d01bf732c8d8df646159a6d91c6ed77439167c7ffa9dd8b","observation_id":"9fe86359-9402-4674-81ae-90c404ae6f0a","resolution":{"observed_at":"2026-08-15T23:50:44.465775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.452359Z","title":"Learning to reason iteratively and parallelly for complex visual reasoning scenarios,","venue":null,"work_id":"4eaa3226-fa95-45f8-a704-13ff75331aac","year":2025},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.064372Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:af9c606747ec2c2e737fd99fdf9bfe0fd9b9f43c552cb34e98a76a0069862bd0","observation_id":"204bab2a-2e42-4642-97bb-685fc96fff48","resolution":{"observed_at":"2026-08-15T23:50:44.456171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.16050","last_updated":"2024-10-03T09:24:56Z","snapshot_observed_at":"2026-08-13T04:10:40.413876Z","submitted_at":"2024-02-25T10:27:46Z","title":"Efficient Temporal Extrapolation of Multimodal Large Language Models with Temporal Grounding Bridge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.16050","snapshot_observed_at":"2026-08-15T23:50:44.067588Z","title":"Efficient temporal extrapolation of multimodal large language models with temporal grounding bridge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.067588Z"},"links":{"cited_paper":"/paper/2402.16050","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:0d9b63ccf515051e731739ffa87fbbd51386bdb0c3dabe15d20b828ab0902c44","observation_id":"7e94c167-1768-4650-bcc6-10a6cde4d138","resolution":{"observed_at":"2026-08-15T23:50:44.067588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03941","last_updated":"2022-10-08T07:03:31Z","snapshot_observed_at":"2026-08-13T14:10:06.537016Z","submitted_at":"2022-10-08T07:03:31Z","title":"Learning Fine-Grained Visual Understanding for Video Question Answering via Decoupling Spatial-Temporal Modeling","version":1},"cited_work":{"arxiv_id":"2210.03941","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.03941","snapshot_observed_at":"2026-08-15T23:50:44.127936Z","title":"Learning Fine-Grained Visual Understanding for Video Question Answering via Decoupling Spatial-Temporal Modeling","venue":"cs.CV","work_id":"a8096dd3-4b9e-4a75-a3f8-0611e545b3c2","year":2022},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.071264Z"},"links":{"cited_paper":"/paper/2210.03941","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:58f0a94ac9ae2f0f8a7c592fd8d6f622448106d40ec3754e7ff61150c52f6b17","observation_id":"fec0ce2d-c820-44ef-b6b5-a7500e15d60f","resolution":{"observed_at":"2026-08-15T23:50:44.134813Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07630","last_updated":"2024-05-27T04:04:40Z","snapshot_observed_at":"2026-08-15T13:52:48.582261Z","submitted_at":"2024-02-12T13:13:04Z","title":"G-Retriever: Retrieval-Augmented Generation for Textual Graph Understanding and Question Answering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07630","snapshot_observed_at":"2026-08-15T23:50:44.074558Z","title":"G-retriever: Retrieval-augmented generation for textual graph understanding and question answering,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.074558Z"},"links":{"cited_paper":"/paper/2402.07630","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:2bb1ef66b32ae6e336524a549e85f802da92289636fb17e3d66748a03c74193f","observation_id":"e527ce8f-8f68-4236-8d92-491033c3c6e2","resolution":{"observed_at":"2026-08-15T23:50:44.074558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:50:44.441095Z","title":"A note on the prize collecting traveling salesman problem,","venue":null,"work_id":"21500474-8afc-4e46-ac53-173aba94ab26","year":1993},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.078054Z"},"links":{"citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:47db24d4793e57524bc1aa4d1bdda71bebfcdf14aff145733ea2b1345fc5a7ad","observation_id":"49832ac9-9274-4e89-9d63-778c489cd185","resolution":{"observed_at":"2026-08-15T23:50:44.445522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17497","last_updated":"2023-06-01T04:56:26Z","snapshot_observed_at":"2026-08-13T11:31:46.973682Z","submitted_at":"2023-05-27T15:38:31Z","title":"FACTUAL: A Benchmark for Faithful and Consistent Textual Scene Graph Parsing","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.17497","snapshot_observed_at":"2026-08-15T23:50:44.081050Z","title":"Factual: A benchmark for faithful and consistent textual scene graph parsing,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:44.081050Z"},"links":{"cited_paper":"/paper/2305.17497","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:8cf8c814ee0f514afc5aa75cd658828356f550a1bd5ac74531c1f58653b161a3","observation_id":"825f9578-8f73-459c-9c8b-16f2f5cc1acb","resolution":{"observed_at":"2026-08-15T23:50:44.081050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.06105","last_updated":"2022-04-12T22:30:12Z","snapshot_observed_at":"2026-08-13T16:03:20.415943Z","submitted_at":"2022-04-12T22:30:12Z","title":"AGQA 2.0: An Updated Benchmark for Compositional Spatio-Temporal Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.06105","snapshot_observed_at":"2026-08-15T23:50:43.919104Z","title":"Available: https://arxiv.org/abs/2204.06105","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-15T23:50:43.919104Z"},"links":{"cited_paper":"/paper/2204.06105","citing_paper":"/paper/2505.03581"},"observation_digest":"sha256:5afe357b33b6ef0a5875d1cc5123e886ee77af8d5fe28ecc98eb30d05ff89c55","observation_id":"c137e4b7-3635-4d63-91ba-aceac1a5e0ed","resolution":{"observed_at":"2026-08-15T23:50:43.919104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.03581","last_updated":"2025-05-06T14:41:42Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T23:45:47.425022Z","submitted_at":"2025-05-06T14:41:42Z","title":"DyGEnc: Encoding a Sequence of Textual Scene Graphs to Reason and Answer Questions in Dynamic Scenes"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":36,"verified_exact":3,"verified_fuzzy":21},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 0 inbound Pith citation observations for arXiv:2505.03581."}