{"as_of":"2026-08-09T06:05:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:817e58ec5e5d0a4ff0cb91abb1c7d04fa5c7c401799e2d4c33744a8e8f75fd61","coverage":[{"denominator":26,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:50:43.557595Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.20550/citation-record","integrity":"/paper/2506.20550/integrity","json":"/paper/2506.20550/citation-record.json","paper":"/paper/2506.20550"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:40.128704Z","title":"YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:40.128704Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:a020c5408813ed737fc43bc949b214aea542947ab28e2d0c7764b8fdbfda4bd6","observation_id":"819cf284-94c8-4197-9474-47130028480b","resolution":{"observed_at":"2026-08-06T22:50:40.128704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:45.973618Z","title":"Recurrent neural networks for video object detection,","venue":null,"work_id":"4f240692-ee3e-47e3-880a-4d14adde102c","year":2020},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:40.196333Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:84c7fc7c5643bbffc4283ab8eb6c0317382f9b79c07c146642686e011134f688","observation_id":"d675ffdc-ce47-4dc7-94cf-d828bbde0c24","resolution":{"observed_at":"2026-08-06T22:50:45.977906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:45.958469Z","title":"Flow-guided feature aggregation for video object detection,","venue":null,"work_id":"e2885a02-1a59-4569-860e-a98079858f84","year":2017},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:40.366268Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:bc6c71ffbde4c7648db912b5f9f86f63698ac1f6af727800d3b8257e4de8e1aa","observation_id":"7a2d0391-6822-4c50-ac39-10360c64c210","resolution":{"observed_at":"2026-08-06T22:50:45.963750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:45.945649Z","title":"Sequence level seman- tics aggregation for video object detection,","venue":null,"work_id":"cfe977e1-b2b9-412c-8299-e5f9fb20c2c2","year":2019},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:40.535850Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:7c9355677f27183d087ad349c2a944a42b721c4209017f283a6c8d2eb20f724d","observation_id":"5f51163e-2f4e-49bb-a51c-149cb56211ce","resolution":{"observed_at":"2026-08-06T22:50:45.949297Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:40.676193Z","title":"Slowfast networks for video recognition,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:40.676193Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:581a322804cb73cf73bc6cad81430148ff00e1a78b5ba4e1c079cb0027cbaf59","observation_id":"aae06bcf-f8d0-49da-a300-b78902796f9c","resolution":{"observed_at":"2026-08-06T22:50:40.676193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:40.812950Z","title":"A brief introduction to weakly supervised learning,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:40.812950Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:b6abd18cb0bb56b5bfb57ea5b2c4f047f053110f23dc291370e8e3440cca0c2a","observation_id":"2e603338-f870-43fd-b7cc-3e32a8c12652","resolution":{"observed_at":"2026-08-06T22:50:40.812950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2003.09003","last_updated":"2020-03-19T20:08:24Z","snapshot_observed_at":"2026-08-01T23:12:45.187111Z","submitted_at":"2020-03-19T20:08:24Z","title":"MOT20: A benchmark for multi object tracking in crowded scenes","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2003.09003","snapshot_observed_at":"2026-08-06T22:50:41.211905Z","title":"Mot20: A bench- mark for multi object tracking in crowded scenes,","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:41.211905Z"},"links":{"cited_paper":"/paper/2003.09003","citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:455ddc22a1ad8b3ece288c8c57927e065486ab3d8908639c89ab7f2b3ba68e50","observation_id":"f543736c-604c-4501-845b-1d477d804e65","resolution":{"observed_at":"2026-08-06T22:50:41.211905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:45.735619Z","title":"Faster r-cnn: Towards real- time object detection with region proposal networks,","venue":null,"work_id":"5dff3aee-ffd3-474c-bf9c-6c30a98b2f0b","year":2015},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:41.851333Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:127a7006915c77c08cd977665baf44fc64389900e1ebef86800f4ec40d999eb3","observation_id":"d3cd499d-a79e-4c78-86a9-3f3421c1317e","resolution":{"observed_at":"2026-08-06T22:50:45.881151Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:45.409115Z","title":"You only look once: Unified, real-time object detection,","venue":null,"work_id":"aa19ef71-f1f8-43cc-8b15-ec83e3d79bf4","year":2016},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:42.048303Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:318267e8867cc7f48bafedcc1f1e27f4d4b8f390d207d4ead4e02d41ce7c74bb","observation_id":"8bcca8c7-d2a2-4ae9-8915-d5c1d56c94af","resolution":{"observed_at":"2026-08-06T22:50:45.577983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:45.170623Z","title":"Focal loss for dense object detection,","venue":null,"work_id":"56068bd8-8539-42c0-9387-41fd47912dae","year":2017},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:42.189727Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:62bd5a1676d7229c82b5d5d2c9f531fa7f223b97f96ed105556b4b4746899880","observation_id":"aac0882c-a9c4-436a-81a8-b3f5b714930f","resolution":{"observed_at":"2026-08-06T22:50:45.266270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:45.044344Z","title":"A detailed study of the association task in tracking-by- detection-based multi-person tracking,","venue":null,"work_id":"f9f2a7af-3925-4105-a540-f84d52bca981","year":2022},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:42.340216Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:cfba06ea8b43d752643e362c83a942422af8be52b680df5d5b58a02913abc15e","observation_id":"6e3995a0-7f80-4580-926d-c92be1bb7f55","resolution":{"observed_at":"2026-08-06T22:50:45.091877Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:44.891077Z","title":"Video object detection with an aligned spatial-temporal memory,","venue":null,"work_id":"f5c53aa5-2aba-4931-b2cf-82efa6778a86","year":2020},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:42.394520Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:67ae93b27560a1c4177449775feca34b0b42a681b2f05b7063b533e3199bdfe8","observation_id":"c8bde5fb-6667-432d-84d3-43eb09eb114f","resolution":{"observed_at":"2026-08-06T22:50:44.961372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:44.740057Z","title":"Learning recurrent memory activation networks for visual tracking,","venue":null,"work_id":"b6df4889-8b24-43ec-a182-fbab971969d7","year":2020},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:42.505935Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:c122e79a5a03bb2fd83d3d3edb44ef343e32d7eb8dbaaea77714c567e3159309","observation_id":"a184505b-27c6-4736-8ccb-45cd45e4f3b8","resolution":{"observed_at":"2026-08-06T22:50:44.800024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:44.618370Z","title":"Video visual relation detection via 3d convolutional neural network,","venue":null,"work_id":"94268be8-837e-4ff3-b720-b59d24c76deb","year":2022},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:42.616257Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:57cea47d85455ea5c77b2d2262beae8116b20183a143520e05e50f7a5bffc615","observation_id":"92a80f96-5f9d-4ec5-a1c4-0e24d7b3bb36","resolution":{"observed_at":"2026-08-06T22:50:44.670775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.08895","last_updated":"2019-07-21T02:05:38Z","snapshot_observed_at":"2026-07-06T08:09:05.455438Z","submitted_at":"2019-07-21T02:05:38Z","title":"An Efficient 3D CNN for Action/Object Segmentation in Video","version":1},"cited_work":{"arxiv_id":"1907.08895","doi":null,"metadata_source":"pith","pith_arxiv_id":"1907.08895","snapshot_observed_at":"2026-08-06T22:50:43.656350Z","title":"An Efficient 3D CNN for Action/Object Segmentation in Video","venue":"cs.CV","work_id":"9333614c-5b4d-4a80-a904-741e2bdb5c93","year":2019},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:42.721245Z"},"links":{"cited_paper":"/paper/1907.08895","citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:2728271a7542283d42dab3ef34182ae0b92aef080178186437ff6d8d2b7ba8ed","observation_id":"95f5e276-5813-43e9-815a-5a3f918887ad","resolution":{"observed_at":"2026-08-06T22:50:43.708852Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:44.463995Z","title":"New generation deep learning for video object detection: A survey,","venue":null,"work_id":"4502084f-6537-4e24-a705-9530294b501f","year":2021},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:42.837489Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:db3c29027205ea2312f80d6d446334d1a2c124a26ef312ba55982185ca67e90d","observation_id":"78cf56d1-e896-4a85-8bfb-0d187447940d","resolution":{"observed_at":"2026-08-06T22:50:44.525546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:44.290019Z","title":"Label- efficient online continual object detection in streaming video,","venue":null,"work_id":"d9108093-74d4-4186-9101-f3ab3abd2ad5","year":2023},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:42.896913Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:581ff8f070a1b42d35a5ee7cdc9ba2eaaa62c9eb1c5b2c1f8c5634bfa9471d84","observation_id":"b89d4ad5-65cc-4e4c-a04f-d3ad2dfaf949","resolution":{"observed_at":"2026-08-06T22:50:44.368875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:44.144158Z","title":"A review of video object detection: Datasets, metrics and methods,","venue":null,"work_id":"2aa379b7-2fee-4ddf-932a-acbc600e8042","year":2020},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:42.940042Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:722f3623a4b949abee39915be94fbcfe707b8fd07dc7a99ee6e2de5c3992d030","observation_id":"e834f025-55b6-486d-a7c4-ceeda7efb76a","resolution":{"observed_at":"2026-08-06T22:50:44.195751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:44.018505Z","title":"Yolov: Making still image object detectors great at video object detection,","venue":null,"work_id":"8c19705c-6ff5-4340-89a9-ade6465eda8c","year":2023},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:43.010015Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:ddf01ecbfe0ec89d9fce221b56570dcf9035897f21c65cf1c2451dedba77227e","observation_id":"f2c17c79-b7af-4a9c-9df3-2dc4c0f6a602","resolution":{"observed_at":"2026-08-06T22:50:44.098627Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:43.861392Z","title":"Seadronessee: A maritime benchmark for detecting humans in open water,","venue":null,"work_id":"7cb46b21-ec3c-4c8f-8e39-e9e85edaab7d","year":2022},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:43.046876Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:146dcc6e37174443f746aa99e713fb9883c02c602b85f2688dc2ab52f75383a4","observation_id":"8c45a11b-ce2c-4d9d-b860-161963fdda8c","resolution":{"observed_at":"2026-08-06T22:50:43.926569Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.14762","last_updated":"2023-11-23T21:01:14Z","snapshot_observed_at":"2026-07-06T16:52:07.175998Z","submitted_at":"2023-11-23T21:01:14Z","title":"The 2nd Workshop on Maritime Computer Vision (MaCVi) 2024","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.14762","snapshot_observed_at":"2026-08-06T22:50:43.153310Z","title":"The 2nd workshop on maritime computer vision (macvi) 2024,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:43.153310Z"},"links":{"cited_paper":"/paper/2311.14762","citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:5680695a513b66484e34a991b86e71a5e74051b4757e940f4ddcbbacd2fa4320","observation_id":"e55fc948-bcba-4147-aaee-ec7062ebcbee","resolution":{"observed_at":"2026-08-06T22:50:43.153310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:43.245635Z","title":"Grad-cam: Visual explanations from deep networks via gradient-based localization,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:43.245635Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:e95de3083890af62274081d6c5438e647aae6b22506695c87a674e57499d3609","observation_id":"aeb32941-9aeb-4d57-9a83-44795dc8bf24","resolution":{"observed_at":"2026-08-06T22:50:43.245635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:43.305234Z","title":"Grad-cam++: Generalized gradient-based visual explanations for deep convolutional networks,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:43.305234Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:f2d1b2742b8c6941b5620621fbc4fe20a15c19eefc533dcabb9ea665c3c1b69d","observation_id":"299c51e6-379c-4aff-acb1-8f1316e73b23","resolution":{"observed_at":"2026-08-06T22:50:43.305234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:43.369754Z","title":"Eigen-cam: Class activation map using principal components,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:43.369754Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:2f29de165cfc8d2df6afb8c58e2f503211f9be1833d6e6c869404ae6f1711e67","observation_id":"5263cb3c-6fb4-4250-80cf-b4fd45eaecf9","resolution":{"observed_at":"2026-08-06T22:50:43.369754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:43.468469Z","title":"Imagenet classification with deep convolutional neural networks,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:43.468469Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:71cb3129029d2ace4a642ddda9321189c883e2099fc216bc2ea2b9280eabc038","observation_id":"ef6b1acd-d479-47ce-bead-59397ff12127","resolution":{"observed_at":"2026-08-06T22:50:43.468469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:50:43.557595Z","title":"Microsoft coco: Common objects in context,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:43.557595Z"},"links":{"citing_paper":"/paper/2506.20550"},"observation_digest":"sha256:ae372b20039d10cb68f5ac8d5f664517b89b47d8a2f7819d743cc2cab1938fda","observation_id":"76791451-a0dc-42d6-a6cd-5ca0fc1fd5b0","resolution":{"observed_at":"2026-08-06T22:50:43.557595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.20550","last_updated":"2025-06-25T15:49:07Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T22:43:47.904845Z","submitted_at":"2025-06-25T15:49:07Z","title":"Lightweight Multi-Frame Integration for Robust YOLO Object Detection in Videos"},"reference_resolution":{"displayed":26,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":1,"verified_fuzzy":15},"total_outbound_references":26},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 26 of 26 outbound references and 0 inbound Pith citation observations for arXiv:2506.20550."}