{"as_of":"2026-08-05T14:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:98d5b74df876de37fd69d7b0c98f038a0b84c1d94e1e25ccec0ab9c6fb5211fb","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":27,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T05:28:44.443964Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T02:04:26.310872Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"1907.06119","last_updated":"2019-07-13T19:23:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-13T19:23:42Z","title":"Understanding Deep Learning Techniques for Image Segmentation","version":1},"reference_index":209,"source":"pdf_text","source_observed_at":"2026-05-24T21:46:17.736097Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/1907.06119"},"observation_digest":"sha256:7de1068f59354e21af22fa7ad3fa516baa9e7a075080f823e41ea7240cbb2a66","observation_id":"2837349d-c4e8-4805-a500-0a3923168857","resolution":{"observed_at":"2026-05-24T21:46:24.412506Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T13:56:25.331304Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2408.00714"},"observation_digest":"sha256:624c9cc00e40e4e980ed5cd927f894eb18b72d4cefa3ddcd04e7a86426421b0a","observation_id":"04a1d9c5-c254-41c2-a2bf-a600c88c39cf","resolution":{"observed_at":"2026-05-10T13:56:25.466836Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2504.12169","last_updated":"2025-08-07T14:22:33Z","snapshot_observed_at":"2026-08-04T14:57:04.485762Z","submitted_at":"2025-04-16T15:19:11Z","title":"Towards a General-Purpose Zero-Shot Synthetic Low-Light Image and Video Pipeline","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-22T20:18:35.764575Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2504.12169"},"observation_digest":"sha256:1dbd4df0e38beb4c648cef8ea8de8347b41ee24bbc297cc1da2f3d0bb971119b","observation_id":"2acf683e-3971-46fc-aa0c-9921b207226e","resolution":{"observed_at":"2026-05-22T20:22:03.834071Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2506.05425","last_updated":"2026-04-28T02:01:09Z","snapshot_observed_at":"2026-07-31T07:37:26.215945Z","submitted_at":"2025-06-05T05:51:35Z","title":"SIV-Bench: A Video Benchmark for Social Interaction Understanding and Reasoning","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-19T11:36:36.687324Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2506.05425"},"observation_digest":"sha256:99a506cec373bc78252755be61b7def632c36027953df80db58915d82999972e","observation_id":"828246b3-e0a2-46a8-9800-d6a586a05c89","resolution":{"observed_at":"2026-05-19T11:37:15.726369Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-08-05T05:28:44.443964Z","title":"Youtube-vos: A large-scale video object segmentation benchmark.arXiv preprint arXiv:1809.03327, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.05297","last_updated":"2025-09-05T17:59:59Z","snapshot_observed_at":"2026-08-05T05:28:30.838132Z","submitted_at":"2025-09-05T17:59:59Z","title":"FlowSeek: Optical Flow Made Easier with Depth Foundation Models and Motion Bases","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-05T05:28:44.443964Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2509.05297"},"observation_digest":"sha256:fbf4cb4f04acd86f7ff0793a17969728e47e64032c8a51e82c561ae6c28eb054","observation_id":"71a37c2a-352c-4fdc-a98e-6d621cf2c18d","resolution":{"observed_at":"2026-08-05T05:28:44.443964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2510.18822","last_updated":"2026-05-18T02:07:26Z","snapshot_observed_at":"2026-08-03T04:29:55.338053Z","submitted_at":"2025-10-21T17:20:15Z","title":"SAM 2++: Tracking Anything at Any Granularity","version":4},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-21T19:51:56.047515Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2510.18822"},"observation_digest":"sha256:03247c8767201af9b3a48ea373a75d85deb26b2dcb808467dc39e29724e44d80","observation_id":"d0042d06-6297-4269-8993-09d056fa8d71","resolution":{"observed_at":"2026-05-21T19:54:20.151923Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2511.16719","last_updated":"2026-03-28T16:54:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-20T18:59:56Z","title":"SAM 3: Segment Anything with Concepts","version":2},"reference_index":145,"source":"arxiv_source","source_observed_at":"2026-05-17T20:22:46.220021Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2511.16719"},"observation_digest":"sha256:76e7cf927ccc981dfe71c94c18f949f8d1ced113738f42f6fd8b4c0ff48de9f5","observation_id":"a3a3a651-84f5-40b2-8a68-ded2354c8081","resolution":{"observed_at":"2026-05-17T20:25:11.536644Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2512.13684","last_updated":"2026-04-21T13:13:49Z","snapshot_observed_at":"2026-07-06T22:39:04.164003Z","submitted_at":"2025-12-15T18:59:48Z","title":"Recurrent Video Masked Autoencoders","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-16T21:55:42.555679Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2512.13684"},"observation_digest":"sha256:88acf8a7d414a5ff7d0e44a7d5e4431cc325672c67f819706f7bad9e3ca546fa","observation_id":"54a755de-dd47-4e30-9ba1-9fcb5002d239","resolution":{"observed_at":"2026-05-16T21:58:35.816176Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2512.22046","last_updated":"2026-04-30T10:55:14Z","snapshot_observed_at":"2026-07-06T22:40:07.339525Z","submitted_at":"2025-12-26T14:48:58Z","title":"Backdoor Attacks on Prompt-Driven Video Segmentation Foundation Models","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-16T19:08:37.460895Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2512.22046"},"observation_digest":"sha256:14bca602ebf35c7db22cb366de7a4664d0a7e80919e9c0869ee6e705e8c76f69","observation_id":"ccd8cf70-8809-4803-b303-04f07816ea3f","resolution":{"observed_at":"2026-05-16T19:11:11.665827Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2601.08831","last_updated":"2026-04-16T12:58:36Z","snapshot_observed_at":"2026-07-06T22:41:39.161087Z","submitted_at":"2026-01-13T18:59:54Z","title":"3AM: 3egment Anything with Geometric Consistency in Videos","version":5},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-16T14:19:34.641108Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2601.08831"},"observation_digest":"sha256:579d085af1bdc182c21d864fa4cd25fa3949870532a16fa7c04bde71f36ebdff","observation_id":"a3327468-b513-46c3-aa64-ded663dcb682","resolution":{"observed_at":"2026-05-16T14:21:01.642414Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-13T16:08:30.031334Z","title":"Youtube-VOS: A large-scale video object segmentation benchmark","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2603.28759","last_updated":"2026-05-31T17:37:40Z","snapshot_observed_at":"2026-07-13T16:08:29.518581Z","submitted_at":"2026-03-30T17:58:12Z","title":"FlowIt: Global Matching via Hierarchical Transformers and Optimal Transport for Optical Flow","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-07-13T16:08:30.031334Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2603.28759"},"observation_digest":"sha256:59a8b9a28048fbde2ba1ac832b8076d2417dc462717858daea9b1f6d0961ffcd","observation_id":"5d7c2b68-6663-4bd6-9907-afd4a0d487fc","resolution":{"observed_at":"2026-07-13T16:08:30.031334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2604.07901","last_updated":"2026-04-09T07:17:47Z","snapshot_observed_at":"2026-07-06T22:57:09.435331Z","submitted_at":"2026-04-09T07:17:47Z","title":"PanoSAM2: Lightweight Distortion- and Memory-aware Adaptions of SAM2 for 360 Video Object Segmentation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T18:27:39.914591Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2604.07901"},"observation_digest":"sha256:04f92aee78963f138a2003a7d6cee9dcead1aa688b8daf1c7f86ab086ea470c9","observation_id":"fea0e949-68d5-4197-bd0c-53d3439aed35","resolution":{"observed_at":"2026-05-11T00:30:55.265816Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2604.14630","last_updated":"2026-04-16T05:14:32Z","snapshot_observed_at":"2026-07-06T23:02:22.790426Z","submitted_at":"2026-04-16T05:14:32Z","title":"CMTM: Cross-Modal Token Modulation for Unsupervised Video Object Segmentation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T12:32:11.911775Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2604.14630"},"observation_digest":"sha256:544b42d80094402d9c870e4f47fcb355687e56a9721274d08ad7ad799079899c","observation_id":"1b41aa70-6c34-4efd-b843-4fb0c7de53ad","resolution":{"observed_at":"2026-05-11T11:51:01.038342Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2604.26488","last_updated":"2026-04-29T09:51:56Z","snapshot_observed_at":"2026-07-06T23:12:06.349884Z","submitted_at":"2026-04-29T09:51:56Z","title":"Featurising Pixels from Dynamic 3D Scenes with Linear In-Context Learners","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-07T13:37:39.954299Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2604.26488"},"observation_digest":"sha256:e1badff48b13b73d4f3cb218d294e535de5e45450b364904696c51947a391862","observation_id":"6c9c2355-208c-4954-b29e-859f2564987a","resolution":{"observed_at":"2026-05-12T08:51:24.798782Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2604.27322","last_updated":"2026-04-30T02:08:13Z","snapshot_observed_at":"2026-07-06T23:12:47.745700Z","submitted_at":"2026-04-30T02:08:13Z","title":"YOSE: You Only Select Essential Tokens for Efficient DiT-based Video Object Removal","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-07T09:12:13.460169Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2604.27322"},"observation_digest":"sha256:e7cda458e6d09708f1d23caa5e48eddcd4432b9f72cb09298479bd135bcbb4f2","observation_id":"060bc62b-c9bf-4756-a4a0-2f6b6e69e453","resolution":{"observed_at":"2026-05-12T09:46:27.935892Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:b0b7de6414d8f484fec14299f85d5973b9719be6de6ed595c5e583bab56c17ef","observation_id":"2cb0fd27-26fd-4a0c-a948-00e046bcd75b","resolution":{"observed_at":"2026-05-11T15:01:05.428048Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2605.03276","last_updated":"2026-05-08T23:14:34Z","snapshot_observed_at":"2026-08-02T06:50:49.392802Z","submitted_at":"2026-05-05T02:05:27Z","title":"VEBench:Benchmarking Large Multimodal Models for Real-World Video Editing","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-08T01:30:19.531699Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2605.03276"},"observation_digest":"sha256:5263984e2fb1d645049eef623181e6797b7bef9f4e49755a56b5b050688708cf","observation_id":"8dc0410c-3933-4020-96a4-8c8d736e2788","resolution":{"observed_at":"2026-05-11T23:06:22.357494Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2605.03276","last_updated":"2026-05-08T23:14:34Z","snapshot_observed_at":"2026-08-02T06:50:49.392802Z","submitted_at":"2026-05-05T02:05:27Z","title":"VEBench:Benchmarking Large Multimodal Models for Real-World Video Editing","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-12T01:43:34.898639Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2605.03276"},"observation_digest":"sha256:fb564a01e579422f2cefc762f98820592a6f5967d48126228ba885dc6c08a6e6","observation_id":"19766d70-a65c-4db4-90d9-718cd387a5f0","resolution":{"observed_at":"2026-05-12T01:46:13.949720Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2605.12006","last_updated":"2026-05-12T11:55:31Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:55:31Z","title":"Robust Promptable Video Object Segmentation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-13T07:18:20.281383Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2605.12006"},"observation_digest":"sha256:e679370507aec3de6bb2ee616ccfd4115bbebd78411c5b3fe910218793f242e2","observation_id":"73302bf2-b071-4a48-9048-9338e3dbac4c","resolution":{"observed_at":"2026-05-13T07:22:29.355364Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2605.18010","last_updated":"2026-05-18T08:05:07Z","snapshot_observed_at":"2026-08-02T12:16:21.427153Z","submitted_at":"2026-05-18T08:05:07Z","title":"Functionalization via Structure Completion and Motion Rectification","version":1},"reference_index":281,"source":"arxiv_source","source_observed_at":"2026-05-20T12:25:07.157086Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2605.18010"},"observation_digest":"sha256:3231892531532b3343d4d18cc41861fc3705f7b976285f8c1a76afc41c731120","observation_id":"f75d5859-c275-486b-a021-7cdec648e0e4","resolution":{"observed_at":"2026-05-20T12:28:17.069803Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2606.00522","last_updated":"2026-06-04T10:14:10Z","snapshot_observed_at":"2026-07-06T23:41:10.825760Z","submitted_at":"2026-05-30T04:29:52Z","title":"A Trajectory-Driven Spatio-Temporal Refinement Solution for CVPR 2026 8th UG2+ Challenge Track 3: DOST","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T18:55:17.222924Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2606.00522"},"observation_digest":"sha256:6be1d581f3cc4e0b283c701afd7844f5c2a310891ae6864bebfd24ba5b66df04","observation_id":"4ec12bb2-e568-476f-aa53-3df3d84aa9b5","resolution":{"observed_at":"2026-06-28T19:42:35.941042Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2607.00902","last_updated":"2026-07-01T13:04:35Z","snapshot_observed_at":"2026-07-07T00:06:29.765578Z","submitted_at":"2026-07-01T13:04:35Z","title":"MG-RWKV: Multi-Grained Context-Aware RWKV for Temporal Forgery Localization","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-07-02T14:03:46.585001Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2607.00902"},"observation_digest":"sha256:38eb212104ac40ddb1abce2dc88e9e98d33e37297566d183ce6df112703cdb5b","observation_id":"93fef265-98ed-4c22-85ce-b1e4f2291e93","resolution":{"observed_at":"2026-07-02T14:07:02.278608Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-12T06:06:47.233814Z","title":"arXiv preprint arXiv:1809.03327 (2018)","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.02922","last_updated":"2026-07-03T03:28:25Z","snapshot_observed_at":"2026-08-04T16:29:40.309150Z","submitted_at":"2026-07-03T03:28:25Z","title":"STAC: Selective Spatiotemporal Aggregation and Compression for Video Reasoning Segmentation","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-07-12T06:06:47.233814Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2607.02922"},"observation_digest":"sha256:6681c3b719c26709462d7f1f5892c8ce55a8ef5cc556fa6b790f5ee9c0eaf7dd","observation_id":"ba25135d-c025-4de1-b363-f149987afac8","resolution":{"observed_at":"2026-07-12T06:06:47.233814Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2607.05247","last_updated":"2026-07-06T15:56:14Z","snapshot_observed_at":"2026-08-01T19:48:27.981186Z","submitted_at":"2026-07-06T15:56:14Z","title":"Vision Pretraining for Dense Spatial Perception","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-07-07T22:20:17.456484Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2607.05247"},"observation_digest":"sha256:8d90387e7ad382f90be7f5072cf160ea45eb781397c3136d28e7aaa09afca07f","observation_id":"5faf9b26-e55b-462f-b2cc-1f560eaf1362","resolution":{"observed_at":"2026-07-07T22:24:11.145551Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2607.06560","last_updated":"2026-07-07T17:58:33Z","snapshot_observed_at":"2026-07-10T23:18:28.655344Z","submitted_at":"2026-07-07T17:58:33Z","title":"Vision as Unified Multimodal Generation","version":1},"reference_index":197,"source":"pdf_text","source_observed_at":"2026-07-08T01:54:30.649092Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2607.06560"},"observation_digest":"sha256:f1e0afc2910b4290f1f299c21d45ab0db2d144f7b57f677161ab9ed02f7cffd0","observation_id":"9c3047a0-4581-474b-adb6-3e7fdf800106","resolution":{"observed_at":"2026-07-08T02:04:26.312361Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-08-01T10:02:03.252658Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.20389","last_updated":"2026-07-22T17:17:33Z","snapshot_observed_at":"2026-08-03T13:35:05.079480Z","submitted_at":"2026-07-22T17:17:33Z","title":"PercepCap: Video Captioner with Structured Spatio-Temporal Perception","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-01T10:02:03.252658Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2607.20389"},"observation_digest":"sha256:8477655448d6bc800c4cb0bd0302af4613969963457b4efe5af4989399d610d6","observation_id":"f6a1186f-f177-43b6-91e8-07412c4b0314","resolution":{"observed_at":"2026-08-01T10:02:03.252658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-08-01T07:07:43.997932Z","title":"YouTube-VOS: A large-scale video object segmentation benchmark.arXiv preprint arXiv:1809.03327, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.21576","last_updated":"2026-07-23T17:55:07Z","snapshot_observed_at":"2026-08-01T07:07:41.854955Z","submitted_at":"2026-07-23T17:55:07Z","title":"Self-Supervised Learning of Structured Dynamics from Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T07:07:43.997932Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2607.21576"},"observation_digest":"sha256:405e15dcb5a5ac679086acc7b03f01399aef4cdf970988b329869633d1004010","observation_id":"35b55820-2081-4f19-8147-b4dd6e62b70f","resolution":{"observed_at":"2026-08-01T07:07:43.997932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/1809.03327/citation-record","integrity":"/paper/1809.03327/integrity","json":"/paper/1809.03327/citation-record.json","paper":"/paper/1809.03327"},"outbound":[],"paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 27 inbound Pith citation observations for arXiv:1809.03327."}