{"as_of":"2026-08-13T09:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1b1ebca225c9b1a6635c9d4f4c08e6dc5f05b1a5caaf8d99b3a124d36412344e","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T04:42:34.488335Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.01147/citation-record","integrity":"/paper/2412.01147/integrity","json":"/paper/2412.01147/citation-record.json","paper":"/paper/2412.01147"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.211313Z","title":"Tarvis: A unified approach for target-based video segmentation, in: Pro- ceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, pp","venue":null,"work_id":"6cb9da23-98b7-493a-a45a-c1f721344b9b","year":2023},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.273043Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:c85c5016ee2c429ce1f07443dcb44630df375cff4a520217bba2c0863cb08a3b","observation_id":"02250719-b0c2-4e95-aa5a-788b98ebbb5c","resolution":{"observed_at":"2026-08-12T04:42:35.216072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.196175Z","title":null,"venue":null,"work_id":"5ec085bc-5214-4e2b-996b-b02c32ee7101","year":2020},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.277975Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:d39bc3b950feb587a98b9b229fb85438ce8234d21822317b86e5d2209f21bed0","observation_id":"630431fe-c9d2-46d5-ab8f-24b8a49dc12f","resolution":{"observed_at":"2026-08-12T04:42:35.200071Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.183190Z","title":"Memot: Multi-object tracking with memory, in: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, pp","venue":null,"work_id":"17316b26-5195-44f8-8f08-6893ff0fd93e","year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.282075Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:55c6f7575ef35e74ba0fb57b58bbe661963f510b9a28b49c4449606e38f9826a","observation_id":"6282fdc8-baad-4751-86b0-51a38cadfc15","resolution":{"observed_at":"2026-08-12T04:42:35.187931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.169097Z","title":null,"venue":null,"work_id":"9a38215c-813b-4e98-912a-8a917da621ec","year":null},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.287663Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:ba61928fb6adcdf74fa18a70ed42ea253db2deab14690c6963b9cd958f5c35e4","observation_id":"0a73cc0e-3c69-4576-a2f3-84399625c81b","resolution":{"observed_at":"2026-08-12T04:42:35.174532Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.141284Z","title":"Observation-centric sort: Rethinking sort for robust multi-object track- ing, in: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, pp","venue":null,"work_id":"c5060958-06b6-4225-bfaf-8e942a53ada0","year":2023},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.296605Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:88ce078471fca194f17a16562abb3c5cb4b51fa88e79f60330cdaf25ba2da415","observation_id":"9b74d68c-af0d-4a45-bae1-7b63d3a6d07b","resolution":{"observed_at":"2026-08-12T04:42:35.146102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10764","last_updated":"2021-12-20T18:59:59Z","snapshot_observed_at":"2026-08-11T22:52:24.521054Z","submitted_at":"2021-12-20T18:59:59Z","title":"Mask2Former for Video Instance Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10764","snapshot_observed_at":"2026-08-12T04:42:34.300822Z","title":"Mask2former for video instance segmentation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.300822Z"},"links":{"cited_paper":"/paper/2112.10764","citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:44b5665e6aa5794d82254a82db91f85dc70509123754aa4eac3dc12ddc5a0608","observation_id":"b52b7ee6-09a8-49a0-8f6c-ada5205370b7","resolution":{"observed_at":"2026-08-12T04:42:34.300822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.306884Z","title":"Masked-attention mask transformer for universal image segmentation, in: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition, pp","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.306884Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:f5c834d41d5740516cca337011bb510f5a90e67db3ee083b84d338cbd7bd2e4d","observation_id":"bc3c520d-7f6e-40c7-9fa9-741952621925","resolution":{"observed_at":"2026-08-12T04:42:34.306884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.119157Z","title":"Selective attention and the organization of visual information","venue":null,"work_id":"3ebd3520-4f12-4b93-9be6-fa9171cc9a1d","year":1984},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.311733Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:feb3c0d70ea08165806a6784dc71051295bd6515fb73adfac3d1604428ac692b","observation_id":"201cfb25-7387-489d-8401-a83a8fd7b31b","resolution":{"observed_at":"2026-08-12T04:42:35.123368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.104327Z","title":null,"venue":null,"work_id":"fe8430c0-0655-450e-86a0-34aff07be5b0","year":2023},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.316177Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:e1d3a6388e13cb21870dc2b0760a0083d2a79322fda4d41e69f5388329a3fdcc","observation_id":"87f94d3c-2f16-490e-bc71-1b623f4d24d8","resolution":{"observed_at":"2026-08-12T04:42:35.108525Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.089021Z","title":null,"venue":null,"work_id":"9f3e7e21-3f81-445d-a0d5-4dba84ac433c","year":null},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.320068Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:275ca3bd3fa2ff3d116fe542484966ced063cde40dce203d9ac6b51409732f19","observation_id":"70f2544b-eb86-4ee6-b26b-f430796a73bc","resolution":{"observed_at":"2026-08-12T04:42:35.093573Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.063566Z","title":"Coarse-to-fine amodal segmentation with shape prior, in: Proceedings of the IEEE/CVF International Conference on Computer Vision, pp","venue":null,"work_id":"4d110fd5-48ad-4882-a165-4a674c3adee5","year":2023},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.328229Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:5eef809cac74c14841b3b1e35e0b357c821a65cf2839c405c554c9a07c8e51f4","observation_id":"ba4c64fb-9483-4fd3-9286-928a866daa19","resolution":{"observed_at":"2026-08-12T04:42:35.067648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.050630Z","title":"Mask r-cnn, in: Proceedings of the IEEE international conference on computer vision, pp","venue":null,"work_id":"f00059fd-518b-4509-b619-7d85508738fc","year":2017},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.332269Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:a9a03affa5d79431eb238e4be15912b30c56d3c42b0275e350d1b33e9d14251f","observation_id":"edcb9025-1267-4991-aa38-a17b71cc8f86","resolution":{"observed_at":"2026-08-12T04:42:35.054926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.336880Z","title":"Deep residual learning for image recognition, in: Proceedings of the IEEE conference on computer vision and pattern recognition, pp","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.336880Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:12a86d358e390113db691c91772984b64c8e081e421c9445ed35cfcce771955a","observation_id":"e49bc68b-d83e-4485-8642-6f78afd5b0c8","resolution":{"observed_at":"2026-08-12T04:42:34.336880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.018361Z","title":"A generalized framework for video instance segmentation, in: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, pp","venue":null,"work_id":"577cf3f5-10dc-471e-b9ef-6b8da2157693","year":2023},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.340989Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:02f8f2ed0e9852ea130949974839b11363d091ec688f3803b909909882c5f219","observation_id":"77709a7c-8a57-495d-8305-c86878c0666f","resolution":{"observed_at":"2026-08-12T04:42:35.023523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.003955Z","title":"Vita: Video instance segmentation via object token association","venue":null,"work_id":"1328e4c0-b012-4c89-9fc1-c69aa99cab3b","year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.344918Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:ca13eb8688da01ffa7c9174bb4a9b24f211b82a4af27b5603d77601bb78b6617","observation_id":"6561fbd1-3956-4620-a01a-e32fac92c8a0","resolution":{"observed_at":"2026-08-12T04:42:35.008450Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.987977Z","title":null,"venue":null,"work_id":"26a658b0-96b0-4ade-83a4-6394ed7014aa","year":2019},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.348836Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:98bb989dac40bdfa888ed635f7d509bccc0ba1425edc52862cf966d408161cce","observation_id":"cf501a0c-b0c8-4230-b2be-48c660213ba9","resolution":{"observed_at":"2026-08-12T04:42:34.994548Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.974343Z","title":"Minvis: A minimal video instance segmentation framework without video-based training","venue":null,"work_id":"4d91a275-bd17-4e36-b672-7a4c23bebd79","year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.352561Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:1952e245f890a671e689085fe278dddea3e3c759023b1aef770f6b772ab0fe8b","observation_id":"43134889-f0d3-4011-aad7-4fc9e9b05427","resolution":{"observed_at":"2026-08-12T04:42:34.979256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.959593Z","title":"Video instance seg- mentation using inter-frame communication transformers","venue":null,"work_id":"31e1d5d0-3445-4782-bca3-ebac1eac30bc","year":2021},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.356670Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:297808d256f8b2ceba39270a8de32bf1faed21944024e9054c351e2bf48d06ce","observation_id":"aa5600fa-6e37-483f-b2ad-4035abaf3b70","resolution":{"observed_at":"2026-08-12T04:42:34.964804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.945546Z","title":"A theory of visual interpolation in object perception","venue":null,"work_id":"c79eaae6-3033-47cd-88bb-08144b86a248","year":1991},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.360563Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:28ffc572599992725fdf5701e29bc5f08986dc593231a32a0e0fd7c006d19d52","observation_id":"ac357b47-1b6b-43f9-99b5-5e9cf45d1339","resolution":{"observed_at":"2026-08-12T04:42:34.949901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.931393Z","title":"Offline-to-online knowl- edge distillation for video instance segmentation, in: Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision, pp","venue":null,"work_id":"929ef8f8-638f-454e-923c-07a7397e7e4c","year":2024},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.364507Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:8f92128d54c027be80052e4d486630b91623355327750c2c53ead120122a865a","observation_id":"9fff41b2-4aad-4520-a85f-3268399046fa","resolution":{"observed_at":"2026-08-12T04:42:34.935800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.917943Z","title":"Amodal instance segmentation, in: European Conference on Computer Vision, Springer","venue":null,"work_id":"3af73bab-843f-492d-80d6-bb2b42c3e45f","year":2016},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.368839Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:c484308abeb56487f7e0caae0403ccd8740efd647bd4a36009f7ee07d225de68","observation_id":"90c60dbe-a6f4-49d9-b52f-dc0769299743","resolution":{"observed_at":"2026-08-12T04:42:34.922191Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.904515Z","title":"Microsoft coco: Common objects in context, in: European conference on computer vision, Springer","venue":null,"work_id":"50d64437-c963-4f96-b8c4-312d99281e54","year":2014},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.372703Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:ca2b79ff38ca9fc569f0200d4c69993c5e76374d3c894014b781bf5d37d8df27","observation_id":"ec1365ed-4239-42c6-a2fb-d50c4ace3cb2","resolution":{"observed_at":"2026-08-12T04:42:34.909110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.891907Z","title":"Sg-net: Spatial granularity network for one-stage video instance segmentation, in: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recogni- tion, pp","venue":null,"work_id":"0d69e27c-80e0-4425-b72c-90a6ff9c50df","year":null},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.376415Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:aadfdecfae39afae526b069ab2f4b651826260f92a01c3c691637c4f8ec55def","observation_id":"382318d1-05f0-426b-b4df-9e6bbc2ef1b0","resolution":{"observed_at":"2026-08-12T04:42:34.896364Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.878433Z","title":"Blade: Box-level supervised amodal segmentation through directed expansion, in: Proceedings of the AAAI Conference on Artificial Intelligence, pp","venue":null,"work_id":"00654541-bec0-47f1-991f-17c6374287d5","year":2024},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.380270Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:d94b937abdbe57d76fbf9c261a3329aa44bf7a6a505bf8180dc7a718a64d0f79","observation_id":"a36f9ed3-e3ca-4a71-8ea2-0b2bbd7f7c35","resolution":{"observed_at":"2026-08-12T04:42:34.883188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.384195Z","title":"Swin transformer: Hierarchical vision transformer using shifted windows, in: Proceedings of the IEEE/CVF international conference on computer vision, pp","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.384195Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:684d83f1d53e248874d421d6f6857bb5e231fae04444cfc6b9cc3c464f910b97","observation_id":"6b31e1ba-4714-4170-b949-1ceeb55e3c3e","resolution":{"observed_at":"2026-08-12T04:42:34.384195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.856920Z","title":"Hota: A higher order metric for evaluating multi-object tracking","venue":null,"work_id":"99537438-15ab-43ff-aaf1-1a43ab2d22a9","year":2021},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.389549Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:acfd68043c750f6d8c50175a97b084c2b5c5c8e6d3ac2d16ab6c50994db94031","observation_id":"7e5e91a2-86b3-4481-aca1-083992f1a49f","resolution":{"observed_at":"2026-08-12T04:42:34.861145Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.843687Z","title":"Trackformer: Multi-object tracking with transformers, in: Proceedings 29 of the IEEE/CVF conference on computer vision and pattern recogni- tion, pp","venue":null,"work_id":"7094312d-0e27-4979-9c2f-cc51241fc8e8","year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.393376Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:b53a3f4ec50157fbeb2635d36386d6caee7cdafc33c085691cad6a4604d8e86f","observation_id":"ca0077ea-ab09-4c88-91d0-a5a85ad83764","resolution":{"observed_at":"2026-08-12T04:42:34.847853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.830595Z","title":"Video object segmentation using space-time memory networks, in: Proceedings of the IEEE/CVF International Conference on Computer Vision, pp","venue":null,"work_id":"74104015-3d1e-4379-adc2-b88ffbfc8c52","year":2019},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.397423Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:b79402be221ea6c621f95b66c3ef6c6e951e66751da5a93a07b213f6f8924e1c","observation_id":"d2cccd63-a54d-426c-a354-a2ca722d53f4","resolution":{"observed_at":"2026-08-12T04:42:34.834950Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.818244Z","title":"pix2gestalt: Amodal segmentation by synthesiz- ing wholes, in: 2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR), IEEE Computer Society","venue":null,"work_id":"c12bf1b2-1267-4b77-8016-d49205b6d65c","year":2024},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.401256Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:67ebff0248bac54b11e46ee6b9204adcf99f4f0e09ba622b85437f9ffb1f10ce","observation_id":"6c29f5ac-e3dd-4254-9b49-827696399741","resolution":{"observed_at":"2026-08-12T04:42:34.822518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.806180Z","title":"Occluded video instance segmentation: A benchmark","venue":null,"work_id":"e73361e2-7952-474e-bcfa-51d5263cbf2c","year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.405286Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:46bbbede075fa2d191a10980b40080438d5823a0b3ec97d93f23e9e028f78bba","observation_id":"61f412a5-113b-49a0-92e1-b618fed8e224","resolution":{"observed_at":"2026-08-12T04:42:34.810742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.794037Z","title":"Coarse-to- fine video instance segmentation with factorized conditional appearance flows","venue":null,"work_id":"6ae0d933-4115-4c29-bb4a-1c331e77075b","year":null},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.409367Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:bd25b6427082c1e222b512718aef037e0bd87efa7056f91599d61308c0acd7cc","observation_id":"2c19913d-9388-4198-989f-78cb4daed242","resolution":{"observed_at":"2026-08-12T04:42:34.798651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.782033Z","title":"Motiontrack: Learning robust short-term and long-term motions for multi-object tracking, in: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, pp","venue":null,"work_id":"10e6f84e-0fe6-4072-826b-b5e1342a5dfd","year":null},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.413941Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:f3f5d8d55cf7151ad5e9901ba2584f7fd3f5c890711b8adba999813776c4b630","observation_id":"6f39d737-190f-4425-9f9d-36c50b80533b","resolution":{"observed_at":"2026-08-12T04:42:34.786431Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.769369Z","title":"Perfor- mance measures and a data set for multi-target, multi-camera tracking, in: European conference on computer vision, Springer","venue":null,"work_id":"79ff05ee-91ea-4c08-8dfc-9547e3a9a5f9","year":2016},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.418236Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:84edd18d91c59f43f14ec19ce44a150eb5a6f90006933912b73663250fbee989","observation_id":"946c38d0-2c76-41b6-adb3-96cdc6967190","resolution":{"observed_at":"2026-08-12T04:42:34.773497Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.756995Z","title":"Dancetrack: Multi-object tracking in uniform appearance and diverse motion, in: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, pp","venue":null,"work_id":"7b44435a-c8cb-40a1-946b-718920f22473","year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.423112Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:a29d25e68438ca959ba5edd40dd9c7020e4442b6d714f2e1796dd1bf7b2c843b","observation_id":"496dee78-d790-49e1-b209-ad22108f6dcf","resolution":{"observed_at":"2026-08-12T04:42:34.761354Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.06562","last_updated":"2023-05-15T12:22:51Z","snapshot_observed_at":"2026-07-06T11:57:21.548494Z","submitted_at":"2021-10-13T08:22:04Z","title":"Unsupervised Object Learning via Common Fate","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.06562","snapshot_observed_at":"2026-08-12T04:42:34.427130Z","title":"Unsuper- vised object learning via common fate","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.427130Z"},"links":{"cited_paper":"/paper/2110.06562","citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:2275c6bc7016d1314e907b1f053aaa80e26cb16e7a1a358f3c3ca9031dfc19b2","observation_id":"8be367cc-46fc-417e-a4ec-72631eb3aee6","resolution":{"observed_at":"2026-08-12T04:42:34.427130Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11376","last_updated":"2024-04-17T16:46:02Z","snapshot_observed_at":"2026-08-13T00:52:20.968398Z","submitted_at":"2024-03-18T00:03:48Z","title":"ShapeFormer: Shape Prior Visible-to-Amodal Transformer-based Amodal Instance Segmentation","version":4},"cited_work":{"arxiv_id":"2403.11376","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.11376","snapshot_observed_at":"2026-08-12T04:42:34.579793Z","title":"ShapeFormer: Shape Prior Visible-to-Amodal Transformer-based Amodal Instance Segmentation","venue":"cs.CV","work_id":"1db8943a-6fcf-42c0-bbca-74a26087a5e7","year":2024},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.431579Z"},"links":{"cited_paper":"/paper/2403.11376","citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:9ede66b2bba3390fbe9c8df930249c151c9c2a592aecaf66e98e23d5062dd20f","observation_id":"241f3bc7-f062-4efc-9eec-6523891c6342","resolution":{"observed_at":"2026-08-12T04:42:34.584410Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.06323","last_updated":"2024-03-17T22:58:03Z","snapshot_observed_at":"2026-07-06T14:04:16.634759Z","submitted_at":"2022-10-12T15:42:40Z","title":"AISFormer: Amodal Instance Segmentation with Transformer","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.06323","snapshot_observed_at":"2026-08-12T04:42:34.436022Z","title":"Aisformer: Amodal instance segmentation with transformer","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.436022Z"},"links":{"cited_paper":"/paper/2210.06323","citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:91e683f912bd90673b359e7ed27fdb6463b0d6f30ea011416c07e8f1f732016f","observation_id":"ea3f3c5b-e088-4985-b297-f96037895e52","resolution":{"observed_at":"2026-08-12T04:42:34.436022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.743224Z","title":"Ov-vis: Open-vocabulary video instance segmentation","venue":null,"work_id":"f31a7b6b-da77-4977-9b93-1c3fae5eb033","year":2024},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.440416Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:39a0682accc850a4f7049a9c74bd69746122ebc93ca5f846521c1014265b7a88","observation_id":"d9e3c107-c224-4223-8b4e-8a6beaf92c06","resolution":{"observed_at":"2026-08-12T04:42:34.747582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.729706Z","title":"Seqformer: Se- quential transformer for video instance segmentation, in: European Con- ference on Computer Vision, Springer","venue":null,"work_id":"533fbebd-3e2f-4ce6-a0d7-9c80553387e0","year":null},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.444113Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:54a73fa6f72c01552cc54a3374300dbe81be7dc330a79e16f3c79a7d7f98ce99","observation_id":"c6cf8a95-24ac-4df2-8d6e-bcb0d9bea37b","resolution":{"observed_at":"2026-08-12T04:42:34.734183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.717152Z","title":"In de- fense of online models for video instance segmentation, in: European Conference on Computer Vision, Springer","venue":null,"work_id":"7085fa62-4d11-46d3-8737-e642815e6711","year":null},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.448371Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:9149af572034faa54b40be5555bc9dc2e717d03fdcf911e9e76d7c4e634eca0f","observation_id":"899314ca-d4f1-4510-879b-a4edea00f9c3","resolution":{"observed_at":"2026-08-12T04:42:34.721257Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.05598","last_updated":"2020-12-19T13:24:36Z","snapshot_observed_at":"2026-07-06T10:23:03.672913Z","submitted_at":"2020-12-10T11:39:09Z","title":"Amodal Segmentation Based on Visible Region Segmentation and Shape Prior","version":2},"cited_work":{"arxiv_id":"2012.05598","doi":null,"metadata_source":"pith","pith_arxiv_id":"2012.05598","snapshot_observed_at":"2026-08-12T04:42:34.549195Z","title":"Amodal Segmentation Based on Visible Region Segmentation and Shape Prior","venue":"cs.CV","work_id":"3f10684e-c258-40ed-855d-cb3f37f09b7a","year":2020},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.452125Z"},"links":{"cited_paper":"/paper/2012.05598","citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:c49ea7f7157007cd7c1633447fabca95921b12050942a32aeef1759356b1feea","observation_id":"af29e1a4-9f47-4436-8416-aaa1126c1e09","resolution":{"observed_at":"2026-08-12T04:42:34.555263Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.703050Z","title":"Video instance segmentation, in: Pro- ceedings of the IEEE/CVF International Conference on Computer Vi- sion, pp","venue":null,"work_id":"6c12c493-9b8f-4d9e-bc53-6c5f1a1c9ff4","year":2019},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.456468Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:3652d3b94e7b394b2e561a8ed86bdf3ba0303fc48c94f62bd81666dffc5bb156","observation_id":"b3e89677-2577-442b-bc51-c77d281c2a1a","resolution":{"observed_at":"2026-08-12T04:42:34.708234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.689902Z","title":"Self-supervised amodal video object segmentation","venue":null,"work_id":"a7249d63-e431-4adb-9500-386865b461c6","year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.460022Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:4496c56cf2af54c004d5fef47b43b573b15f5d4cf045b858d9f7d5739703a3f1","observation_id":"6c7a8bc6-fbad-473a-8e55-931e6ddc4478","resolution":{"observed_at":"2026-08-12T04:42:34.694038Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.676853Z","title":"Motr: End-to-end multiple-object tracking with transformer, in: Euro- pean Conference on Computer Vision, Springer","venue":null,"work_id":"939fedbc-ab26-4a3b-871c-89b2b150c774","year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.463754Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:1fc2e9008bd5048afd201228a3eb5c893e22b7ae7025caf81424af36ca06a050","observation_id":"3b3e1f54-6077-475c-bf77-70a71f68f954","resolution":{"observed_at":"2026-08-12T04:42:34.680934Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.663811Z","title":"Amodal ground truth and completion in the wild, in: Proceedings of the IEEE/CVF 31 Conference on Computer Vision and Pattern Recognition, pp","venue":null,"work_id":"f3b9cb09-6e2d-4894-9b40-fb5a28a72ccf","year":2024},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.467514Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:ca9bff16a83d1ab9b80dad058f90bad95eb13fd525e897395871510bc0dc0997","observation_id":"48bbdeca-5207-4b94-b70f-66524cb3d383","resolution":{"observed_at":"2026-08-12T04:42:34.668095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03413","last_updated":"2023-07-14T08:46:08Z","snapshot_observed_at":"2026-07-06T15:39:02.609692Z","submitted_at":"2023-06-06T05:24:15Z","title":"DVIS: Decoupled Video Instance Segmentation Framework","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03413","snapshot_observed_at":"2026-08-12T04:42:34.471447Z","title":"Dvis: Decoupled video instance segmentation framework","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.471447Z"},"links":{"cited_paper":"/paper/2306.03413","citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:bba5d09a83604c9774b959a6239aaf6915f1cc2d56ff40365e58fd1eee7c9aef","observation_id":"c2c124e0-31c0-4aff-9e4b-2d794abfe11b","resolution":{"observed_at":"2026-08-12T04:42:34.471447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.651371Z","title":"Bytetrack: Multi-object tracking by associat- ing every detection box, in: European Conference on Computer Vision, Springer","venue":null,"work_id":"2b45ac61-1df6-4a8a-b2c4-d829cdb52c63","year":2022},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.475834Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:20d4c177c9bee79a3499f2979464075d523d7c340966a187075b9f97b5f31e35","observation_id":"96c7501d-659c-4a37-8812-844f635c5301","resolution":{"observed_at":"2026-08-12T04:42:34.655587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.637093Z","title":"Fairmot: On the fairness of detection and re-identification in multiple object tracking","venue":null,"work_id":"3dcf1ec1-110a-441e-979a-cce1bb563e7d","year":2021},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.480109Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:be3899ffc32c665ecfff35c263f17bd9c568f3bc29f421adefe5b9cdd6e84128","observation_id":"4d0502c3-7402-4d06-b133-cbf9cd1c8ec3","resolution":{"observed_at":"2026-08-12T04:42:34.642755Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:34.621777Z","title":"Motrv2: Bootstrapping end-to- end multi-object tracking by pretrained object detectors, in: Proceed- ings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, pp","venue":null,"work_id":"761e0238-1006-4f4b-9129-b250f1fdea37","year":null},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.484338Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:f4e3a915099e4ce7207c611e6e017cf27870610f1a31265d66bff9887ac7026e","observation_id":"4243bee1-1125-4d31-a920-73ad513c6358","resolution":{"observed_at":"2026-08-12T04:42:34.627472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.04159","last_updated":"2021-03-18T03:14:26Z","snapshot_observed_at":"2026-08-13T07:13:04.991235Z","submitted_at":"2020-10-08T17:59:21Z","title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.04159","snapshot_observed_at":"2026-08-12T04:42:34.488335Z","title":"Deformable detr: Deformable transformers for end-to-end object detection","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.488335Z"},"links":{"cited_paper":"/paper/2010.04159","citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:830dbd35ace408f7dab280033507a26f4f832647697cddf06e0ccf6e7f977bdd","observation_id":"1afbdc35-72bb-49b2-ae75-feb87855366b","resolution":{"observed_at":"2026-08-12T04:42:34.488335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.075762Z","title":null,"venue":null,"work_id":"d375151a-64ea-4372-b43a-2ad827f36772","year":2019},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.324251Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:20ab8d27e4029b6b46c3058b22a26a1d65e1c3826559f495c0cbddb0a4efd843","observation_id":"96b61c04-92ab-4614-8758-18ba7008c178","resolution":{"observed_at":"2026-08-12T04:42:35.079844Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:42:35.155433Z","title":null,"venue":null,"work_id":"3542c45a-418b-4ffc-b638-df41c5b1b996","year":2020},"citing_paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-12T04:42:34.292294Z"},"links":{"citing_paper":"/paper/2412.01147"},"observation_digest":"sha256:7172d030274bbfd9e156438b200eacba35c870d8579607b1ad4b3f474cdb909b","observation_id":"0b1f5a2d-6b89-4965-a2be-6fb61268462f","resolution":{"observed_at":"2026-08-12T04:42:35.159910Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.01147","last_updated":"2025-04-09T21:26:06Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T23:52:06.541011Z","submitted_at":"2024-12-02T05:44:29Z","title":"A2VIS: Amodal-Aware Approach to Video Instance Segmentation"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":15,"verified_exact":2,"verified_fuzzy":35},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 0 inbound Pith citation observations for arXiv:2412.01147."}