{"as_of":"2026-08-07T18:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:03aa02f74cc1c51264a406f490a18e1293933270a366ced6f608717811bed60d","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:53:10.443418Z","state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.06748/citation-record","integrity":"/paper/2506.06748/integrity","json":"/paper/2506.06748/citation-record.json","paper":"/paper/2506.06748"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.665226Z","title":"One- shot video object segmentation","venue":null,"work_id":"1c69b9fc-2141-428b-9b61-b78f9733f462","year":null},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.369033Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:6ac74076f194e08c82c53995ee13885564f4f71bac199a31cb02de4d4054495e","observation_id":"19a82926-2c95-4aa3-bcea-5bd2d73fed26","resolution":{"observed_at":"2026-08-07T05:53:10.668238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.657533Z","title":"Xmem: Long- term video object segmentation with an atkinson-shiffrin memory model","venue":null,"work_id":"62c969c8-0ea9-4542-b6ef-92b7d7eac7f0","year":2022},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.372867Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:aeba16fcf86217cdd18d9da7743c776d8c7cb5e7cad5e16f1a323a7201ca55ef","observation_id":"d4b009cf-e521-4bee-b322-0214b6f4f35e","resolution":{"observed_at":"2026-08-07T05:53:10.660577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.649914Z","title":"Rethink- ing space-time networks with improved memory coverage for efficient video object segmentation","venue":null,"work_id":"679a21df-72b4-401d-9d7e-e493950f1bb2","year":null},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.376144Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:a9fd668025c5267f54082f357ad254f918e62f18036390428971370d582606a5","observation_id":"d25ee1b0-3f5f-43ae-91e3-888e0fe39b4a","resolution":{"observed_at":"2026-08-07T05:53:10.652929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.641776Z","title":"Putting the object back into video object segmentation","venue":null,"work_id":"ab33dffc-6984-46d7-b388-9546974a319a","year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.379426Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:f4882201fd08b6c8c15cf03608358ee815920951bc0d290fb55a69531f485dc7","observation_id":"a1c78d10-fc14-446f-8dac-9c7e340d831f","resolution":{"observed_at":"2026-08-07T05:53:10.644530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.632117Z","title":"Epic-kitchens visor benchmark: Video segmenta- tions and object relations","venue":null,"work_id":"c0793f17-2a58-40f4-adc5-3c300520cbfe","year":2022},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.382571Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:c8e0693d80c2a22e42b74d0dc28007f91165eecf42e2fdfd7fb437724f2e4969","observation_id":"941f7b66-6ae6-4f07-8b74-270d4f96a4b5","resolution":{"observed_at":"2026-08-07T05:53:10.635716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.622568Z","title":"MOSE: A new dataset for video object segmentation in complex scenes","venue":null,"work_id":"a4e205e2-0048-4035-9bc1-187fb8d29fa8","year":null},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.385897Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:c0fc4d4c7d5e2e9fa1a370b70a16df4c24aa558713aab02a6cbf5c91b5c280a5","observation_id":"a1cf217f-134b-4a30-a321-5aeec0587cf2","resolution":{"observed_at":"2026-08-07T05:53:10.625621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16268","last_updated":"2025-07-29T00:08:01Z","snapshot_observed_at":"2026-07-06T19:37:16.203681Z","submitted_at":"2024-10-21T17:59:19Z","title":"SAM2Long: Enhancing SAM 2 for Long Video Segmentation with a Training-Free Memory Tree","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16268","snapshot_observed_at":"2026-08-07T05:53:10.389194Z","title":"Sam2long: Enhancing sam 2 for long video seg- mentation with a training-free memory tree.arXiv preprint arXiv:2410.16268, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.389194Z"},"links":{"cited_paper":"/paper/2410.16268","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:66ea793d3ec47b58f76f33ba82462b26709a8300a7eeee8a5bcfced94f47bd64","observation_id":"fc2dfb4c-ce12-47bb-9e93-18b824d08f2d","resolution":{"observed_at":"2026-08-07T05:53:10.389194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.614040Z","title":"Deep learning for video object segmentation: a review.Artificial Intelligence Review, 56(1):457–531, 2023","venue":null,"work_id":"733d7633-867c-4285-83f2-3c6bd58606a9","year":2023},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.392628Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:8e01c7144df2cfa380bb179ce3408092c4f226bc0d2b990c6473b60f082a9250","observation_id":"1102344f-e822-4535-961d-45abdfe6b9f2","resolution":{"observed_at":"2026-08-07T05:53:10.617011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.603602Z","title":"Ego-exo4d: Understanding skilled human activity from first-and third-person perspectives","venue":null,"work_id":"43447279-23b9-4d0d-9ad5-a0f9c40503e5","year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.395481Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:77d8cee8ef4e21051e73ff717aa971e78e9ad50bb4a075380eafcc342323dd99","observation_id":"c2a2926a-1205-499e-a68b-6ac7e00dc630","resolution":{"observed_at":"2026-08-07T05:53:10.607594Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19326","last_updated":"2024-05-01T01:30:58Z","snapshot_observed_at":"2026-07-06T18:07:30.890899Z","submitted_at":"2024-04-30T07:50:29Z","title":"LVOS: A Benchmark for Large-scale Long-term Video Object Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19326","snapshot_observed_at":"2026-08-07T05:53:10.398572Z","title":"Lvos: A benchmark for large- scale long-term video object segmentation.arXiv preprint arXiv:2404.19326, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.398572Z"},"links":{"cited_paper":"/paper/2404.19326","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:814a48aac689cf5e6da30f0a07b8e010c6e8e15455795415ec62c0ca64bfad2b","observation_id":"43f6b820-f8e6-443c-8814-23572813fa5d","resolution":{"observed_at":"2026-08-07T05:53:10.398572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.594991Z","title":"Video object segmentation with adaptive feature bank and uncertain-region refinement","venue":null,"work_id":"a1114936-59a9-4e06-856a-80cf577464ac","year":2020},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.401905Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:9aaa29b02f39df1a12c381d4386fd16f61b04fd66140a9f32b97f142798a26a9","observation_id":"16d95078-7ebb-454b-a0c4-3c27e8b383aa","resolution":{"observed_at":"2026-08-07T05:53:10.598545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.587092Z","title":"Video object segmentation using space-time memory networks","venue":null,"work_id":"ce641882-4404-4ce0-81dc-da6fe314d29b","year":2019},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.404328Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:ff35235b0b579295066a125f1b0c8c90e835466dd5da78cd3f4a85105fd2822e","observation_id":"687cdb10-da89-4b1b-b2ee-628cc908583a","resolution":{"observed_at":"2026-08-07T05:53:10.590004Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.577864Z","title":null,"venue":null,"work_id":"e97296fe-373b-48e0-88bb-c9c9a980d08b","year":2023},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.406886Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:50a6de3208e47229646408088cc00dfcc48d94c2dd6a6d054ca592809cd2f880","observation_id":"ff231224-9342-405b-a099-a12e4d4485d3","resolution":{"observed_at":"2026-08-07T05:53:10.581092Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.567631Z","title":"Hd-epic: A highly-detailed egocentric video dataset","venue":null,"work_id":"410d3634-f274-4fec-941c-c82e620c23e7","year":2025},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.410176Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:2399e6cb36edae4a7b766ebc883a2a3f2674615f9c10370b4395db1d61f20892","observation_id":"ad4e75bf-c6ed-4297-a0c7-36e540a42f4e","resolution":{"observed_at":"2026-08-07T05:53:10.570757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.558702Z","title":"An outlook into the fu- ture of egocentric vision.IJCV, 132(11):4880–4936, 2024","venue":null,"work_id":"83cf767b-8697-46a5-8b76-a1b66b9b158e","year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.413282Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:035b8d33808ec2d5b908fedc3527a17acf1f282bd2fdde1bfab48e247833d159","observation_id":"e8fba540-06d8-4cbb-bba9-fb6528151569","resolution":{"observed_at":"2026-08-07T05:53:10.561663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.00675","last_updated":"2018-03-01T17:50:08Z","snapshot_observed_at":"2026-08-02T10:51:13.194643Z","submitted_at":"2017-04-03T16:44:46Z","title":"The 2017 DAVIS Challenge on Video Object Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.00675","snapshot_observed_at":"2026-08-07T05:53:10.415724Z","title":"The 2017 davis challenge on video object segmentation.arXiv preprint arXiv:1704.00675, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.415724Z"},"links":{"cited_paper":"/paper/1704.00675","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:83630d25db6e5b30508d8596649e0f2799f5b0f0c20564728178d2aba829ed1d","observation_id":"037fd328-2d0d-4774-8016-4bd121ff6e34","resolution":{"observed_at":"2026-08-07T05:53:10.415724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.549810Z","title":"Vi- sion transformers for dense prediction","venue":null,"work_id":"c42b93a9-86ef-435f-a284-2e03d8bdcfbc","year":2021},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.419058Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:b978c5287bac880d1cb53742b2ec7c7ddea6ccfad78c8f1a3f459bae4bf05c55","observation_id":"f98f7708-bbaa-42eb-8627-2d312e3d8eed","resolution":{"observed_at":"2026-08-07T05:53:10.553197Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-07T05:53:10.423028Z","title":"Sam 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.423028Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:997cc94622844b6e0d24b3dac63bdd35c90bcf301c311cf0b5c4c781d1a73d40","observation_id":"c84fca61-5c02-477e-9a22-5b40a08f1b92","resolution":{"observed_at":"2026-08-07T05:53:10.423028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.540429Z","title":"Hi- era: A hierarchical vision transformer without the bells-and- whistles","venue":null,"work_id":"6dc97f21-9573-4285-8f01-dce26afb7ef2","year":2023},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.426737Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:00350195eb2c9c245c178261ce72b99c20fa58b52914f4418caf587ccefa1853","observation_id":"a9d641a0-7279-4079-8b35-921d99958325","resolution":{"observed_at":"2026-08-07T05:53:10.543682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17576","last_updated":"2024-12-04T08:58:53Z","snapshot_observed_at":"2026-07-06T19:57:24.407368Z","submitted_at":"2024-11-26T16:41:09Z","title":"A Distractor-Aware Memory for Visual Object Tracking with SAM2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17576","snapshot_observed_at":"2026-08-07T05:53:10.430158Z","title":"A distractor-aware memory for visual object tracking with sam2.arXiv preprint arXiv:2411.17576, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.430158Z"},"links":{"cited_paper":"/paper/2411.17576","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:433bf46a010c2b0ef5d9ff306440aa250edd265a94a7edb9c40087560b5fdfe6","observation_id":"290196ea-093f-41f3-87bd-c9a12f9af0f8","resolution":{"observed_at":"2026-08-07T05:53:10.430158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:53:10.527631Z","title":"Feelvos: Fast end-to-end embedding learning for video object segmenta- tion","venue":null,"work_id":"f6b0ab9d-f745-49e9-bacd-4a6fb6a9a961","year":2019},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.434401Z"},"links":{"citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:607c8f4b43eae5fae0fdab19c51e2d1d8353f54f66852b1d465c777521857682","observation_id":"70887bb4-34a3-4fe3-af64-a5ec28897730","resolution":{"observed_at":"2026-08-07T05:53:10.533170Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-08-07T05:53:10.437191Z","title":"Youtube-vos: A large-scale video object segmentation benchmark.arXiv preprint arXiv:1809.03327, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.437191Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:52d6bdf6bfcd64c397418633c2cb2d23b811de828a07695f47da7b7970c3d181","observation_id":"68d8fad7-d0e9-4b5d-bb39-930d446d4197","resolution":{"observed_at":"2026-08-07T05:53:10.437191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.11922","last_updated":"2024-11-30T22:32:34Z","snapshot_observed_at":"2026-08-07T02:44:57.260422Z","submitted_at":"2024-11-18T05:59:03Z","title":"SAMURAI: Adapting Segment Anything Model for Zero-Shot Visual Tracking with Motion-Aware Memory","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.11922","snapshot_observed_at":"2026-08-07T05:53:10.440098Z","title":"Samurai: Adapting segment anything model for zero-shot visual tracking with motion-aware memory.arXiv preprint arXiv:2411.11922,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.440098Z"},"links":{"cited_paper":"/paper/2411.11922","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:9c029ab3521adf57d20bf5f12df3c086631cdfde0aec56b956fd5236089914b3","observation_id":"1f06af49-3be1-4c6d-b98c-c8c3f86b6ab7","resolution":{"observed_at":"2026-08-07T05:53:10.440098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09414","last_updated":"2024-10-20T11:24:09Z","snapshot_observed_at":"2026-07-06T18:30:32.982860Z","submitted_at":"2024-06-13T17:59:56Z","title":"Depth Anything V2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09414","snapshot_observed_at":"2026-08-07T05:53:10.443418Z","title":"Depth any- thing v2.arXiv:2406.09414, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T05:53:10.443418Z"},"links":{"cited_paper":"/paper/2406.09414","citing_paper":"/paper/2506.06748"},"observation_digest":"sha256:c33a1d79b43105d5f528a5f636805aab6d1e21b4f5c98fdc61a44e026e2d24f1","observation_id":"8407f388-274e-47a0-91f8-bb5cbced8cab","resolution":{"observed_at":"2026-08-07T05:53:10.443418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.06748","last_updated":"2025-06-07T10:33:16Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T05:48:00.486161Z","submitted_at":"2025-06-07T10:33:16Z","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":0,"verified_fuzzy":15},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 0 inbound Pith citation observations for arXiv:2506.06748."}