{"as_of":"2026-08-10T04:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8e7e7d708037eb6579dd3ff76d24143339cbcd7b903761897c92f25e2262034a","coverage":[{"denominator":61,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:52:49.988237Z","state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.01304/citation-record","integrity":"/paper/2506.01304/integrity","json":"/paper/2506.01304/citation-record.json","paper":"/paper/2506.01304"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.734427Z","title":"Xmem++: Production-level video segmenta- tion from few annotated frames","venue":null,"work_id":"6dab7723-9664-4c39-8532-3bc56666f3b7","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.181841Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:c2e236d53da7d5aed88b11f02a7c68f232181d28be630b65243831a2a1f1d28b","observation_id":"11a488a3-9d79-45bd-84bf-3beb1059ecc5","resolution":{"observed_at":"2026-08-07T11:52:53.738085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.722117Z","title":"One- shot video object segmentation","venue":null,"work_id":"5829c8e3-d0c2-456d-8a83-b1f07b2497f4","year":2017},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.234531Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:e8a97b335987cca17dfae11b52ebe8bcacad2e0280c8216bf22f39ca1400b20a","observation_id":"366b346b-2fb1-42a3-9b20-82c3210325bf","resolution":{"observed_at":"2026-08-07T11:52:53.726052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.710802Z","title":"Rsprompter: Learning to prompt for remote sensing instance segmenta- tion based on visual foundation model.IEEE Transactions on Geoscience and Remote Sensing, 2024","venue":null,"work_id":"2bb12ff6-8add-4f0f-9e43-55dc09e9a1e7","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.287685Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:b776e05c27f11487207d4881e2ffd8b4320e1e5ce1bb4578ab2f48434c413f23","observation_id":"11561c84-e797-4fbf-950d-36de8a62d960","resolution":{"observed_at":"2026-08-07T11:52:53.714060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.699760Z","title":"Adaptformer: Adapting vision transformers for scalable visual recognition","venue":null,"work_id":"d47147b2-0b65-46aa-af1e-5af8dceac751","year":2022},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.368695Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:901d20d46e0dc291ba869b4912a3916153f72fda45fad03810b9b0fc885089dc","observation_id":"ba02cf08-f5f4-4072-95c3-ddd96b478355","resolution":{"observed_at":"2026-08-07T11:52:53.703563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04579","last_updated":"2024-08-10T11:20:52Z","snapshot_observed_at":"2026-07-06T18:58:30.942582Z","submitted_at":"2024-08-08T16:40:15Z","title":"SAM2-Adapter: Evaluating & Adapting Segment Anything 2 in Downstream Tasks: Camouflage, Shadow, Medical Image Segmentation, and More","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04579","snapshot_observed_at":"2026-08-07T11:52:38.439660Z","title":"Sam2-adapter: Evaluating & adapting segment any- thing 2 in downstream tasks: Camouflage, shadow, medical image segmentation, and more.arXiv:2408.04579, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.439660Z"},"links":{"cited_paper":"/paper/2408.04579","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:061ffe8ae01743e796a5bab26f2401952c0b78d0989d89ff1a085c6bf5ead596","observation_id":"4a334688-76cb-41b5-8674-3a3d106dde9e","resolution":{"observed_at":"2026-08-07T11:52:38.439660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.689355Z","title":"0.1% data makes segment anything slim.NeurIPS,","venue":null,"work_id":"d94cf53c-02f7-4520-8ceb-b29d639f50d1","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.533843Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:ae6fb9acb8abc3e873f181036ccdbdcd22aad66e980bc9e4db3ab2a5ae3ca01a","observation_id":"db998cdb-5490-40f5-97fe-37adf7115483","resolution":{"observed_at":"2026-08-07T11:52:53.692814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.677987Z","title":"Xmem: Long- term video object segmentation with an atkinson-shiffrin memory model","venue":null,"work_id":"33e59893-66de-40d6-bc58-21dc23460140","year":2022},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.585479Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:a650dc39fa5e4748d0f0bb3e536bf02e3e13f6aef465df30388faf8638c957d5","observation_id":"4c34bce8-2e12-47db-abee-d3958b50182d","resolution":{"observed_at":"2026-08-07T11:52:53.682205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.664897Z","title":"Modular interactive video object segmentation: Interaction-to-mask, propagation and difference-aware fusion","venue":null,"work_id":"b9a4ccf9-d709-48af-bf44-c26902ca759c","year":2021},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.657606Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:31be8c26907fcb677f4a7086d90c6133c2f8c439de9096667f4ced9fd1a03051","observation_id":"4e577a05-e56e-4644-8cdd-8399b2e354d6","resolution":{"observed_at":"2026-08-07T11:52:53.669131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.652440Z","title":"Rethink- ing space-time networks with improved memory coverage for efficient video object segmentation.NeurIPS, 2021","venue":null,"work_id":"14bb0bc5-81b3-4bac-af7b-c1b1c906b0d1","year":2021},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.735290Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:27d83956ddcd0a53a4f567c4a047a259b5bb464ebe73f5aad20a2fb482ebd2d9","observation_id":"7e4ddc88-dcfd-4521-8ca4-3f5f80beee02","resolution":{"observed_at":"2026-08-07T11:52:53.656205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:38.821648Z","title":"Tracking anything with de- coupled video segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.821648Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:05b421c04aa90352815fd0df69d3a17373e3b5d1e19768d8051b7e735ed6b797","observation_id":"f07074bf-67a3-4cd5-8cd0-9247113ec8ae","resolution":{"observed_at":"2026-08-07T11:52:38.821648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.632970Z","title":"Putting the object back into video object segmentation","venue":null,"work_id":"abbc699c-2aae-4361-a6b0-829db8e20020","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.900297Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:a55c84f7db5b29974d653e4e08b5e73bdffe580da25890a2a37622a723941d60","observation_id":"68e2cdb2-f115-42c5-b0f3-2c108187e7a2","resolution":{"observed_at":"2026-08-07T11:52:53.636635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06558","last_updated":"2023-05-11T04:33:08Z","snapshot_observed_at":"2026-08-03T19:49:16.800693Z","submitted_at":"2023-05-11T04:33:08Z","title":"Segment and Track Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06558","snapshot_observed_at":"2026-08-07T11:52:38.984260Z","title":"Segment and track anything.arXiv:2305.06558, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:38.984260Z"},"links":{"cited_paper":"/paper/2305.06558","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:f50055cfc680884c312b414fbf9df6def385b818ea0dcc0f9aecb773bbbeb776","observation_id":"9b1cc4fd-526c-497a-b21e-a3484ac88c49","resolution":{"observed_at":"2026-08-07T11:52:38.984260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2003.10555","last_updated":"2020-03-23T21:17:42Z","snapshot_observed_at":"2026-08-06T14:37:14.514749Z","submitted_at":"2020-03-23T21:17:42Z","title":"ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2003.10555","snapshot_observed_at":"2026-08-07T11:52:39.082937Z","title":"Electra: Pre-training text encoders as discrimina- tors rather than generators.arXiv:2003.10555, 2020","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.082937Z"},"links":{"cited_paper":"/paper/2003.10555","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:51433e7a82bc18e36e541b23060608154396d8f538a10310452e38dedec30f7c","observation_id":"a33b334d-66b1-4181-a184-740ff8cbc34e","resolution":{"observed_at":"2026-08-07T11:52:39.082937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.622448Z","title":"Learning the what and how of annotation in video object segmentation","venue":null,"work_id":"df732d4d-cdf6-4b71-affe-1285be6e98fc","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.212385Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:0d19b5d3a02989084750f6ebb843423c5ea1d60abbf3ca9cd421d2e5d88711f9","observation_id":"409ef4a8-5aeb-4347-a106-dceda66c567b","resolution":{"observed_at":"2026-08-07T11:52:53.625812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00756","last_updated":"2024-08-22T16:38:20Z","snapshot_observed_at":"2026-07-06T18:55:45.493755Z","submitted_at":"2024-08-01T17:57:25Z","title":"Segment anything model 2: an application to 2D and 3D medical images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00756","snapshot_observed_at":"2026-08-07T11:52:39.281260Z","title":"Segment anything model 2: an ap- plication to 2d and 3d medical images.arXiv:2408.00756,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.281260Z"},"links":{"cited_paper":"/paper/2408.00756","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:9e686e01bc60183193155c5c6917b66580a901efbda59447e759bf5f0e967d28","observation_id":"35c29cac-6a12-481e-84a0-ac4ac2ca8add","resolution":{"observed_at":"2026-08-07T11:52:39.281260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-10T01:12:16.468283Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-07T11:52:39.420425Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale.arXiv:2010.11929,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.420425Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:b0de194453639c1da527e7a352883f87e682cd1a87f0f8c0f6a49a7780c20daa","observation_id":"642f041d-7bee-48c2-94c4-6b3bac3f550a","resolution":{"observed_at":"2026-08-07T11:52:39.420425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.611718Z","title":"Interactive video object segmentation using global and local transfer modules","venue":null,"work_id":"aff4b6b9-c607-4d3c-a4cf-985c2f96afef","year":2020},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.603848Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:c644d5be1bbb0cee384bf3c668b03076f0e8ca98aefad8788b6ddb1e0e5c6730","observation_id":"8c099b13-52de-4c6f-820c-5c523858aadf","resolution":{"observed_at":"2026-08-07T11:52:53.615372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.06301","last_updated":"2023-02-17T08:33:28Z","snapshot_observed_at":"2026-08-02T15:18:34.660553Z","submitted_at":"2023-02-13T12:02:51Z","title":"A Neuromorphic Dataset for Object Segmentation in Indoor Cluttered Environment","version":2},"cited_work":{"arxiv_id":"2302.06301","doi":null,"metadata_source":"pith","pith_arxiv_id":"2302.06301","snapshot_observed_at":"2026-08-07T11:52:50.447917Z","title":"A Neuromorphic Dataset for Object Segmentation in Indoor Cluttered Environment","venue":"cs.CV","work_id":"0529405f-902c-4269-ab54-54351c0c87a9","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:39.861021Z"},"links":{"cited_paper":"/paper/2302.06301","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:220799018013bfc894644387dc6d9885a061af6adbd55f2b1523883d46c1d918","observation_id":"58d1537e-41f7-47b0-a316-1aeda0409a70","resolution":{"observed_at":"2026-08-07T11:52:50.520855Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.600557Z","title":"Prompting visual-language models for efficient video understanding","venue":null,"work_id":"b9c30936-f732-44e8-87a5-431324f37815","year":2022},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:40.192127Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:96dd16a92e239f16f0f296de23bf73838188396d62b7b554aa588879c0c1258a","observation_id":"d6944856-87ef-495f-a04c-90eb2ac2d4c0","resolution":{"observed_at":"2026-08-07T11:52:53.604426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.590900Z","title":"Segment anything in high qual- ity.NeurIPS, 2023","venue":null,"work_id":"304cdcf4-d95a-4fb2-bef0-f2510cb29892","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:40.888261Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:0a681bed6f30f4a6d5fbd9f36a2f7a7566dffeab8f46015a0020e65c9fc15663","observation_id":"3c3594b6-e9d5-4f84-86ef-35f0455f93e5","resolution":{"observed_at":"2026-08-07T11:52:53.593770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.581365Z","title":"Segment any- thing","venue":null,"work_id":"8b8d963d-adfe-40de-8f1f-cf7a9695c75c","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:42.580715Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:75ff23a7af5fdf835a058d3cab79a9fbf222dcdf9edfa41ab9b22f0e727b007e","observation_id":"50853442-a2d1-4ad6-8d6a-4f62cb46ffe0","resolution":{"observed_at":"2026-08-07T11:52:53.584383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.571280Z","title":"Frozen clip models are efficient video learners","venue":null,"work_id":"d28a290f-a247-4a6e-9543-4eb9fc582a58","year":2022},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:44.078694Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:676fdb5047d1e24304f40744c4db8ee4806eb5a2523a2f476725b48776bfce2b","observation_id":"06580f31-8f5e-41a2-a50f-92f1890d0de8","resolution":{"observed_at":"2026-08-07T11:52:53.574762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.07931","last_updated":"2025-03-11T12:57:43Z","snapshot_observed_at":"2026-08-06T05:55:21.187619Z","submitted_at":"2024-08-15T04:59:12Z","title":"Surgical SAM 2: Real-time Segment Anything in Surgical Video by Efficient Frame Pruning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.07931","snapshot_observed_at":"2026-08-07T11:52:44.860444Z","title":"Surgical sam 2: Real-time segment anything in surgical video by efficient frame pruning.arXiv:2408.07931,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:44.860444Z"},"links":{"cited_paper":"/paper/2408.07931","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:63b2725bc58305dc701b512965148a5db61f6f264c35cca877ac643f8c7ec729","observation_id":"3d97282a-0322-42fa-a8d3-8595e413b810","resolution":{"observed_at":"2026-08-07T11:52:44.860444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-09T20:34:52.923500Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-07T11:52:45.975938Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:45.975938Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:8015253faca09f4dd374f5a331d1c06a2dbab6bed1de2b61a3caf59f849d0c59","observation_id":"a27319da-c951-430b-95ce-aa2214fc1cdd","resolution":{"observed_at":"2026-08-07T11:52:45.975938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.553556Z","title":"Segment anything in medical images.Nature Communications, 2024","venue":null,"work_id":"a2981150-d034-45e7-b11b-5151a247219d","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:46.692481Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:9ecd99439abcca243cc2ccb68f1368a46087b0f6a987e1486df100354a7133af","observation_id":"46538667-b3ba-4a45-9504-1aea96602a29","resolution":{"observed_at":"2026-08-07T11:52:53.562809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.542330Z","title":"Video object segmentation without temporal information.IEEE TPAMI, 2018","venue":null,"work_id":"1d442f76-427f-4c0b-a960-c3d2560c6b49","year":2018},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:46.789541Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:ed4109fad432d031b4200f896bf3077db749ee4de0f718c53a7f4c50e6e16749","observation_id":"296753c5-d4e9-4a3e-9d0c-e1f19ce65fb1","resolution":{"observed_at":"2026-08-07T11:52:53.545539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.463407Z","title":"Segment anything model for medical image analysis: an experimental study","venue":null,"work_id":"7a36c96b-f21d-4a60-88f3-6aabbcebdbae","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:46.896233Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:5fc23b617f30a28f4526961f50b3805d15c2040e47f5e7981ca542de6e870600","observation_id":"97ffd79f-007d-4c78-84ff-5e994a6b19fc","resolution":{"observed_at":"2026-08-07T11:52:53.517771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.269816Z","title":"Fast video object segmentation by reference- guided mask propagation","venue":null,"work_id":"ac4ce4cc-d8f0-4f3c-8c73-7a577843630a","year":2018},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.024429Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:98b9c57251aa161e6973fb0634130395442603ee01aebbeab63cd643a0b831f6","observation_id":"cf47f950-2c8b-4f84-9960-b8b403de068f","resolution":{"observed_at":"2026-08-07T11:52:53.363558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:53.114877Z","title":"St-adapter: Parameter-efficient image-to-video transfer learning.NeurIPS, 2022","venue":null,"work_id":"e4148757-683f-4c5f-9275-dccd68e25f68","year":2022},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.074668Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:2e7aa6204ecbc059defcbb40c3b6492609e6aa720661689cf2c3461b375a4843","observation_id":"ade024f6-0491-46a6-86ce-d70d8e35fd4f","resolution":{"observed_at":"2026-08-07T11:52:53.187973Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.975695Z","title":"Dual- path adaptation from image to video transformers","venue":null,"work_id":"41c7dd3b-6c01-4f12-9a22-ab9e2e0a27c5","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.194040Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:0b86be154d10b9816fc396c017f59366d9a76ccc45a2db46fff7a7050f3f1897","observation_id":"fce007a0-bd8a-485a-9603-4ae670e231c8","resolution":{"observed_at":"2026-08-07T11:52:53.020964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:47.273487Z","title":"Pytorch: An imperative style, high-performance deep learning library","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.273487Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:cd88d80899754b4f1d2465aa02e352d924679c289d65d9aedcda77229df22cc3","observation_id":"1c66845a-b417-4fe3-be90-b52433a7d20f","resolution":{"observed_at":"2026-08-07T11:52:47.273487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1704.00675","last_updated":"2018-03-01T17:50:08Z","snapshot_observed_at":"2026-08-02T10:51:13.194643Z","submitted_at":"2017-04-03T16:44:46Z","title":"The 2017 DAVIS Challenge on Video Object Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.00675","snapshot_observed_at":"2026-08-07T11:52:47.338977Z","title":"The 2017 davis challenge on video object segmentation","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.338977Z"},"links":{"cited_paper":"/paper/1704.00675","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:e1c678cbd98c54c152820a7a5d3a5ec1e4652882686e2c14e914dba08c51957b","observation_id":"54246d1f-937a-4143-8f44-ca248c0a50a6","resolution":{"observed_at":"2026-08-07T11:52:47.338977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.935987Z","title":"Disentangling spatial and temporal learning for efficient image-to-video transfer learning","venue":null,"work_id":"2e9ba308-3634-4e9f-8c81-78dd258d543a","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.428687Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:2e66848f752ed1365a329f1f911d05dc26b492c3cddac2f78ffa72e6ccf51b7e","observation_id":"5ba348c6-d502-4d91-869a-c166418a4c9c","resolution":{"observed_at":"2026-08-07T11:52:52.962159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:47.507661Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.507661Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:2acf04413d8b1459b9e3f506b40ca02a9da79afeef16b34349ff30be4d5587bc","observation_id":"4cdc9513-ab3e-4027-a34e-7f0167211ff8","resolution":{"observed_at":"2026-08-07T11:52:47.507661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.01197","last_updated":"2023-12-03T23:57:43Z","snapshot_observed_at":"2026-07-06T15:49:47.586513Z","submitted_at":"2023-07-03T17:58:01Z","title":"Segment Anything Meets Point Tracking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.01197","snapshot_observed_at":"2026-08-07T11:52:47.613125Z","title":"Segment anything meets point tracking.arXiv:2307.01197, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.613125Z"},"links":{"cited_paper":"/paper/2307.01197","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:a780731fc34163986b33db0754420c204c06187ece8940370cc92a4e7d8bb9d6","observation_id":"54567cfc-c79c-4116-b667-2137cf7a270a","resolution":{"observed_at":"2026-08-07T11:52:47.613125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-07T11:52:47.689544Z","title":"Sam 2: Seg- ment anything in images and videos.arXiv:2408.00714,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.689544Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:e7860381dd3d81368c28e4870414bd8a7d9815e4318afe52a4661e5c41504bf1","observation_id":"32d05b03-b271-4e21-ad9f-ba434b00d35e","resolution":{"observed_at":"2026-08-07T11:52:47.689544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.543220Z","title":"Seg- ment anything, from space? InWACV, 2024","venue":null,"work_id":"8734ce5c-9455-4e7a-8fe1-e30d8cbb28ac","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.882759Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:e3a56c893365408fedef394925491003f50a50c6363f5be0eab11c0f772d440a","observation_id":"b46505a9-ab4f-4412-a50f-b4457b8763be","resolution":{"observed_at":"2026-08-07T11:52:52.634283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.424864Z","title":"Learning fast and robust target models for video object segmentation","venue":null,"work_id":"5f59319b-8202-44fe-81df-0c3b10d91c8b","year":2020},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.977450Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:7e12e44d7ae49d6d726019f66dfc485a7e36809f4ad12945ee3cf115408c9202","observation_id":"d2d55a5b-8207-4979-97ea-d2284c541ef0","resolution":{"observed_at":"2026-08-07T11:52:52.441859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.02635","last_updated":"2025-01-06T03:00:07Z","snapshot_observed_at":"2026-08-07T10:08:12.416376Z","submitted_at":"2024-08-05T16:58:56Z","title":"Interactive 3D Medical Image Segmentation with SAM 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.02635","snapshot_observed_at":"2026-08-07T11:52:48.066282Z","title":"Interactive 3d medical image segmentation with sam 2.arXiv:2408.02635, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.066282Z"},"links":{"cited_paper":"/paper/2408.02635","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:758bbfb80e04f7b6038a6831608fa903b0ef3baef6f45943c9ef7f0e1525cdfd","observation_id":"6367c98a-127d-4bb1-8f02-dc5c67f8f2ba","resolution":{"observed_at":"2026-08-07T11:52:48.066282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13789","last_updated":"2025-01-08T10:01:20Z","snapshot_observed_at":"2026-07-06T17:06:29.162729Z","submitted_at":"2023-12-21T12:26:11Z","title":"TinySAM: Pushing the Envelope for Efficient Segment Anything Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.13789","snapshot_observed_at":"2026-08-07T11:52:48.139202Z","title":"Tinysam: Pushing the envelope for efficient segment any- thing model.arXiv:2312.13789, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.139202Z"},"links":{"cited_paper":"/paper/2312.13789","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:c1117f0931536fb4247a2d31b32668d64ac3605f21de8b5f516ce282d2bb17cb","observation_id":"0af786a1-4ab5-4cc8-94af-fd5a658cad6c","resolution":{"observed_at":"2026-08-07T11:52:48.139202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.352881Z","title":"Towards open-vocabulary video instance segmentation","venue":null,"work_id":"022dbc8c-2559-465a-ac5a-4216eb14fae4","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.219346Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:5de1ce2f20a82056d642ab934d76d3f4d1506ed8cc1b189c4cf1494fcd58fa16","observation_id":"145770c0-a7d4-449f-ad6e-75da32188b50","resolution":{"observed_at":"2026-08-07T11:52:52.416992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.064899Z","title":"F3net: Fu- sion, feedback and focus for salient object detection","venue":null,"work_id":"c8cb5369-5fc3-4917-82bd-4dc60029365d","year":2020},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.404865Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:90c7f3979915348dc7f21a9c45e7850c9d167b6d05518663305c9ebdcb097a95","observation_id":"a28ac603-12cc-45d1-b433-35d72d35297e","resolution":{"observed_at":"2026-08-07T11:52:52.133490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.12620","last_updated":"2023-12-29T03:40:59Z","snapshot_observed_at":"2026-07-06T15:19:39.223171Z","submitted_at":"2023-04-25T07:34:22Z","title":"Medical SAM Adapter: Adapting Segment Anything Model for Medical Image Segmentation","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.12620","snapshot_observed_at":"2026-08-07T11:52:48.476606Z","title":"Medical sam adapter: Adapting segment anything model for medical image segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.476606Z"},"links":{"cited_paper":"/paper/2304.12620","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:2a1f323e92694ccae052bf6967c9f6895513035ff0607615cc3580e3efa824e1","observation_id":"8fa47d21-827d-415a-8d7b-019097e79d86","resolution":{"observed_at":"2026-08-07T11:52:48.476606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.900798Z","title":"Scalable video object segmentation with simplified frame- work","venue":null,"work_id":"0e0954a2-fb2e-475b-868e-27fadc441cc1","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.565837Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:d93d0177989fd554d7f01da3634440401935a3854b7e35467c37d74d059cd26f","observation_id":"f66ab37f-aead-4a37-b22e-914904281cd9","resolution":{"observed_at":"2026-08-07T11:52:51.963083Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.879578Z","title":"Cat-sam: Con- ditional tuning for few-shot adaptation of segment anything model","venue":null,"work_id":"c6817636-a9be-471c-ab06-a5b6d0821b02","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.671961Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:0e20882098f25da25016800e2fc412bada27eea7dc56e8bdc9c74f32f233eb5f","observation_id":"3cf8432a-bead-4b98-9543-340adc1bda27","resolution":{"observed_at":"2026-08-07T11:52:51.886849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:48.758340Z","title":"Sam2-unet: Segment anything 2 makes strong encoder for natural and medical image segmentation.arXiv:2408.08870,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.758340Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:c9ad921ac9c21c901350d51df6f2f51f79eab1f2899f1bf98e53b32a30eb0c88","observation_id":"2d21e214-7128-4169-94a9-69c759c75ee2","resolution":{"observed_at":"2026-08-07T11:52:48.758340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.862521Z","title":"Efficientsam: Leveraged masked image pretraining for efficient segment anything","venue":null,"work_id":"583c761c-2927-4ab9-abc2-d261561d0350","year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.859621Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:b712925414980643295f4b826709832e1e86da7f5d190656deb2ed7fe6d31e51","observation_id":"f6ec5bef-6694-4bca-b495-7d1b0097f919","resolution":{"observed_at":"2026-08-07T11:52:51.871711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.507461Z","title":"Youtube-vos: Sequence-to-sequence video object segmentation","venue":null,"work_id":"9ad873af-0954-4cb9-be90-a224b49ddb32","year":2018},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.966528Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:f139afcfe41eba309243bda60c05ec3262f666cd6049afd93e09d4dee321623d","observation_id":"66dd7019-f941-415a-b79a-1d07b6480998","resolution":{"observed_at":"2026-08-07T11:52:51.629829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11968","last_updated":"2023-04-28T03:21:27Z","snapshot_observed_at":"2026-08-09T08:50:57.337514Z","submitted_at":"2023-04-24T10:04:06Z","title":"Track Anything: Segment Anything Meets Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11968","snapshot_observed_at":"2026-08-07T11:52:49.056362Z","title":"Track anything: Segment anything meets videos.arXiv:2304.11968, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.056362Z"},"links":{"cited_paper":"/paper/2304.11968","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:fcb3c66508125bda8f2c78384066f75121c129a36cd271c934caf9e437be3f72","observation_id":"9ac3aeba-a833-4a41-8e9e-70ccdfed6daa","resolution":{"observed_at":"2026-08-07T11:52:49.056362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.215096Z","title":"Efficient video object segmen- tation via network modulation","venue":null,"work_id":"0a9564ae-bb80-4f26-b4d0-c9268a7820c7","year":2018},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.155216Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:4f8f9909fdd8923e3d4d7ec1659a2d0d3b50cd00fba43b8e76813a1078ce42f7","observation_id":"960e79a4-35db-4e07-83ec-aca882a801dc","resolution":{"observed_at":"2026-08-07T11:52:51.325468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:51.107857Z","title":"AIM: Adapting image models for efficient video action recognition","venue":null,"work_id":"4ea9a138-d8c0-490c-94a6-c9628adb61e4","year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.257921Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:516262ab605b311f142e1b4e210498f854964107f5b2622903167f33f580df71","observation_id":"71beb037-fafc-43c3-8e60-8a94f15c8c43","resolution":{"observed_at":"2026-08-07T11:52:51.156717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:50.961851Z","title":"Collaborative video object segmentation by foreground-background inte- gration","venue":null,"work_id":"467fd5c5-9cd2-4c34-9775-8af0ed92ab05","year":2020},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.351850Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:95a218d482def31ee972629684458028c217f18b53d1e2f6c0f8540222556869","observation_id":"907785b0-b6af-41a1-98b3-a28ef3db9675","resolution":{"observed_at":"2026-08-07T11:52:51.066669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14289","last_updated":"2023-07-01T07:26:22Z","snapshot_observed_at":"2026-08-08T16:17:33.420400Z","submitted_at":"2023-06-25T16:37:25Z","title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14289","snapshot_observed_at":"2026-08-07T11:52:49.455141Z","title":"Faster segment anything: Towards lightweight sam for mo- bile applications.arXiv:2306.14289, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.455141Z"},"links":{"cited_paper":"/paper/2306.14289","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:7ab02cafa38476e18e9487fba4c50a2a6f8edfba2c3d5e91ab4513e5b95114c9","observation_id":"5404a6f2-5757-4575-aca4-3a46ccf51623","resolution":{"observed_at":"2026-08-07T11:52:49.455141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.09579","last_updated":"2023-12-15T07:21:12Z","snapshot_observed_at":"2026-08-09T19:51:26.770539Z","submitted_at":"2023-12-15T07:21:12Z","title":"MobileSAMv2: Faster Segment Anything to Everything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.09579","snapshot_observed_at":"2026-08-07T11:52:49.547609Z","title":"Mobilesamv2: Faster segment anything to everything.arXiv:2312.09579,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.547609Z"},"links":{"cited_paper":"/paper/2312.09579","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:3d07ef6693e0b5fbff3d738af526fadd09e5a19788e2630f05187cfadcaec63a","observation_id":"b14219f4-0ea2-41f4-8266-4983cc0e55d0","resolution":{"observed_at":"2026-08-07T11:52:49.547609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.13505","last_updated":"2025-04-30T07:19:18Z","snapshot_observed_at":"2026-07-06T16:10:31.087739Z","submitted_at":"2023-08-25T17:30:08Z","title":"Joint Modeling of Feature, Correspondence, and a Compressed Memory for Video Object Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.13505","snapshot_observed_at":"2026-08-07T11:52:49.642417Z","title":"Joint modeling of feature, correspondence, and a compressed memory for video object segmentation.arXiv:2308.13505,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.642417Z"},"links":{"cited_paper":"/paper/2308.13505","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:bd0102b399a0d2a7144f8433b20dba161585e6593346c18faa879aa77cc6a867","observation_id":"2ee8bb8f-c9f7-4a37-b36b-eaaa7dc82e06","resolution":{"observed_at":"2026-08-07T11:52:49.642417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.12156","last_updated":"2023-06-21T10:08:29Z","snapshot_observed_at":"2026-07-06T15:45:01.961896Z","submitted_at":"2023-06-21T10:08:29Z","title":"Fast Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.12156","snapshot_observed_at":"2026-08-07T11:52:49.758396Z","title":"Fast segment any- thing.arXiv:2306.12156, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.758396Z"},"links":{"cited_paper":"/paper/2306.12156","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:5111a002e549dbfc6bebc1467a75e7ac19a937090a07ffeb34c28a51401d1223","observation_id":"a027915a-2f96-4f65-8176-e2d4128ae780","resolution":{"observed_at":"2026-08-07T11:52:49.758396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06660","last_updated":"2025-09-07T19:09:41Z","snapshot_observed_at":"2026-07-06T17:00:00.321552Z","submitted_at":"2023-12-11T18:59:52Z","title":"EdgeSAM: Prompt-In-the-Loop Distillation for SAM","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06660","snapshot_observed_at":"2026-08-07T11:52:49.890149Z","title":"Edgesam: Prompt-in-the-loop distillation for on-device de- ployment of sam.arXiv:2312.06660, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.890149Z"},"links":{"cited_paper":"/paper/2312.06660","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:ac7b771fb0931d1cc68671489e952591e5172597c7022f931df3793f6a18d36a","observation_id":"19b37adb-16b0-4d3f-ad7a-f5ceaa97255e","resolution":{"observed_at":"2026-08-07T11:52:49.890149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00874","last_updated":"2024-12-04T23:51:25Z","snapshot_observed_at":"2026-08-04T09:38:10.661882Z","submitted_at":"2024-08-01T18:49:45Z","title":"Medical SAM 2: Segment medical images as video via Segment Anything Model 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00874","snapshot_observed_at":"2026-08-07T11:52:49.909797Z","title":"Medical sam 2: Seg- ment medical images as video via segment anything model 2.arXiv:2408.00874, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.909797Z"},"links":{"cited_paper":"/paper/2408.00874","citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:edc41821d8b4d91afcc594b85794a123026b89df40a99bb5dc11a5d72d197074","observation_id":"d6722ea7-d27d-4d4b-a968-83bc9046d004","resolution":{"observed_at":"2026-08-07T11:52:49.909797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:50.686759Z","title":"-” indicates directly combining existing pre-trained models for inference. “Cost","venue":null,"work_id":"6254e01a-83c2-4054-9032-8d2cb93ca836","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:49.988237Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:fa2b15376168b3644f1a4d4f9bae0c09536b6b2d02ffdbdc367c5693b46d31f2","observation_id":"e62b112e-18b2-4ce8-8a43-a0b8f4e6efd4","resolution":{"observed_at":"2026-08-07T11:52:50.813473Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.202306Z","title":null,"venue":null,"work_id":"1251ac23-8615-4c17-bb76-dd20024903f3","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:48.320982Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:086393f8283ce1dffad2d242004c402cc126bd570b2ab976443aa77320456776","observation_id":"56a8c35a-5af4-4fc7-8199-6c8eecbe5760","resolution":{"observed_at":"2026-08-07T11:52:52.280421Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:52:52.723673Z","title":null,"venue":null,"work_id":"c7b0fd38-1e4a-4537-a1d2-1bff6b80c13f","year":null},"citing_paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:47.760620Z"},"links":{"citing_paper":"/paper/2506.01304"},"observation_digest":"sha256:f3a289089646bce2187b4902181c8f3da51bc69885e33057e235581ce21dfa19","observation_id":"4a59372e-94cc-41ba-9a79-00605792d2b9","resolution":{"observed_at":"2026-08-07T11:52:52.837961Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.01304","last_updated":"2025-06-02T04:30:14Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T02:38:04.027009Z","submitted_at":"2025-06-02T04:30:14Z","title":"SAM-I2V: Upgrading SAM to Support Promptable Video Segmentation with Less than 0.2% Training Cost"},"reference_resolution":{"displayed":61,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":26,"verified_exact":1,"verified_fuzzy":33},"total_outbound_references":61},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 61 of 61 outbound references and 0 inbound Pith citation observations for arXiv:2506.01304."}