{"as_of":"2026-08-11T07:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ad397b1ec4e2e2da76059b78f1620e2b933ccdf9e6725a14df7d5911238a0e86","coverage":[{"denominator":61,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T15:47:59.927188Z","state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.13667/citation-record","integrity":"/paper/2501.13667/integrity","json":"/paper/2501.13667/citation-record.json","paper":"/paper/2501.13667"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:01.031022Z","title":"Xmem++: Production-level video segmentation from few annotated frames","venue":null,"work_id":"ab990034-e7aa-4c77-bb36-e8175c4cb0f3","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.638768Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:9e25ffb5d87c0209b280cf4964ee2ffaefcd48b88734ad000bdac45c13eff452","observation_id":"4e1c3f6c-bbc6-4ba5-8b13-9314fc777a8e","resolution":{"observed_at":"2026-08-10T15:48:01.036615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.00263","last_updated":"2020-10-01T09:10:53Z","snapshot_observed_at":"2026-08-10T00:39:27.297853Z","submitted_at":"2020-10-01T09:10:53Z","title":"RefVOS: A Closer Look at Referring Expressions for Video Object Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.00263","snapshot_observed_at":"2026-08-10T15:47:59.644137Z","title":"Refvos: A closer look at referring expressions for video object segmen- tation.arXiv preprint arXiv:2010.00263, 2020","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.644137Z"},"links":{"cited_paper":"/paper/2010.00263","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:7f892a1a22db8799d70eae782db9bc52bf671775bd22dd4a74b897eacfb323f6","observation_id":"bbc84f67-177f-4a4d-9f71-db45046f149d","resolution":{"observed_at":"2026-08-10T15:47:59.644137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:01.013515Z","title":"End-to-end referring video object segmentation with multi- modal transformers","venue":null,"work_id":"b51fa50d-b93e-4244-875f-2fbe6f9e143b","year":2022},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.649698Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:37234241516a03e7796cb6dcc363bb67c3787f3a3226a4a32482745aa0749a08","observation_id":"6bf26444-f89a-4366-b158-ec586a8a771a","resolution":{"observed_at":"2026-08-10T15:48:01.019502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.995932Z","title":"End-to-end referring video object segmentation with multi- modal transformers","venue":null,"work_id":"3d4a1a25-89b9-4156-b3af-bc333cd2b8b9","year":2022},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.654829Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:0cd414daee84bb1b833cf8a7307d61dfbdd55a00b9e8ba487e5be576e7174857","observation_id":"fdc8ee81-920a-4426-9562-f90f532b5172","resolution":{"observed_at":"2026-08-10T15:48:01.001502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.978813Z","title":"End- to-end object detection with transformers","venue":null,"work_id":"1c4b89c2-7f12-4897-b9e8-0f621ec60afb","year":2020},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.659696Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:1eb76f8f9e26d7058a71f6a2ae79af24237a2efe6a4cf355bcfce57ae72efdc9","observation_id":"987ce447-7a12-4674-bc10-519771e68145","resolution":{"observed_at":"2026-08-10T15:48:00.984032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.961807Z","title":"Xmem: Long- term video object segmentation with an atkinson-shiffrin memory model","venue":null,"work_id":"fb027db8-329c-4a61-8a01-b894e77fb9a8","year":2022},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.664751Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:ff1c62768fd320772daff5d9da375288db80bf89b3a4dbbc2d1eaa5be2d599b2","observation_id":"755fb4b8-5a47-4de1-9e55-e8b1463f784c","resolution":{"observed_at":"2026-08-10T15:48:00.966930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.944242Z","title":"Putting the object back into video object segmentation","venue":null,"work_id":"20e6c225-7ab9-48a7-bbf7-a7e550b122a9","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.670130Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:2de884b793855cf13cb658adb11a433add03b71b41c70b2c5cbbe8c46a358441","observation_id":"ff279dd1-d879-4b85-adb0-7c78be7d3270","resolution":{"observed_at":"2026-08-10T15:48:00.950457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06558","last_updated":"2023-05-11T04:33:08Z","snapshot_observed_at":"2026-08-03T19:49:16.800693Z","submitted_at":"2023-05-11T04:33:08Z","title":"Segment and Track Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06558","snapshot_observed_at":"2026-08-10T15:47:59.674573Z","title":"Segment and track anything.arXiv preprint arXiv:2305.06558, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.674573Z"},"links":{"cited_paper":"/paper/2305.06558","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:28481cae3ec9869469429f93c985268eb17e4b9374aedf7eee40f6cf38ddc643","observation_id":"4546e337-6ea6-4a03-8f0c-ec6d3e3a81d0","resolution":{"observed_at":"2026-08-10T15:47:59.674573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1911.02116","last_updated":"2020-04-08T01:02:17Z","snapshot_observed_at":"2026-08-02T09:09:10.011185Z","submitted_at":"2019-11-05T22:42:00Z","title":"Unsupervised Cross-lingual Representation Learning at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.02116","snapshot_observed_at":"2026-08-10T15:47:59.679455Z","title":"Unsupervised cross-lingual representation learning at scale.arXiv preprint arXiv:1911.02116, 2019","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.679455Z"},"links":{"cited_paper":"/paper/1911.02116","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:715ab286487a9b59b19dc6a900d4cec771af70475d325d55f41885d0630c4786","observation_id":"055a0632-966b-491b-95e1-cc2a2fbb4493","resolution":{"observed_at":"2026-08-10T15:47:59.679455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.926973Z","title":"Vision-language transformer and query generation for refer- ring segmentation","venue":null,"work_id":"878ff9ab-9cf9-4375-863b-4da1e69ac13a","year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.684899Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:74d77d81453a6cd94e4809465c0c6cc86d8c02c085834849c110de47a3c98cbc","observation_id":"ff018f14-f355-42b4-b98e-04a998bfec9a","resolution":{"observed_at":"2026-08-10T15:48:00.932309Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.908046Z","title":"Mevis: A large-scale benchmark for video segmentation with motion expressions","venue":null,"work_id":"8a80f355-90a7-41ea-bb44-13106c6d22b2","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.690233Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:ad746891a0ce1ff5213f88a219c07d546fd4af5fae3002efeaf6efd4b17e22ca","observation_id":"0fa86d86-2c75-4729-a592-455d0adf2496","resolution":{"observed_at":"2026-08-10T15:48:00.913517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.891826Z","title":"Language-bridged spatial-temporal interaction for referring video object segmentation","venue":null,"work_id":"ec35246f-7850-408a-ad50-511b6e2c84fd","year":2022},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.695176Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:130bb11a57153e9afbf424decbe922cebddc7e9ef725ea44757ef099613f7bce","observation_id":"5971daea-c441-47f0-b780-85b500c13e4d","resolution":{"observed_at":"2026-08-10T15:48:00.896976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.875855Z","title":"Unified embedding alignment for open-vocabulary video instance segmentation","venue":null,"work_id":"cf992954-c995-423a-ac52-145184ac65df","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.700026Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:41d08f9b8da60bcbbb05581feeec735914e33a15e77bd314017d2b3025d35898","observation_id":"f94ac849-ef94-41d6-ab2c-b0ef10281cb4","resolution":{"observed_at":"2026-08-10T15:48:00.881133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.859906Z","title":"Html: Hybrid temporal-scale mul- timodal learning framework for referring video object seg- mentation","venue":null,"work_id":"01449708-d1e9-4a95-810c-f0ed4c0c5067","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.704741Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:373726f30cbd958ce890314a7b8cc10f3d9762d936e0286f913b29456b41aa0e","observation_id":"64500324-77b2-49a1-b524-4c5921b9300e","resolution":{"observed_at":"2026-08-10T15:48:00.865317Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.843057Z","title":"Decoupling static and hier- archical motion perception for referring video segmentation","venue":null,"work_id":"a080f141-2f3a-4c27-a393-318d65f509e7","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.709388Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:66d01d6c88c49c8f2bc8f88e2794cd9dbace9cae0a9a834ded2f9eda50fe3e93","observation_id":"160ea41c-b019-473c-9647-601d3cdd12c5","resolution":{"observed_at":"2026-08-10T15:48:00.848532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15876","last_updated":"2024-12-23T08:10:30Z","snapshot_observed_at":"2026-08-09T17:26:46.735554Z","submitted_at":"2024-08-28T15:47:32Z","title":"Unleashing the Temporal-Spatial Reasoning Capacity of GPT for Training-Free Audio and Language Referenced Video Object Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15876","snapshot_observed_at":"2026-08-10T15:47:59.714033Z","title":"Unleashing the temporal-spatial reasoning capacity of gpt for training-free audio and language referenced video object segmentation.arXiv preprint arXiv:2408.15876, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.714033Z"},"links":{"cited_paper":"/paper/2408.15876","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:37308340138ea81a2769021165f5f00d6e5bb1fb06e44581ca8abb5ee8f76df8","observation_id":"7161f8c8-2472-4e51-8763-388d1fb61b66","resolution":{"observed_at":"2026-08-10T15:47:59.714033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.718988Z","title":"Segment anything in high qual- ity.Advances in Neural Information Processing Systems, 36,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.718988Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:698f2afab9892d408b55c2f144878827a17ab589f8601cbefbacb4bd36882980","observation_id":"69c6c784-e421-4ba8-806c-7309be4bb745","resolution":{"observed_at":"2026-08-10T15:47:59.718988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.809057Z","title":"Video object segmentation with language referring expressions","venue":null,"work_id":"af3a1bb4-d5ae-4eda-8524-9f06ae0bfe7d","year":2018},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.723933Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:8fae4071a3954cc25104814001867b6a61f393c43eb29f2b78a53c2de53747c6","observation_id":"dfe47324-1f8c-4d7a-aaad-9288711f0ff0","resolution":{"observed_at":"2026-08-10T15:48:00.815303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.792973Z","title":"Segment any- thing","venue":null,"work_id":"7ca258a8-01ed-4ccf-a610-2363e57a1378","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.728435Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:a32a42c3dc493a26f4028423d1c2f029cdc8d8a1a80dd51323787a6182724be5","observation_id":"4e3d5cfb-376c-4f68-a950-94455ae29c78","resolution":{"observed_at":"2026-08-10T15:48:00.797935Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.775526Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":"65d16c4d-35d8-452d-9cb1-ed3dd86af795","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.733316Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:bdb2d692f6bf02fb821d532412396b4a686e25744e691678c415aaed67525000","observation_id":"b89bf872-4ce4-4000-89c9-2bd0ba6b7222","resolution":{"observed_at":"2026-08-10T15:48:00.780736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.759769Z","title":"Learning to learn better for video object segmentation","venue":null,"work_id":"5327e26d-d0f9-49c5-bd71-612d5993fbd0","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.737900Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:9c25cdf797701b626fb5fd4b6785523ca62b1e7c0ab05588954521fad2f062ab","observation_id":"ced4a7ee-c08d-48ca-af59-a115b9104a63","resolution":{"observed_at":"2026-08-10T15:48:00.764725Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.744169Z","title":null,"venue":null,"work_id":"9790b4f4-881c-4c89-89ba-08f4c8be6d09","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.742330Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:46b735ada7b7a6097a2707d7c5cb4b0e8ea880cf50f54fbc57edb91a93f100ab","observation_id":"328add4e-aec4-408f-9b64-772b9545be10","resolution":{"observed_at":"2026-08-10T15:48:00.749283Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.728462Z","title":"Bidirectional correlation-driven inter-frame inter- action transformer for referring video object segmentation","venue":null,"work_id":"35918d2c-a36a-4d7c-ba03-44b12de51c71","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.747025Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:bbd28c0d3e274c7dde76f9a61f688c8e12b7ea06926a4e3dbb5758f463ec11c0","observation_id":"5b791fd6-2d71-4e4c-a4a1-6985029cb2a2","resolution":{"observed_at":"2026-08-10T15:48:00.733566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.712885Z","title":"You only infer once: Cross-modal meta-transfer for referring video object segmentation","venue":null,"work_id":"02a7424d-6162-4e41-a379-cbf42e8ecbe4","year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.751536Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:2ee8a51bd6d62a678babc75657cc58d690902db2b91f669b7c845bd91085d0a3","observation_id":"9701dfa4-8e8e-4eef-886e-cd73a0ea8d1c","resolution":{"observed_at":"2026-08-10T15:48:00.717969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.00997","last_updated":"2024-09-03T07:25:51Z","snapshot_observed_at":"2026-08-06T19:45:17.476087Z","submitted_at":"2023-07-03T13:21:58Z","title":"RefSAM: Efficiently Adapting Segmenting Anything Model for Referring Video Object Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.00997","snapshot_observed_at":"2026-08-10T15:47:59.756198Z","title":"Refsam: Efficiently adapting segmenting any- thing model for referring video object segmentation.arXiv preprint arXiv:2307.00997, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.756198Z"},"links":{"cited_paper":"/paper/2307.00997","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:e016022594483e24472a82f2f1495cbb20aba05c4b44535be024f561370526f0","observation_id":"e5c6324e-ed1f-44bc-9d23-12fb89929b46","resolution":{"observed_at":"2026-08-10T15:47:59.756198Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.761037Z","title":"Visual instruction tuning.Advances in neural information processing systems, 36, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.761037Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:31532c7c3b63c1610e7fae2e09a6d3882b65ce6c1408d32e6eff41658974d3d9","observation_id":"3e9f7077-2c38-4e6e-99f5-159afcd2d098","resolution":{"observed_at":"2026-08-10T15:47:59.761037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-10T15:47:59.765578Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection.arXiv preprint arXiv:2303.05499, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.765578Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:2e620a319c412bd4e03be9d448e00f769aa28ca547135c2c13d2bad44a9a848e","observation_id":"fd5bc5ec-19e1-40f6-b456-e6f50d648946","resolution":{"observed_at":"2026-08-10T15:47:59.765578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.770551Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.770551Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:cb7b8c248aea17c6a6dcb230c96ee1d35521654fad43308c6d6641d53f4dfbae","observation_id":"230a054c-9723-4a4a-93ea-8bf4aac4c325","resolution":{"observed_at":"2026-08-10T15:47:59.770551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.677172Z","title":"Soc: Semantic-assisted object cluster for referring video object segmentation.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"de74a9c3-3a11-4ed1-a2c9-80e7e4cc7623","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.775051Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:1f22348a214fced4804c40e68faa6a0ae0b095382d6ee0cd93d0101e6c6da417","observation_id":"d947f84c-2012-4c4a-94d4-3e539f1415bd","resolution":{"observed_at":"2026-08-10T15:48:00.682289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.779445Z","title":"Generation and comprehension of unambiguous object descriptions","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.779445Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:83d896ced3108c116b60babfe3fc70106a935dc2b5d953f3bac1e0b34ba9c917","observation_id":"50a38455-8280-4a0f-9131-321f32e3234f","resolution":{"observed_at":"2026-08-10T15:47:59.779445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.650871Z","title":"Visual-textual capsule routing for text-based video segmentation","venue":null,"work_id":"39a76a48-caac-40ff-b514-6d052f7de5ee","year":2020},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.784047Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:08248759282434ccb968d77ddace436de45e01c4c43d58dba871fb8b2542cc01","observation_id":"bd1b8c88-a717-449d-a81c-135e74c8e4ab","resolution":{"observed_at":"2026-08-10T15:48:00.655677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.635363Z","title":"Spectrum-guided multi-granularity referring video object segmentation","venue":null,"work_id":"b92db058-377a-4ee7-937f-7fabd02554c9","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.788681Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:f83987940672be791d77fac7685f86a96b8ccecce0ff68d0eecd037165a85cfb","observation_id":"38e25771-d08d-407a-a24a-19c3b8c25675","resolution":{"observed_at":"2026-08-10T15:48:00.640795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.619907Z","title":"V-net: Fully convolutional neural networks for volumetric medical image segmentation","venue":null,"work_id":"98f17e2f-ea00-4d12-84ca-b2b7ec1db621","year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.793025Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:619230e71ecbaf92e8585f2e7743a7a96a093b34e644c6d387d11b7f2be2a306","observation_id":"04df44cf-137c-43d2-9023-03e4e5a2394f","resolution":{"observed_at":"2026-08-10T15:48:00.624776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.604201Z","title":"Video object segmentation using space-time memory networks","venue":null,"work_id":"97d27f07-7733-4e6c-abc7-43bd4a8e339f","year":2019},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.797817Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:305ddb1d2d3a9f8992225ef2619fdb78cab75f3e46f6989da4149207b1406293","observation_id":"ad7d8cdc-6449-4416-bddc-15badaeee255","resolution":{"observed_at":"2026-08-10T15:48:00.609152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.588013Z","title":"Semantic and sequential alignment for referring video object segmentation","venue":null,"work_id":"4e4fe5b6-8902-416b-873a-66e18f4b8197","year":2025},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.802324Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:9674ac74ecd85b3b486630548cc84803c6e7646b144d8c0e1863e23b9929a1df","observation_id":"b476880e-13e2-4bc3-ba23-53d3216bf272","resolution":{"observed_at":"2026-08-10T15:48:00.593023Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.00675","last_updated":"2018-03-01T17:50:08Z","snapshot_observed_at":"2026-08-02T10:51:13.194643Z","submitted_at":"2017-04-03T16:44:46Z","title":"The 2017 DAVIS Challenge on Video Object Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.00675","snapshot_observed_at":"2026-08-10T15:47:59.807087Z","title":"The 2017 davis challenge on video object segmentation.arXiv preprint arXiv:1704.00675, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.807087Z"},"links":{"cited_paper":"/paper/1704.00675","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:132347d6f06a85ace1f5a803cfcd01886684c90e995445b6f8e4f9c884157cff","observation_id":"a2d4b4e7-3df5-46c6-88d9-7eb0fb05c1c2","resolution":{"observed_at":"2026-08-10T15:47:59.807087Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.811931Z","title":"Glamm: Pixel grounding large multimodal model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.811931Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:e134ff649e5c7f868ea8dd2c7b21b7f5634675df17526761affd301549e251c7","observation_id":"e384604e-d98d-4289-b000-5cd9177d1b1e","resolution":{"observed_at":"2026-08-10T15:47:59.811931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-10T15:47:59.817076Z","title":"Sam 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.817076Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:edeb88ca7185b322e235fa367a5d931d184d3f3a183331f490db3999c88cdfe2","observation_id":"36430e9f-d503-4473-b3e4-ed7f044aac35","resolution":{"observed_at":"2026-08-10T15:47:59.817076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.821941Z","title":"Cus- tomized sam 2 for referring remote sensing image segmenta- tion.arXiv preprint arXiv:2503.07266, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.821941Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:dc7300a1fc14403ce8052951a82695cadd70ec86f3682489708446382cdb86b8","observation_id":"1dd84681-af88-483f-aac6-521cc1ddbc2e","resolution":{"observed_at":"2026-08-10T15:47:59.821941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.562920Z","title":"Urvos: Unified referring video object segmentation network with a large-scale benchmark","venue":null,"work_id":"4dddb96e-f91d-403d-b63f-f44041a038c7","year":2020},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.826708Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:72a8b73e24c8575ec76aa7b278814e005bdc8b9d5feb9cb02275c4d2ce259e9f","observation_id":"92ed7359-2cfb-4b8d-985e-b3e434a333e0","resolution":{"observed_at":"2026-08-10T15:48:00.567693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.547743Z","title":"Temporal collection and distribution for referring video object segmentation","venue":null,"work_id":"9103b88e-1063-40ed-9222-06296f3e6278","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.831198Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:d6aefe895683d4bf2eeee8b282c7f7487c3c6aeb0a57fb40a3eabfd51a888dab","observation_id":"94e8d278-d03b-4049-8de5-aca91dbf4e0b","resolution":{"observed_at":"2026-08-10T15:48:00.552878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.532861Z","title":"Samrs: Scaling-up re- mote sensing segmentation dataset with segment anything model.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"f3f82495-706d-4e25-b873-3c8ab643ac8f","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.835790Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:a9774338a61f277f5fb21150a1576338e17999032cd069b9f6df9d89c370ab59","observation_id":"3065cf5b-7afd-4526-833f-13bb09643fd2","resolution":{"observed_at":"2026-08-10T15:48:00.537906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.515607Z","title":"Asymmetric cross-guided attention network for actor and ac- tion video segmentation from natural language query","venue":null,"work_id":"8965fc92-46d6-4ca6-9dd4-feea4152b7a3","year":2019},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.840270Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:639dad524cd05943e5c23a18937709d0e4d25b89bec3dc08a1e2472cf757cbbd","observation_id":"a1d6c23c-9577-466b-b38a-327dcc6a4f30","resolution":{"observed_at":"2026-08-10T15:48:00.521970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.499414Z","title":"Image as a foreign language: Beit pretraining for vision and vision- language tasks","venue":null,"work_id":"a07344ed-9ea5-4ba5-8f90-b22c4c899ede","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.845067Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:d521ec1f8f912ae4f2e8037f838563dda7979662458bd2bd0202ebd923745bea","observation_id":"5959649d-ce2c-46b4-9e05-e8c3ca604f26","resolution":{"observed_at":"2026-08-10T15:48:00.504760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17606","last_updated":"2024-12-02T03:19:04Z","snapshot_observed_at":"2026-08-09T10:15:00.225568Z","submitted_at":"2024-11-26T17:18:20Z","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17606","snapshot_observed_at":"2026-08-10T15:47:59.849646Z","title":"Hyperseg: Towards univer- sal visual segmentation with large language model.arXiv preprint arXiv:2411.17606, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.849646Z"},"links":{"cited_paper":"/paper/2411.17606","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:1f6dfb168d8929989f7686b8e0628db354303e58903e7230e71ff42c0d9d6c72","observation_id":"8f1e15ea-a785-4cc4-ace7-b3b1e43c0986","resolution":{"observed_at":"2026-08-10T15:47:59.849646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.482983Z","title":"Multi-level representation learning with semantic alignment for referring video object segmentation","venue":null,"work_id":"8766dfd0-3656-4dfb-b9ca-61f6d676c214","year":2022},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.854413Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:2f5386f723ba13b372e1989fcccd992d292cf882f49f77ec1724361c717f01e9","observation_id":"99a2b93d-69c9-4020-8819-a18dc653a03a","resolution":{"observed_at":"2026-08-10T15:48:00.488432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.466345Z","title":"Onlinerefer: A simple online baseline for referring video object segmentation","venue":null,"work_id":"9e2bf597-263e-4a99-9a64-7da22d80a0cc","year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.858650Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:1f06b54ce3897dc5b2e67ad1a154397f65b893013c01defdda2e86b4d486818a","observation_id":"dcaacb23-6094-40d8-a4eb-12d4d8299c9d","resolution":{"observed_at":"2026-08-10T15:48:00.471795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.450579Z","title":"Language as queries for referring video object segmen- tation","venue":null,"work_id":"77c7d988-ddab-4e4d-81e9-ff11b04341af","year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.863235Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:989bd158d7c981a509fe0d70aca826469b7b7dcbca56341c6746571a64ad1504","observation_id":"27dad8b8-5237-45d9-b191-642f246c9a2d","resolution":{"observed_at":"2026-08-10T15:48:00.455585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.433653Z","title":"Logiczsl: Exploring logic- induced representation for compositional zero-shot learning","venue":null,"work_id":"aad39284-cb7d-4db5-ad50-e148deda8e5a","year":2025},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.868031Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:efcb0e791ccae2f013c2d27a2ac912540351f27cce8ae6a32890e546ca696221","observation_id":"af69f5b1-cba9-4ea5-bb47-dfc980093eef","resolution":{"observed_at":"2026-08-10T15:48:00.438701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.417449Z","title":"Efficientsam: Leveraged masked image pretraining for efficient segment anything","venue":null,"work_id":"89bb53a8-4b28-4370-9a79-3ebb2b8e2828","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.873004Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:fc29832b99244fbd9b493e0ef508440be74c111f386be50c99bb1b099264786a","observation_id":"3690eaf6-802f-4159-9288-f20a2e6b9847","resolution":{"observed_at":"2026-08-10T15:48:00.422489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.05348","last_updated":"2024-08-28T14:26:07Z","snapshot_observed_at":"2026-08-10T14:48:57.090454Z","submitted_at":"2023-11-09T13:18:27Z","title":"u-LLaVA: Unifying Multi-Modal Tasks via Large Language Model","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.05348","snapshot_observed_at":"2026-08-10T15:47:59.877607Z","title":"u-llava: Uni- fying multi-modal tasks via large language model.arXiv preprint arXiv:2311.05348, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.877607Z"},"links":{"cited_paper":"/paper/2311.05348","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:7dc7a28d8685f8dfaf898a1f49e64e1bdc2ce9eaeb9e49a0ee4763ccc3bb5d62","observation_id":"d52ffb01-c86d-4286-9866-779efd545e76","resolution":{"observed_at":"2026-08-10T15:47:59.877607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.400756Z","title":"Visa: Reasoning video object segmentation via large language models","venue":null,"work_id":"99834d2f-208d-44a8-bbbb-5d0f6430c26e","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.882837Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:5448e1b016672cb389a5181e4b268d97dc551f6225ed7d24b25606b70ee1e82d","observation_id":"de3c20d8-e016-42fd-8746-f16c4ab22481","resolution":{"observed_at":"2026-08-10T15:48:00.405801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.384247Z","title":"Referred by multi-modality: A unified tem- poral transformer for video object segmentation","venue":null,"work_id":"598d03fb-81f0-4730-adc2-e3fa7e16cce5","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.887488Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:bc0575857df128ffc56e599995f8c68a53cf8c93b3477c118f38d1a6509e9e61","observation_id":"f36f6979-76b4-4c7c-9717-e82879d29075","resolution":{"observed_at":"2026-08-10T15:48:00.390008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.367370Z","title":"Modeling context in referring expres- sions","venue":null,"work_id":"5b05721c-31c2-4449-9318-2b925f409c88","year":2016},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.892095Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:ea04a2ebb891f30c2080655f2ce03a6a68f82a44fa734e571dd2638932d3e0ab","observation_id":"39fab523-0cbf-4cde-a3ea-496bc9ffb894","resolution":{"observed_at":"2026-08-10T15:48:00.372685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15521","last_updated":"2025-06-17T06:34:15Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T04:14:01Z","title":"A Simple Baseline with Single-encoder for Referring Image Segmentation","version":3},"cited_work":{"arxiv_id":"2408.15521","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.15521","snapshot_observed_at":"2026-08-10T15:47:59.998003Z","title":"A Simple Baseline with Single-encoder for Referring Image Segmentation","venue":"cs.CV","work_id":"53d639f0-c4ba-4522-b0cb-edd8c2ae3901","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.897025Z"},"links":{"cited_paper":"/paper/2408.15521","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:517b73c1830637422246317d1f64ec4e233d8d296cce98d2cc31005678debc4c","observation_id":"cc420157-1065-4e3b-a348-dada5e697f3f","resolution":{"observed_at":"2026-08-10T15:48:00.007696Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.901850Z","title":"Losh: Long-short text joint prediction network for referring video object segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.901850Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:99ad2fecd011cbbf85ed9c22440d9553d28dbdeaa3b99ea7e4abad768e7e613d","observation_id":"959aed02-6d41-47d3-9e66-7763d8836a05","resolution":{"observed_at":"2026-08-10T15:47:59.901850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.906540Z","title":"Surgicalsam: Efficient class prompt- able surgical instrument segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.906540Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:4350eeebcb07b05caa7a3ca8c035b75e224d9d2d60f9621aed9cc0972ba8c0fe","observation_id":"015c654e-2242-41cf-a9d4-161540a8f6cb","resolution":{"observed_at":"2026-08-10T15:47:59.906540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14289","last_updated":"2023-07-01T07:26:22Z","snapshot_observed_at":"2026-08-08T16:17:33.420400Z","submitted_at":"2023-06-25T16:37:25Z","title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14289","snapshot_observed_at":"2026-08-10T15:47:59.911548Z","title":"Faster segment anything: Towards lightweight sam for mo- bile applications.arXiv preprint arXiv:2306.14289, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.911548Z"},"links":{"cited_paper":"/paper/2306.14289","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:11220cc7409a10f612c0b209771606d12694bde991a69e6c079ca65684806310","observation_id":"42fa24c1-7262-43ff-885e-fd50ec4fb259","resolution":{"observed_at":"2026-08-10T15:47:59.911548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.20076","last_updated":"2025-03-10T12:34:24Z","snapshot_observed_at":"2026-08-06T23:39:09.398813Z","submitted_at":"2024-06-28T17:38:18Z","title":"EVF-SAM: Early Vision-Language Fusion for Text-Prompted Segment Anything Model","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.20076","snapshot_observed_at":"2026-08-10T15:47:59.916575Z","title":"Evf- sam: Early vision-language fusion for text-prompted seg- ment anything model.arXiv preprint arXiv:2406.20076,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.916575Z"},"links":{"cited_paper":"/paper/2406.20076","citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:166805236b9611fc0728cc022c64b35fec6a6338b2bbd17e0176cce899cdd347","observation_id":"747b87e0-fc36-405d-a4e0-52a21ae4e60d","resolution":{"observed_at":"2026-08-10T15:47:59.916575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:47:59.922407Z","title":"Deformable detr: Deformable transformers for end-to-end object detection","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.922407Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:92114758bc24999ebf24161408439a74318085d2880e6235aa8874796663c20a","observation_id":"a569ecdb-9f60-4755-9229-2bedc8ca6b6c","resolution":{"observed_at":"2026-08-10T15:47:59.922407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:48:00.320204Z","title":"Exploring pre-trained text- to-video diffusion models for referring video object segmen- tation","venue":null,"work_id":"3b9fc9f1-ac7a-4741-8dd6-8591355af344","year":2024},"citing_paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation","version":5},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T15:47:59.927188Z"},"links":{"citing_paper":"/paper/2501.13667"},"observation_digest":"sha256:48c4e5a82ea2561f094a5e58f98e5996cd342ff0f859305eb06f1ab9a371f214","observation_id":"0cd8a6d0-c9f7-4798-b2df-3d797451fb83","resolution":{"observed_at":"2026-08-10T15:48:00.326524Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.13667","last_updated":"2025-08-08T10:17:56Z","latest_version":5,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-11T04:49:08.862771Z","submitted_at":"2025-01-23T13:53:33Z","title":"MPG-SAM 2: Adapting SAM 2 with Mask Priors and Global Context for Referring Video Object Segmentation"},"reference_resolution":{"displayed":61,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":22,"verified_exact":1,"verified_fuzzy":38},"total_outbound_references":61},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 61 of 61 outbound references and 0 inbound Pith citation observations for arXiv:2501.13667."}