{"as_of":"2026-08-06T13:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:04571fe0ffcf549c341c087677485230e08f657642eb41bdd8473f1a2c1ef340","coverage":[{"denominator":68,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":68,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-09T20:47:06.698475Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.00891/citation-record","integrity":"/paper/2605.00891/integrity","json":"/paper/2605.00891/citation-record.json","paper":"/paper/2605.00891"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:597a75bed4726c4ab4a9333f6503cfabcdcb8c809556bd8363ad994405f514fa","observation_id":"189a533b-936a-4955-8d15-8da0e87348e6","resolution":{"observed_at":"2026-05-11T15:01:05.310910Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:8bfa1a27c6f7a39cdcad776f8053f0033ceb51460cf49ec7ed09a44d4b33382f","observation_id":"ee46ec15-f3da-40f0-beb6-71859a1d303e","resolution":{"observed_at":"2026-05-11T15:01:05.240526Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T06:20:45.322418Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"29916027-cee6-4d61-a66a-57ca905e76a2","year":2021},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:a670a6ada194ad85d9d1505e5b469b00adda29a50a3d7a895df344e4f3c45212","observation_id":"112e7e39-cdcb-4a74-bed1-3d0328ae86cd","resolution":{"observed_at":"2026-05-24T08:29:12.536243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision","venue":null,"work_id":"3c4a825c-5c2d-4f1c-a084-243ea28aabde","year":2021},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:3957fcbfab46b93e26cd5e348e2662973672690ed79a53a0773277cbf3ee8d84","observation_id":"d661328e-2fc3-4b90-9f66-18185220cd93","resolution":{"observed_at":"2026-05-24T08:29:12.498229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Show, attend and tell: Neural image caption generation with visual attention","venue":null,"work_id":"3221b5c8-8821-470c-93d3-800f43cf3ee1","year":2048},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:8fb1a4284ed7651fce9d1bc3f6b7c24920308892168c543ec2cc4a157f087e4b","observation_id":"1a9c4b47-0ff5-4ef6-9da4-ce5691db9d97","resolution":{"observed_at":"2026-05-24T08:29:12.539604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vqa: Visual question answering","venue":null,"work_id":"1b99ce10-2448-4b58-b823-1153f873ca28","year":2015},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:5f0a1c26a0ab7de16c400e2aae1bee1e4df2dc037df649a3a410f288de417ec5","observation_id":"efbb2c68-ac56-46c8-9834-5c011cd288bd","resolution":{"observed_at":"2026-05-24T08:29:12.494715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Language-based image editing with recurrent attentive models","venue":null,"work_id":"dc24d993-5c54-4bf6-ba9b-2a5ddf272a22","year":2018},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:b10291bc2eb10b7b15834ad75c6a431d5bb62b12ac5e584a552bffee39d4898d","observation_id":"8a666af3-d6f6-40ee-95c4-935b067aeaff","resolution":{"observed_at":"2026-05-24T08:29:12.501296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T03:06:43.908517Z","title":"Segment anything","venue":null,"work_id":"ffd866b4-b8b4-4924-887d-af662dde5bdc","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:ab296f3cd5b3693640eaccc9d9893edb215711e47d3354a77d11a4551f53341e","observation_id":"73d5b524-410c-4d1c-a816-5c329c2a46cb","resolution":{"observed_at":"2026-05-24T08:29:12.504364Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":"2408.00714","doi":"10.1038/s41598-025-97590-3","metadata_source":"pith","pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SAM 2: Segment Anything in Images and Videos","venue":"cs.CV","work_id":"acc13f66-d814-44f9-9688-375688bf2d4a","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:d0148bf9a53330ce422a847af92310285e9a63688af66e707337a565f06e7b64","observation_id":"57511c89-3be4-4704-b86d-a1b3c87186d1","resolution":{"observed_at":"2026-05-11T15:01:05.356359Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-24T04:24:23.885301+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T04:24:23.885301+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T03:06:43.966629Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":"65ea6a8f-17d6-47e9-b126-0d1ea82dd36d","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:09cb30d01dee4bfcc59bbd8f6e949686fed3e0d8a1601c46b0f7de3d7901a9d5","observation_id":"9a1001aa-4c7b-42f2-8239-3d622cc1ab50","resolution":{"observed_at":"2026-05-24T08:29:12.510978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visa: Reasoning video object segmentation via large language models.ECCV","venue":null,"work_id":"c6100c36-0123-4359-95ba-f6f2d57f708b","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:10c71fd9ca0637bf8d735f358c366bc967c55bf0df2715b857c0326c4dc01f85","observation_id":"83a0d30e-093f-4cc7-8039-969fd5a5b955","resolution":{"observed_at":"2026-05-24T08:29:12.476693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"One token to seg them all: Language instructed reasoning segmentation in videos.NeurIPS","venue":null,"work_id":"c94215c2-ecae-4551-9d4d-f74517a26f01","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:0f51759127669c7f4ac5b2a47e93716c5d592edc535223af5a938b5c89af178e","observation_id":"d02ec514-18dc-494d-abc9-69eba5c3511d","resolution":{"observed_at":"2026-05-24T08:29:12.480003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","venue":null,"work_id":"60f9ecec-cb13-494f-a80b-2d0c37388b82","year":2022},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:5a9b9619f05408522f047cb501405480d5c6e7917ef2ec1f6d1abbcbe7d5ec1b","observation_id":"ef49ba74-329f-4077-92a8-5222ed275bfb","resolution":{"observed_at":"2026-05-24T08:29:12.483298Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"6607c1a9-ce29-49ca-a51d-8415fc83531d","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:c28ba831d23f8ce5bbc5bb30746b6f32d1773578aa56b6138b1c59de14a3cf04","observation_id":"843c713f-225c-47ee-b7f7-0f8ab3275bbc","resolution":{"observed_at":"2026-05-24T08:29:12.514451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tarvis: A unified approach for target-based video segmentation","venue":null,"work_id":"5ee4e5ef-960c-4917-85cf-195e17ebdce9","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:55df0c84e274b561baa033daf2977fac446ed9aafc2c741c69dfd49b0a1dd9d6","observation_id":"c95311b6-83b1-40e2-8254-c9c2a49a22b4","resolution":{"observed_at":"2026-05-24T08:29:12.521963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T03:06:43.904462Z","title":"Oneformer: One transformer to rule universal image segmentation","venue":null,"work_id":"33a3f729-9e29-446a-9bc3-a10f800c1e8c","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:42713ebe7057bc849f4cc1f543664981c3cf7eb7cfdc8cee73dc97a4425dbe11","observation_id":"aeba86b3-ffcb-4e68-b3b7-56a9ca1126c1","resolution":{"observed_at":"2026-05-24T08:29:12.528451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Omg-seg: Is one model good enough for all segmentation? InCVPR, pages 27948–27959","venue":null,"work_id":"d20cf456-0613-40f5-8a3e-564095158699","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:6f0b9b142510c6536780fa8b4f5a0171dd0cdc49a311c61cf0ab08442ac7abe3","observation_id":"2df7d21c-1b2b-4b3b-9f7c-1ade3bd75d29","resolution":{"observed_at":"2026-05-24T08:29:12.682610Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Omg-llava: Bridging image-level, object-level, pixel-level reasoning and understanding.NeurIPS, 37:71737–71767","venue":null,"work_id":"b47c8ff8-7d92-4212-afbb-c996ab03f9b4","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:0be7835c3f575d80e65472c00a894aeace4168bf72aa4b67ee1a1ac1b0b41dc0","observation_id":"c7c79173-c57b-4342-bb2f-8493a602bb72","resolution":{"observed_at":"2026-05-24T08:29:12.678888Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Temporal memory attention for video semantic segmentation","venue":null,"work_id":"c58da8dc-8406-49d4-a6e7-4be03316f9e5","year":2021},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:519a2008f7bcda49edddccbfdfbfe57ae67376d7634bc92ea5b0c7ecfadd7fe8","observation_id":"00024a58-bc61-40ad-8391-b81693f6e39e","resolution":{"observed_at":"2026-05-24T08:29:12.686154Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video k-net: A simple, strong, and unified baseline for video segmentation","venue":null,"work_id":"b6b258b1-780a-4a19-89b3-f9c7bf9c9834","year":2022},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:f0d2fdcf9106723aadf0a538467f585a628564b24ff66f2a27651123daaece46","observation_id":"f512585c-0589-43c6-ae74-50191e885586","resolution":{"observed_at":"2026-05-24T08:29:12.693255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"X-sam: From segment anything to any segmentation","venue":null,"work_id":"355cb31e-e86c-4312-9332-0d1ad26e54ee","year":2026},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:bb17b9bfa8942e02d0413e205ebb15a8d27012040a9b26dd6ab4b7c3503a7d97","observation_id":"af656a47-b134-4013-b0d3-5e5c5ed71abe","resolution":{"observed_at":"2026-05-24T08:29:12.667366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual instruction tuning.NeurIPS, 36:34892–34916","venue":null,"work_id":"b6ac582f-2fb8-433c-a94c-a857187b3025","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:821c2bfe477802b02e7d596c8736f8acdb1bcefa600ade6abe5761303a369803","observation_id":"6974a5d1-5310-4dd2-baeb-57ed19f590ea","resolution":{"observed_at":"2026-05-24T08:29:12.671074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llavanext: Improved reasoning, ocr, and world knowledge","venue":null,"work_id":"582bb26f-85eb-46d9-968c-f0df911b400e","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:9faf2346781311d5591011b6c2377fc02cd97eec1eb98dd6e0c1ee4a733e224f","observation_id":"64660bb0-4f1c-4366-924a-f0b000bbab47","resolution":{"observed_at":"2026-05-24T08:29:12.518515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":"2408.03326","doi":"10.48550/arxiv.2408.03326","metadata_source":"pith","pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","venue":"cs.CV","work_id":"f5f2452b-f2a9-49ac-b38d-c76e18cdfe49","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:c55734b69c96d860de0bf576a8ab5ee63cef0fd96bcd66430b1ebebc14578c79","observation_id":"02ce4c79-32d9-49c6-941c-5a1e0530e1d5","resolution":{"observed_at":"2026-05-11T15:01:05.327718Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How far are we to gpt-4v? closing the gap to commercial multimodal models with open-source suites.Science China Information Sciences, 67(12):220101","venue":null,"work_id":"d11e14a2-3976-4af5-bdfd-ad5f9b29d04a","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:f909b33370416379f7089aeb477c2b8e3f19d863d00e1990cad48c42c29e7675","observation_id":"c1ff8307-044d-4a7e-be60-6a2c5bfcf9cf","resolution":{"observed_at":"2026-05-24T08:29:12.660214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":"2308.12966","doi":"10.48550/arxiv.2308.12966","metadata_source":"pith","pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","venue":"cs.CV","work_id":"cbc2bb21-b6bb-46c0-80bf-107e195ffe10","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:5df895560af8ab00676b7148609c9db5e8194d4871d5897dea19ccf857df5394","observation_id":"896b17eb-d5e7-4d8f-939c-172841dd903a","resolution":{"observed_at":"2026-05-11T15:01:05.456688Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Glamm: Pixel grounding large multimodal model","venue":null,"work_id":"992d95c7-e1dd-4dc3-b674-76ad8ce85854","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:6350ec94cfe5f8a25e3dde0f1b3a4ca56881447a5685b23cac502a0e208d550d","observation_id":"8a919a35-e3c3-4889-8b0a-33b4989eb4b0","resolution":{"observed_at":"2026-05-24T08:29:12.652775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videoglamm: A large multimodal model for pixel-level visual grounding in videos.CVPR","venue":null,"work_id":"55f94a5c-8b29-48cf-9169-557dcc0e3d3b","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:3a3769d175d26f422c31edfe4b9b38475dcc36de5973f7fcc837329e4638de54","observation_id":"0196f2a1-bdb6-4125-8449-43c5cd76b280","resolution":{"observed_at":"2026-05-24T08:29:12.656287Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Psalm: Pixelwise segmentation with large multi-modal model","venue":null,"work_id":"9466dc49-670d-4d72-9dab-0e0bacc55ba5","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:619b48ad012c8ccdd31b4b01add24c4ee4ea77705d00f68b1329ea4f03f89179","observation_id":"af0980ee-7629-4667-9e72-759526988977","resolution":{"observed_at":"2026-05-24T08:29:12.663736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17606","last_updated":"2024-12-02T03:19:04Z","snapshot_observed_at":"2026-07-06T19:57:24.407368Z","submitted_at":"2024-11-26T17:18:20Z","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","version":2},"cited_work":{"arxiv_id":"2411.17606","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.17606","snapshot_observed_at":"2026-07-03T00:47:30.621522Z","title":"Hyperseg: Towards universal visual segmentation with large language model","venue":null,"work_id":"027e4858-ce79-47ad-a12b-af062f5af5e8","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2411.17606","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:d294cd8c4e526c88fa320ac37cda8b5a704013da824fab21f6310ffe8a1aabaf","observation_id":"bba4b529-0457-41d8-84ff-7f7bc8d4facf","resolution":{"observed_at":"2026-05-11T15:01:05.436176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04001","last_updated":"2025-11-03T17:35:29Z","snapshot_observed_at":"2026-07-06T20:17:47.599899Z","submitted_at":"2025-01-07T18:58:54Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","version":3},"cited_work":{"arxiv_id":"2501.04001","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.04001","snapshot_observed_at":"2026-07-10T03:06:43.598138Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","venue":"cs.CV","work_id":"fbffda10-106c-432c-b84e-5d0908666921","year":2025},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2501.04001","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:198703fdf4865ade7c301af2ef36370eadd07a47e5e35a44bf44f1fb42776f3e","observation_id":"b81759f9-1147-4d4a-819b-121ec2ec0a0b","resolution":{"observed_at":"2026-05-16T11:39:22.805579Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:0e8b1afec102a99e7b2299610445533b4168c5c4744a8f9935a0b372f54a1600","observation_id":"f2b0c637-ef50-418c-992d-c7bd6da00cb1","resolution":{"observed_at":"2026-05-11T15:01:05.383742Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07704","last_updated":"2023-10-11T17:55:15Z","snapshot_observed_at":"2026-07-06T16:31:25.350087Z","submitted_at":"2023-10-11T17:55:15Z","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","version":1},"cited_work":{"arxiv_id":"2310.07704","doi":"10.48550/arxiv.2310.07704","metadata_source":"pith","pith_arxiv_id":"2310.07704","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","venue":"cs.CV","work_id":"0db2bf55-f649-4428-bcc9-688724b38e57","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2310.07704","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:3071878cc69d7abf4d84334de2e9b3f63e1bce71bc5af2bfddc7f8180e3c94c4","observation_id":"ec346914-1640-4eef-afb5-371ccabdbf0e","resolution":{"observed_at":"2026-05-15T11:29:05.422888Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V-net: Fully convolutional neural networks for volumetric medical image segmentation","venue":null,"work_id":"12597ee5-09e9-42f6-8d6a-fd1a722ea03b","year":2016},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:d0d32c843b11bb5a1adbb9c7a728d184295a8905408b064845f0672d6a839e66","observation_id":"426e0b0d-f9b4-4e99-9fbf-3abad408de99","resolution":{"observed_at":"2026-05-24T08:29:12.674761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improving language understanding by generative pre-training","venue":null,"work_id":"fd186a44-bd70-482d-9601-58322941000c","year":2018},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:25cda972aac71fd8eada0535a5aca43272864be1d1c8abaad92af12e844b6c50","observation_id":"0e9620a1-d871-43b7-bbab-e92b044516c7","resolution":{"observed_at":"2026-05-24T08:29:12.635492Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Focal loss for dense object detection","venue":null,"work_id":"700c0af1-cbc3-41c3-811d-c357113b3512","year":2017},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:0713cd35fcdc770c2167d20be52e07e3155d48f4ff1ba8c984d738be78bd2153","observation_id":"48ae41de-27a5-4684-b2ce-c0d10650fb2b","resolution":{"observed_at":"2026-05-24T08:29:12.639720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Large-scale video panoptic segmentation in the wild: A benchmark","venue":null,"work_id":"37cb660a-dc12-4d6a-a070-5157bf32de15","year":2022},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:7450bfff7495587b07b7fe3419d69f5c23aaa35f4266773e6c0496889de189ee","observation_id":"5714d776-dbe0-48c9-a429-e69de5135a60","resolution":{"observed_at":"2026-05-24T08:29:12.627025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vspw: A large-scale dataset for video scene parsing in the wild","venue":null,"work_id":"133d7e3d-919b-4adb-8138-691544507a51","year":2021},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:01335074e5bb659a87727c092ccdb3cfb6a263cafe1f0d2ab09313804c0f125b","observation_id":"ecd7e4e6-cd0b-47d2-be74-ce892fff8d39","resolution":{"observed_at":"2026-05-24T08:29:12.619200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video instance segmentation","venue":null,"work_id":"b1c34287-1582-4990-b7c1-a7f55d6f54cc","year":2019},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:215458320bc20c326f90779ad4296f7cbac44e081cf6a434ebb793433c121e3f","observation_id":"b59f63e8-6e66-4f52-ba88-c9b6cacba9d5","resolution":{"observed_at":"2026-05-24T08:29:12.622865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Urvos: Unified referring video object segmentation network with a large-scale benchmark","venue":null,"work_id":"ff3cb254-72b3-476f-9dca-69812d6feecc","year":2020},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:e8fe2bf34951f178017d1cc9ab46f0644befc15e71cb232f8de89504b2be909b","observation_id":"9cf3bdca-9cc3-4380-b93c-f4fa1c233029","resolution":{"observed_at":"2026-05-24T08:29:12.631094Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.03327","last_updated":"2018-09-06T04:19:45Z","snapshot_observed_at":"2026-07-06T07:00:12.122241Z","submitted_at":"2018-09-06T04:19:45Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","version":1},"cited_work":{"arxiv_id":"1809.03327","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.03327","snapshot_observed_at":"2026-07-08T02:04:26.310872Z","title":"YouTube-VOS: A Large-Scale Video Object Segmentation Benchmark","venue":"cs.CV","work_id":"2798c43d-6f6c-445b-8a43-14cba092aa4b","year":2018},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/1809.03327","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:7a47c64585ac8893b6f6421d4c54086e21cd9af27dd9d656abff420579432c19","observation_id":"2cb0fd27-26fd-4a0c-a948-00e046bcd75b","resolution":{"observed_at":"2026-05-11T15:01:05.428048Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A benchmark dataset and evaluation methodology for video object segmentation","venue":null,"work_id":"b7e1cc71-1363-4ca4-ac3f-6bacaff2991e","year":2016},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:b56a1be912d14713018e84509a6581de31b0c185adc3f6584de44942515e6e6f","observation_id":"1b3a79e1-793e-47b5-bfbb-8acd67dd255f","resolution":{"observed_at":"2026-05-24T08:29:12.642997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":"50c5e35b-7241-4f6b-835b-23a10d4e3583","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:88e64fb28a16b9e6c9dd9a10baf88cca64f21167105f6ad88b3b56c08a6a6ed1","observation_id":"979af393-7e8b-4cc5-8da4-0f631655a1df","resolution":{"observed_at":"2026-05-24T08:29:12.646842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gres: Generalized referring expression segmentation","venue":null,"work_id":"139d2186-2474-4d0e-a70b-3c3467b8aa65","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:c7cdd7a6257c33a13b9988f72a49928d4ad1b9a23616df2f81c30a5c5da86546","observation_id":"4ff49862-0d0c-4d00-bc18-6d5cc73c35f3","resolution":{"observed_at":"2026-05-24T08:29:12.689595Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Semantic understanding of scenes through the ade20k dataset.IJCV, 127(3):302–321","venue":null,"work_id":"70f23762-21f8-4ad2-b2cd-63124f3bc430","year":2019},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:9d2f0890508be09a08614bebc6bc805d3dcce8e4b99aab2d17fa38013d60f5d1","observation_id":"c65917f6-a72f-48a0-9a66-37df14f5e1ea","resolution":{"observed_at":"2026-05-24T08:29:12.696795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"LoRA: Low-rank adaptation of large language models","venue":null,"work_id":"a656fe0c-1358-466a-b49d-80501ef18d57","year":2022},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:7d3b3324b90f36364176180b09baa4f89d6887e3d031e4e6a0331056c67a97ba","observation_id":"a8702369-aeaf-4897-814e-cf6494515560","resolution":{"observed_at":"2026-05-24T08:29:12.598446Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":"1711.05101","doi":"10.1137/1.9781611972825.47","metadata_source":"pith","pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Decoupled Weight Decay Regularization","venue":"cs.LG","work_id":"07ef7360-d385-4033-83f7-8384a6325204","year":2017},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:1d40e1f04b329c1ef7cbb43259f49c33b32d99701c02cef19cfabc4985e1e20b","observation_id":"d1bf84a3-8b50-4e07-bb31-8e73d1a805c3","resolution":{"observed_at":"2026-05-11T15:01:05.228126Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Open-vocabulary panoptic segmentation with text-to-image diffusion models","venue":null,"work_id":"fffdd936-cc5f-4258-abc8-203485def900","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:0407a4e1de6ddd3228603e7be94fa07ee3032fd330c5dd00a176bb283414088b","observation_id":"185942e1-e13f-4040-8fc0-7373101d478f","resolution":{"observed_at":"2026-05-24T08:29:12.580826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Uniref++: Segment every reference object in spatial and temporal spaces.ICCV","venue":null,"work_id":"6a3cb09a-a8bb-4764-9832-a11bd105567d","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:3a7b41e3530831102bcbf0533a4f06913e863127db103a8db7df64dbcd46465f","observation_id":"5775c423-951a-4a6d-866a-f7911a22f73e","resolution":{"observed_at":"2026-05-24T08:29:12.486963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unipixel: Unified object referring and segmentation for pixel-level visual reasoning","venue":null,"work_id":"0042f3e7-6eaa-4eec-b91c-1bec99083f83","year":2025},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:d4e294c53ec79085bc1e1cc6f783982492ae4727753e959144da9bc3a5eb3f06","observation_id":"ddd436b2-739a-471c-bb30-79590767642d","resolution":{"observed_at":"2026-05-24T08:29:12.491118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Segment everything everywhere all at once.NeurIPS, 36:19769–19782","venue":null,"work_id":"0975b046-2f46-455c-a82d-db2424b81df6","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:1b1aa259dfb520e06decc5dd37b89d05df38cfd0ea881862bee83cab55a0a313","observation_id":"5442f6fb-7a08-4af4-be2b-a5640366d55e","resolution":{"observed_at":"2026-05-24T08:29:12.507529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Language as queries for referring video object segmentation","venue":null,"work_id":"6e59847c-8dc6-4814-a7d7-1b02d1214f19","year":2022},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:ea0d768eb74450abaa2f064e5893bd2e9f9c5c8bcdcbddb08abceb8da1a9253f","observation_id":"9bc49f70-74a6-4281-ae04-8ba5b8871e34","resolution":{"observed_at":"2026-05-24T08:29:12.525512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08506","last_updated":"2024-04-12T14:40:45Z","snapshot_observed_at":"2026-08-02T23:06:41.458983Z","submitted_at":"2024-04-12T14:40:45Z","title":"LaSagnA: Language-based Segmentation Assistant for Complex Queries","version":1},"cited_work":{"arxiv_id":"2404.08506","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.08506","snapshot_observed_at":"2026-07-08T23:55:43.060633Z","title":"LaSagnA: Language-based segmentation assistant for complex queries","venue":"cs.CV","work_id":"04d02749-2ed7-4d09-a611-d752d054bc4a","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2404.08506","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:096d7d5bce2fd52ae0d6dab36c2facc5d965e068943e92d1a2e028d14f07f864","observation_id":"2af7add8-b3e8-4b47-9326-6b86450a8b53","resolution":{"observed_at":"2026-05-11T15:01:05.418347Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mmbench: Is your multi-modal model an all-around player? InECCV, pages 216–233","venue":null,"work_id":"2ca49b6c-7fbb-45ab-af73-1f5dd30ea63a","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:a7d06850d03d918876e75d7bf17cc938bd764792c17783393aad0c6ee816e42c","observation_id":"fc734836-8251-4641-914b-d10e09e118b0","resolution":{"observed_at":"2026-05-24T08:29:12.532469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Seed-bench: Benchmarking multimodal large language models","venue":null,"work_id":"15dd8e8e-bf5c-49b0-9484-afd2b7794158","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:f6a4d2f8fc4e62f224a894e384ce95e957dceb27b265f52df1d06fc1050b05ef","observation_id":"f03881bb-577e-44d7-8196-ef4ba1862f66","resolution":{"observed_at":"2026-05-24T08:29:12.584571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10355","last_updated":"2023-10-26T02:52:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-17T16:34:01Z","title":"Evaluating Object Hallucination in Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2305.10355","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.10355","snapshot_observed_at":"2026-07-10T11:37:03.198858Z","title":"Evaluating Object Hallucination in Large Vision-Language Models","venue":"cs.CV","work_id":"66d8ac3e-c134-4995-b528-550afa17586f","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2305.10355","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:3b53240ac122e20c0cc1368860d8bab69acf364843fe7339e2ded336f9b24d0f","observation_id":"5fdedad0-1ec3-4043-96e2-c902833b1ce8","resolution":{"observed_at":"2026-05-11T15:01:05.291358Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A diagram is worth a dozen images","venue":null,"work_id":"e1741732-b8c3-4927-9699-8057004d77fc","year":2016},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:47dde56befe3305e3abaafb3d1872024c41e2d8bfb8ef7107dc94d007dd72386","observation_id":"9068ae36-3707-43b3-ac8a-39d266f118ab","resolution":{"observed_at":"2026-05-24T08:29:12.590763Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"c34c4776-44a4-4b45-8805-d166a50ebe3b","year":2014},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:9422da21f6633a2c3370b4386da9186a0ccc70fb20086abddffef5994a8772ed","observation_id":"d7dc8b0b-ed25-4f49-90b6-46dbdaae1da6","resolution":{"observed_at":"2026-05-24T08:29:12.574063Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":"5ce3c027-94e7-4be1-85d6-149065d236d0","year":2025},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:06e4bff793634106922b7483a4158d5f7f3d0ddd435faa926b6134d190ce02bd","observation_id":"01815b34-207c-46b7-a5d7-16668649f050","resolution":{"observed_at":"2026-05-24T08:29:12.566380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mvbench: A comprehensive multi-modal video understanding benchmark","venue":null,"work_id":"2509a7b0-1e43-4873-97f5-580863f27b4e","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:5f25041eac084ef455834141cddd45a146867fc43b812bb8d42dfd6e456fd3a6","observation_id":"671e29d7-b81b-4d47-b747-e2a39e9f0744","resolution":{"observed_at":"2026-05-24T08:29:12.570596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mlvu: Benchmarking multi-task long video understanding","venue":null,"work_id":"7a17de59-f606-4ca7-be01-cc6bb55c09a5","year":2025},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:53bf5bd43bbd32c7f85e61fec35401258d76402115c6ea119f08d4f90afb8e2b","observation_id":"9f13f2ab-6e5e-450e-92a2-663be295ff92","resolution":{"observed_at":"2026-05-24T08:29:12.577708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding.NeurIPS, 37:28828–28857","venue":null,"work_id":"ac932942-4b15-4ddb-a605-2ce3f0630338","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:bddb98b4c28e4d31ff2536b9e4e3080d7f22ccc079092bcff0c45df81972255f","observation_id":"85d37fd3-c47c-45a1-8d4b-29d6ce4dc423","resolution":{"observed_at":"2026-05-24T08:29:12.594663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T03:06:43.920377Z","title":"Masked-attention mask transformer for universal image segmentation","venue":null,"work_id":"bf98b5b2-dbd2-4dd6-b33f-e08cabbb5ad9","year":2022},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:2cfd7c1ec7105c609a0fd5c2087b0efa5621218199f4ba0edcdc8ad30a0b7ed9","observation_id":"03bffbe7-5c54-49a6-8635-8d9a376f551f","resolution":{"observed_at":"2026-05-24T08:29:12.608794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cris: Clip-driven referring image segmentation","venue":null,"work_id":"693b0746-4d03-4dbe-abcf-019c64798e35","year":2022},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:7b5f5e3431b14119aa6f20a2ec1bd2abd2969ff3d77ebbe8b1c482c4a7214a1f","observation_id":"0d3a54b7-25d4-4926-8063-01edbde880b2","resolution":{"observed_at":"2026-05-24T08:29:12.562547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.13435","last_updated":"2023-12-13T17:24:10Z","snapshot_observed_at":"2026-07-06T16:51:06.781541Z","submitted_at":"2023-11-22T14:48:30Z","title":"PG-Video-LLaVA: Pixel Grounding Large Video-Language Models","version":2},"cited_work":{"arxiv_id":"2311.13435","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.13435","snapshot_observed_at":"2026-06-28T19:22:34.433487Z","title":"Pg-video-llava: Pixel grounding large video-language models","venue":null,"work_id":"d3e76033-493a-4877-96aa-223793394c52","year":2023},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"cited_paper":"/paper/2311.13435","citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:ce1de37b4008a709ad0df7690e762ebd3d4267b6f3be0fc0ba1864f4c7872427","observation_id":"94abc434-0dfc-4bc3-ace5-c915712e418f","resolution":{"observed_at":"2026-05-11T15:01:05.448353Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videochat: Chat-centric video understanding.Science China Information Sciences, 68(10):200102","venue":null,"work_id":"365c0967-1b1a-4988-af69-b5b560a1f136","year":2025},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:54211cf6afaa6dd466aaa642a7acf39a02eddfec3cc4d3d0a586db972afdef6b","observation_id":"a2142cf6-256c-4bea-b154-a585cc137bd9","resolution":{"observed_at":"2026-05-24T08:29:12.552357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chat-univi: Unified visual representation empowers large language models with image and video understanding","venue":null,"work_id":"99b9fbc6-57d1-44a4-b147-c5df049afdac","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:9c670a9fbe07a2dfb8bb8fec78e3761225c919bc1046c70e23a1ab7a4451bf31","observation_id":"7feb229b-18d6-4698-89c7-3d85fe25ea84","resolution":{"observed_at":"2026-05-24T08:29:12.543460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vila: On pre-training for visual language models","venue":null,"work_id":"8832e4f3-62df-4b62-9cac-506f41c5f064","year":2024},"citing_paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-09T20:47:06.698475Z"},"links":{"citing_paper":"/paper/2605.00891"},"observation_digest":"sha256:96341006a05b6d4ac8bb4ecd714839ce095418f78a92dacac05af447ec1c707e","observation_id":"167756bb-13c3-4fc1-b365-064c3411a38c","resolution":{"observed_at":"2026-05-24T08:29:12.547625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.00891","last_updated":"2026-04-27T16:24:45Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-27T16:24:45Z","title":"X2SAM: Any Segmentation in Images and Videos"},"reference_resolution":{"displayed":68,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":13,"verified_fuzzy":54},"total_outbound_references":68},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 68 of 68 outbound references and 0 inbound Pith citation observations for arXiv:2605.00891."}