{"as_of":"2026-08-07T22:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:329e77a1156a58f8170c535782e48fbfb24b267e995c6ead4e955d9ecb8e9257","coverage":[{"denominator":89,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":89,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:28:14.907661Z","state":"measured"},{"denominator":104,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":104,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":15,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T16:51:24.717294Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T17:27:15.782237Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-06T16:51:24.717294Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.12441","last_updated":"2025-08-02T17:35:59Z","snapshot_observed_at":"2026-08-07T08:51:18.386431Z","submitted_at":"2025-07-16T17:28:19Z","title":"Describe Anything Model for Visual Question Answering on Text-rich Images","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T16:51:24.717294Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2507.12441"},"observation_digest":"sha256:043cd48806bd532bd9bfa5a6e3ad89f3275adc4926c0c43f1f574d45cd1ab170","observation_id":"1309521c-a60a-444f-ba0b-13861ed211f6","resolution":{"observed_at":"2026-08-06T16:51:24.717294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-05T20:29:11.675850Z","title":"Lin et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.10955","last_updated":"2025-08-14T07:25:45Z","snapshot_observed_at":"2026-08-07T18:23:08.889281Z","submitted_at":"2025-08-14T07:25:45Z","title":"Empowering Multimodal LLMs with External Tools: A Comprehensive Survey","version":1},"reference_index":293,"source":"arxiv_source","source_observed_at":"2026-08-05T20:29:11.675850Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2508.10955"},"observation_digest":"sha256:2dbb9d96179e6f92b6743109c5ee87c2c5960d5a321fd493544ccf0c805e9b62","observation_id":"54b3d678-10f0-4408-b6bb-06bdb6520ea5","resolution":{"observed_at":"2026-08-05T20:29:11.675850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-05T10:58:16.722899Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.03501","last_updated":"2025-09-03T17:33:20Z","snapshot_observed_at":"2026-08-07T14:11:47.550063Z","submitted_at":"2025-09-03T17:33:20Z","title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T10:58:16.722899Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2509.03501"},"observation_digest":"sha256:6dfa806711003fb33298ffc6ab8207bf0af0b4b5e19db30adbed1aa01bba6360","observation_id":"2b0f3896-8ba2-43c2-ac5a-aa020adf157d","resolution":{"observed_at":"2026-08-05T10:58:16.722899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-03T18:15:05.829843Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos.arXiv preprint arXiv:2506.05302, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.06276","last_updated":"2026-07-28T07:07:11Z","snapshot_observed_at":"2026-08-07T08:10:23.313193Z","submitted_at":"2025-12-06T03:59:21Z","title":"RefBench-PRO: Perceptual and Reasoning Oriented Benchmark for Referring Expression Comprehension","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T18:15:05.829843Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2512.06276"},"observation_digest":"sha256:a3e67ffb5290b8fe5ef36155e9a43c3d85edebee3d63c63b1b0a344ea182a6c6","observation_id":"5bdff3cf-8f46-4932-b4a7-efc5fd55bbdc","resolution":{"observed_at":"2026-08-03T18:15:05.829843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2512.07348","last_updated":"2026-04-28T10:02:14Z","snapshot_observed_at":"2026-08-06T12:32:11.849043Z","submitted_at":"2025-12-08T09:40:11Z","title":"MICo-150K: A Comprehensive Dataset Advancing Multi-Image Composition","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-17T00:20:58.483350Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2512.07348"},"observation_digest":"sha256:d2a93be0f323a5478f05eed664e55926bfb4e6967a35f4d476a7d2571f236329","observation_id":"64af0091-4638-4aaa-8598-57b8cbcfa627","resolution":{"observed_at":"2026-05-17T00:21:23.379690Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2512.12675","last_updated":"2026-06-09T11:29:25Z","snapshot_observed_at":"2026-08-03T16:37:02.565200Z","submitted_at":"2025-12-14T12:58:19Z","title":"Scone: Bridging Composition and Distinction in Subject-Driven Image Generation via Unified Understanding-Generation Modeling","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T22:39:32.955779Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2512.12675"},"observation_digest":"sha256:e2355077e0b1bf8f8f370ffeee64106f368341a9fddcccfe9ce8b278053623f5","observation_id":"f78f9f04-29ac-4021-89e9-4f639e9d59ea","resolution":{"observed_at":"2026-05-16T22:41:19.224758Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-08-03T16:37:05.085568Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos.arXiv preprint arXiv:2506.05302, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.12675","last_updated":"2026-06-09T11:29:25Z","snapshot_observed_at":"2026-08-03T16:37:02.565200Z","submitted_at":"2025-12-14T12:58:19Z","title":"Scone: Bridging Composition and Distinction in Subject-Driven Image Generation via Unified Understanding-Generation Modeling","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T16:37:05.085568Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2512.12675"},"observation_digest":"sha256:db264e4adb435338a7494c71a825ef8c654b7fd528a6d3b4b15e0769fc35b247","observation_id":"59ec8268-c8fa-4276-9aee-c8c8bb744edd","resolution":{"observed_at":"2026-08-03T16:37:05.085568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2602.03151","last_updated":"2026-04-06T04:51:44Z","snapshot_observed_at":"2026-07-06T22:44:18.875334Z","submitted_at":"2026-02-03T06:06:35Z","title":"Enhancing Foundation VLM Robustness to Missing Modality: Scalable Diffusion for Bi-directional Feature Restoration","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-16T08:33:49.841678Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2602.03151"},"observation_digest":"sha256:2c06843c31ceb0af3ad7dc60b798611b9db59ba2b04ae2b74b2c0ffdc4ea13a8","observation_id":"4d313a4e-0c65-4f54-b147-e665b3c3d17c","resolution":{"observed_at":"2026-05-16T08:37:37.229588Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2604.04707","last_updated":"2026-05-25T06:28:44Z","snapshot_observed_at":"2026-07-13T09:42:22.607962Z","submitted_at":"2026-04-06T14:19:48Z","title":"OpenWorldLib: A Unified Codebase and Definition of Advanced World Models","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-10T19:36:42.100191Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2604.04707"},"observation_digest":"sha256:93698570d377e7e70d34b24d4b91859e0482538ebbebc5c4cf508ee9e5d228eb","observation_id":"728775e5-abdc-40d7-84b4-a2b3a515999e","resolution":{"observed_at":"2026-05-10T22:45:49.048227Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-13T09:42:23.808691Z","title":"Perceive anything: Recognize, explain, caption, and segment anything in images and videos, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.04707","last_updated":"2026-05-25T06:28:44Z","snapshot_observed_at":"2026-07-13T09:42:22.607962Z","submitted_at":"2026-04-06T14:19:48Z","title":"OpenWorldLib: A Unified Codebase and Definition of Advanced World Models","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-07-13T09:42:23.808691Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2604.04707"},"observation_digest":"sha256:d905d375129a96a1c050991fc57f8c238e3f617419f87de73551406377e8ab7b","observation_id":"87f48f3a-8a30-4c2e-baec-3acd0e31e9bc","resolution":{"observed_at":"2026-07-13T09:42:23.808691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2604.11789","last_updated":"2026-04-20T14:38:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-13T17:55:02Z","title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","version":2},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-05-10T15:35:37.095627Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2604.11789"},"observation_digest":"sha256:c0a028a36e1d6c8c0d75bd2965f069806b67252cede469f88107dbe4150a4286","observation_id":"c25c039a-fdd0-4fdc-a0c6-8609bd93d72d","resolution":{"observed_at":"2026-05-11T10:11:06.525843Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2605.16903","last_updated":"2026-05-16T09:28:46Z","snapshot_observed_at":"2026-07-06T23:27:57.805018Z","submitted_at":"2026-05-16T09:28:46Z","title":"WOW-Seg: A Word-free Open World Segmentation Model","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-19T21:23:14.311122Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2605.16903"},"observation_digest":"sha256:940edd03f3fe96f6968367f60b2dc5956391920e7bac45f89bc6c3c7f5dfc8f0","observation_id":"fecadecc-bf3c-4c06-850e-e596473f3e66","resolution":{"observed_at":"2026-05-19T21:27:48.052668Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2605.18018","last_updated":"2026-05-18T08:09:37Z","snapshot_observed_at":"2026-07-06T23:28:55.079333Z","submitted_at":"2026-05-18T08:09:37Z","title":"See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-20T12:10:54.874012Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2605.18018"},"observation_digest":"sha256:77d6f768bc26e0ba8b0b08e929965d9859538aefc91867e5462661dd3bfbf02e","observation_id":"d1ce6cc2-ff2d-43fa-bcdf-2036aa9755d6","resolution":{"observed_at":"2026-05-20T12:13:16.316928Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":"2506.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-02T17:27:15.782237Z","title":"Perceive anything: Recog- nize, explain, caption, and segment anything in images and videos","venue":null,"work_id":"47327aa1-d59d-4d8f-93e4-19149de0935b","year":2025},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":114,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:c278e7fdebeff7e7a08211df010ccf79f5614dd3d8f3d17311bbb811d1583d39","observation_id":"694d2d02-f883-4042-9944-73eace34e136","resolution":{"observed_at":"2026-07-02T17:27:15.783743Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05302","snapshot_observed_at":"2026-07-11T21:31:04.773453Z","title":"Perceive anything: Recognize, explain, cap- tion, and segment anything in images and videos,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04125","last_updated":"2026-07-05T05:42:56Z","snapshot_observed_at":"2026-08-06T15:56:42.943701Z","submitted_at":"2026-07-05T05:42:56Z","title":"FRFDet: Efficient UAV Small Object Detection with Symmetric Sampling and Scalable Fusion","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-07-11T21:31:04.773453Z"},"links":{"cited_paper":"/paper/2506.05302","citing_paper":"/paper/2607.04125"},"observation_digest":"sha256:c164f4ce2b5a56842319431d82a7bf0b2fc06f3f8f8e770ec2b978e18ace9a9b","observation_id":"f067d0b1-ed72-4548-b86f-4a796ce217ed","resolution":{"observed_at":"2026-07-11T21:31:04.773453Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.05302/citation-record","integrity":"/paper/2506.05302/integrity","json":"/paper/2506.05302/citation-record.json","paper":"/paper/2506.05302"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:07.834487Z","title":"Mc-llava: Multi-concept personalized vision-language model.arXiv preprint arXiv:2411.11706, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:07.834487Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:836bba8b0b356dfb7c842477ff3cd0291f363caa29beb87b0649f8ec4e7a458f","observation_id":"5e6c0141-9f53-4a7a-9fb3-8161ec7a3f0d","resolution":{"observed_at":"2026-08-07T10:28:07.834487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-07T10:28:07.986779Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:07.986779Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:b963b0f3ada7f89e12dcf08c8759bcefff8aba118e7ad98cbc36097d470cb5da","observation_id":"9caa0b70-e2a3-4d5d-a811-fa791f80c4c2","resolution":{"observed_at":"2026-08-07T10:28:07.986779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.106718Z","title":"Meteor: An automatic metric for mt evaluation with improved correlation with human judgments","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.106718Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:b8346744b93b07fa004c9977b295f25d85312d435405673eec9deb5dbf7f3111","observation_id":"b44229d7-89fb-458d-934d-5bdf41176fcd","resolution":{"observed_at":"2026-08-07T10:28:08.106718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.211037Z","title":"Abductive commonsense reasoning, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.211037Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:0595bdfe2557cbfec4f05ac9af160d1c811048e6c00a1709cec6622e6dcb941a","observation_id":"78c355f2-42bb-4a4e-886c-ad97de098a34","resolution":{"observed_at":"2026-08-07T10:28:08.211037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.339094Z","title":"Graph cuts in vision and graphics: Theories and applications","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.339094Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:00d551ea8bb4c07bd9f2d66c574d1ef7cfe12c1a01a6e21774c12fb63f6cfd47","observation_id":"53317ec5-f5f6-4ddc-b81d-c892bb967dcc","resolution":{"observed_at":"2026-08-07T10:28:08.339094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.465002Z","title":"Scene-text oriented referring expression comprehension.IEEE Transactions on Multimedia, 25:7208–7221, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.465002Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:56250ed4d332436b3dc569d0132a853fdee98ec765acd0418ca32fd63a2c16b5","observation_id":"b074da60-bfa4-4d8e-9662-7c15f588d927","resolution":{"observed_at":"2026-08-07T10:28:08.465002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.647907Z","title":"Activitynet: A large-scale video benchmark for human activity understanding","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.647907Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:96d4b564c3e36ba5945eb79117fa2384ad1db66b382df592747728c4cb7ef807","observation_id":"b8625921-7c21-4aa7-9a75-856e626b190f","resolution":{"observed_at":"2026-08-07T10:28:08.647907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.783473Z","title":"Vip-llava: Making large multimodal models understand arbitrary visual prompts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.783473Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:d7826b0d8677f58fdb3d3e3fcb850fbf22c9b873b06693855a19f0500d994bfe","observation_id":"6d1cb01a-b711-45b9-835b-124c53b2aa96","resolution":{"observed_at":"2026-08-07T10:28:08.783473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:08.952935Z","title":"Active contours without edges.IEEE Transactions on image processing, 10(2):266–277, 2001","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:08.952935Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:a9c1411d0bbc25cd503481712520222a54982ef64ba068db86891c7d2a109eec","observation_id":"9dac110d-8043-47ee-ae47-2bbdb3a948b2","resolution":{"observed_at":"2026-08-07T10:28:08.952935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:09.089611Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.089611Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:0830c3fa4f630a832c10a00055841c689dd37bccaaad3654ec8cf12a2612ab69","observation_id":"8a0a7662-5adb-4947-9e25-a5fee112ded4","resolution":{"observed_at":"2026-08-07T10:28:09.089611Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:09.157074Z","title":"Videollm-online: Online video large language model for streaming video","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.157074Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:8b57b58e74f5f9ed3c9552723c9387332f689251a912cb2561657a78da3de4a0","observation_id":"b362ae58-7c9a-44a4-9249-8f628c09b56a","resolution":{"observed_at":"2026-08-07T10:28:09.157074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-07-06T15:47:07.545213Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-07T10:28:09.209098Z","title":"Shikra: Unleashing multimodal llm’s referential dialogue magic.arXiv preprint arXiv:2306.15195, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.209098Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:13701abf9be769f83e7d066d46204d1ed168997dad082d3ba40de108194c63bb","observation_id":"3d57f293-dc7e-400f-8b59-ed0e79f1b019","resolution":{"observed_at":"2026-08-07T10:28:09.209098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06558","last_updated":"2023-05-11T04:33:08Z","snapshot_observed_at":"2026-08-03T19:49:16.800693Z","submitted_at":"2023-05-11T04:33:08Z","title":"Segment and Track Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06558","snapshot_observed_at":"2026-08-07T10:28:09.287312Z","title":"Segment and track anything.arXiv preprint arXiv:2305.06558, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.287312Z"},"links":{"cited_paper":"/paper/2305.06558","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:1ac1d9455353192b46f4d0087f5162d05d1e152592f0a1b8952cb323429d5baa","observation_id":"917015bf-08bc-43be-87e3-d8b9bc2d7aa5","resolution":{"observed_at":"2026-08-07T10:28:09.287312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:22.068971Z","title":"Total-text: A comprehensive dataset for scene text detection and recognition, 2017","venue":null,"work_id":"2e4b2f07-784c-4359-aeec-e3bb895cf44f","year":2017},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.366044Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:1a51110beed771d47b4c9a1739541bf4eacbbbbfb4cf067d29a4039420cfd2f1","observation_id":"000a4c48-a6f7-46bc-8812-11e99f4309c4","resolution":{"observed_at":"2026-08-07T10:28:22.167513Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:21.866232Z","title":"V ocabulary-free image classification, 2024","venue":null,"work_id":"36727d25-b59d-477b-b169-4d20aa55642f","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.461655Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:d58ff854e0f2c0a5ec4ca885d7dd03bd7a960be25552c8e0c0b0c779520cefa5","observation_id":"9e559b13-e1bc-4755-b972-952d73d0b449","resolution":{"observed_at":"2026-08-07T10:28:21.982352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:21.299284Z","title":"Online action detection","venue":null,"work_id":"c3094ac9-ef89-4a97-a1f8-95a59bbee56e","year":2016},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.531677Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:e9b1702239502c7a2f0db469310df0fa2f437a9a726d06e4d5d5cdc1df81afb5","observation_id":"6bccd78c-1e66-4279-9cda-1f1ba90c107c","resolution":{"observed_at":"2026-08-07T10:28:21.636264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:21.069853Z","title":"Mevis: A large-scale benchmark for video segmentation with motion expressions, 2023","venue":null,"work_id":"0ecd032d-3ad7-4044-af6e-7754b58a032d","year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.622757Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:feefd49a23a722f1284d2b619fd85b09ece3910c94acc37b2d3e47821f8e9005","observation_id":"90bd38f4-2b46-4711-8c71-cd433798ce8c","resolution":{"observed_at":"2026-08-07T10:28:21.160819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.957306Z","title":"Actor and action video segmentation from a sentence","venue":null,"work_id":"495a8a83-683d-4dfb-ae5f-c51f637ccdb6","year":2018},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.710833Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:6f32f758b7e7c0cc131ecb2b1000be13478c150d48655236edc00b77bf45a1e6","observation_id":"ec455c19-e304-4fd8-a4fd-c922bb6e03b0","resolution":{"observed_at":"2026-08-07T10:28:21.019625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.772810Z","title":"Icdar2017 robust reading challenge on coco-text","venue":null,"work_id":"2731c5b4-322f-463c-89b4-65c793652234","year":2017},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.778550Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:ac226c08af29c02d79b7aa798268f1eed5b8773d2ee5a2e170c3fc8d7c141947","observation_id":"23fc4dd0-9b5d-4320-828c-a68112c74745","resolution":{"observed_at":"2026-08-07T10:28:20.857060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:09.831040Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.831040Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:32bd3c42593cd34932ec06e091ab5ba2814a996dc5be6f222f79662e21419c7a","observation_id":"93ee6ab1-f238-48e7-bb87-02d288870d6f","resolution":{"observed_at":"2026-08-07T10:28:09.831040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.609377Z","title":"Regiongpt: Towards region understanding vision language model, 2024","venue":null,"work_id":"d8dceb18-3f0a-4506-b6a2-5b7d3201a8e1","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:09.953283Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:aba7b42c7867228c75d2523f0f6dce41ce0ff90e417c0efaae7470bd82b9a223","observation_id":"0fccdb72-31f9-4b8f-8922-eae20a984627","resolution":{"observed_at":"2026-08-07T10:28:20.698363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05643","last_updated":"2025-03-03T10:28:30Z","snapshot_observed_at":"2026-08-07T05:23:25.250251Z","submitted_at":"2024-10-08T02:46:30Z","title":"TRACE: Temporal Grounding Video LLM via Causal Event Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05643","snapshot_observed_at":"2026-08-07T10:28:10.068680Z","title":"Trace: Temporal grounding video llm via causal event modeling.arXiv preprint arXiv:2410.05643, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.068680Z"},"links":{"cited_paper":"/paper/2410.05643","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:77813e233040d8779c2e7511fdf03bfec5d5eb214ff1fb04915d48f317072573","observation_id":"8e817e83-8877-43fe-bf03-842cf459c6cb","resolution":{"observed_at":"2026-08-07T10:28:10.068680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:10.160396Z","title":"Lvis: A dataset for large vocabulary instance segmentation, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.160396Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:e4ff72f97b30969fb03e9cf5a1b238321351fb82d9a5aa395aa201e5fc4f1165","observation_id":"b1a95b82-b009-4ffe-87f4-a71ce956a7b2","resolution":{"observed_at":"2026-08-07T10:28:10.160396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.456966Z","title":"Synthetic data for text localisation in natural images, 2016","venue":null,"work_id":"ec608a97-f88f-40b4-99f0-05762fdb35fb","year":2016},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.245238Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:9f93274f4bddf6277dc4159647b2ce8fc90a05ff50a84a14a70e506ccd994558","observation_id":"bf17f7ba-2721-403b-921c-fcb3e4933269","resolution":{"observed_at":"2026-08-07T10:28:20.526290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.269637Z","title":"Omni-rgpt: Unifying image and video region-level understanding via token marks, 2025","venue":null,"work_id":"a86bd877-3e47-45c6-8c99-009f463efcd8","year":2025},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.341914Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:52c41b1a7686b9ac317cdf9200316623ecc700b47b867c137be24b4eafe46fd2","observation_id":"9f6b5a53-9742-454a-9744-3d2693faf534","resolution":{"observed_at":"2026-08-07T10:28:20.368507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:20.079593Z","title":"Segment and caption anything","venue":null,"work_id":"7fdc2332-e986-4f08-8e9e-43e5d52d1754","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.442313Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:6b6bb035560bc55be5695d3aec2e4bce749cd39bc1b30536b100fc9b11721b26","observation_id":"329c1ee0-6d7c-4793-b219-11a137c499d1","resolution":{"observed_at":"2026-08-07T10:28:20.174249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T10:28:10.544501Z","title":"Gpt-4o system card.arXiv preprint arXiv:2410.21276, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.544501Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:3b97720549c667a095e1857f0bdd58e285fc2e6dc737223bfe4e431366d4710e","observation_id":"8f5ee2b8-10b5-4d15-9577-c1b75af42a49","resolution":{"observed_at":"2026-08-07T10:28:10.544501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:10.660851Z","title":"Visual prompt tuning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.660851Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:26b5bfb23816c677f865d10fc9523e6bc10299483483a9e4c6e0ee455133c7cd","observation_id":"da1fde41-df61-47a4-b776-fe83df5a39a0","resolution":{"observed_at":"2026-08-07T10:28:10.660851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.18363","last_updated":"2025-03-11T14:19:42Z","snapshot_observed_at":"2026-08-02T05:48:34.429861Z","submitted_at":"2024-11-27T14:11:10Z","title":"ChatRex: Taming Multimodal LLM for Joint Perception and Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.18363","snapshot_observed_at":"2026-08-07T10:28:10.735473Z","title":"Chatrex: Taming multimodal llm for joint perception and understanding.arXiv preprint arXiv:2411.18363, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.735473Z"},"links":{"cited_paper":"/paper/2411.18363","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:2c52775310d82554f10346e94d820b6ff80624681efdb8efe0c7e47176a1e34f","observation_id":"7e4fafc4-c40b-4f2c-abba-fae1b630d798","resolution":{"observed_at":"2026-08-07T10:28:10.735473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.915868Z","title":"Icdar 2015 competition on robust reading","venue":null,"work_id":"c90a6807-829d-4411-842e-bbb58b1bdcbf","year":2015},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.822007Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:2a6554c63178d1767c9e68a0f9c3191416cf2c0c1405092201a7581e02f75391","observation_id":"0320ae6a-736d-466a-9137-320cf8b0beb0","resolution":{"observed_at":"2026-08-07T10:28:19.977543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.764195Z","title":"Icdar 2013 robust reading competition","venue":null,"work_id":"2dcae961-b5e0-4a8b-8221-d9e14d44f3bf","year":2013},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.902744Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:5be35558e5d8a46eb2d7e07faf55c64b73cdffb8ad5cd6948d6507c0dc366ff9","observation_id":"79fca956-3d36-4d13-8c15-fc1fce4ea5a4","resolution":{"observed_at":"2026-08-07T10:28:19.819618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:10.990032Z","title":"Referitgame: Referring to objects in photographs of natural scenes","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:10.990032Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:b2a2ada93da93273eae8aefe566b5f925aeed2d2dd463c2e75076e50ad7e27d7","observation_id":"75fff697-834b-493d-9c0e-ebd32d1da19c","resolution":{"observed_at":"2026-08-07T10:28:10.990032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:11.093403Z","title":"Segment anything in high quality.Advances in Neural Information Processing Systems, 36:29914–29934, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.093403Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:7e559893a53dc85b2ad2bb4081944005bdcd40cf24ff461d9a97503ef06433e7","observation_id":"7dd22ac1-8d1d-418a-872d-1c1d175764ca","resolution":{"observed_at":"2026-08-07T10:28:11.093403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.02643","last_updated":"2023-04-05T17:59:46Z","snapshot_observed_at":"2026-07-06T15:12:37.953383Z","submitted_at":"2023-04-05T17:59:46Z","title":"Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.02643","snapshot_observed_at":"2026-08-07T10:28:11.178747Z","title":"Segment anything.arXiv preprint arXiv:2304.02643, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.178747Z"},"links":{"cited_paper":"/paper/2304.02643","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:f154d0ec156d2e4810c77fb1ff99a1fea3897723325815f2c39816cfe2ca4db3","observation_id":"9594d68a-989f-41aa-835f-5303e38fd1d0","resolution":{"observed_at":"2026-08-07T10:28:11.178747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.655205Z","title":"Openimages: A public dataset for large-scale multi-label and multi-class image classification.Dataset available from https://github","venue":null,"work_id":"cef24270-7ec2-44ae-ae98-f91c9c02d4fe","year":2017},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.254087Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:14ec2e536397ca13fa4bc84b86db46c8b4f18f1b0dd42e6f76694cfce5303742","observation_id":"02a13c13-2da8-45f8-bc92-16ded190fae2","resolution":{"observed_at":"2026-08-07T10:28:19.690256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.491412Z","title":"Shamma, Michael S","venue":null,"work_id":"1b7d255e-6025-4d3a-90d2-1a93a1ade2fe","year":2016},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.338922Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:8cd9a1ec419e22da79c9249eaa2a58b00e479f3639b81556a58e0012b7ce7503","observation_id":"dbc8b219-ba86-4eeb-a622-83fad787060b","resolution":{"observed_at":"2026-08-07T10:28:19.558320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.325530Z","title":"Beyond mot: Semantic multi-object tracking","venue":null,"work_id":"78027468-0052-4c2d-a810-1424276215c0","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.411250Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:011c974cb379986a61ff1d70724a91e34541c115ee225ad413f22de82e4ef9ed","observation_id":"2343811e-b4e8-4161-99c3-e1eed22d929b","resolution":{"observed_at":"2026-08-07T10:28:19.413523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16072","last_updated":"2025-04-22T17:51:41Z","snapshot_observed_at":"2026-08-07T16:00:09.944765Z","submitted_at":"2025-04-22T17:51:41Z","title":"Describe Anything: Detailed Localized Image and Video Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16072","snapshot_observed_at":"2026-08-07T10:28:11.485753Z","title":"Describe anything: Detailed localized image and video captioning.arXiv preprint arXiv:2504.16072, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.485753Z"},"links":{"cited_paper":"/paper/2504.16072","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:e9b938de1c83c8b561bc4679e23e4f6b8e48531cf09acd52c74ba58dcfa0a69b","observation_id":"3de15e83-db5f-4f6d-8e47-8de78f50c331","resolution":{"observed_at":"2026-08-07T10:28:11.485753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:11.573027Z","title":"Rouge: A package for automatic evaluation of summaries","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.573027Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:3b3677dace00322e88137098ed55c2c200053be1ca741c21cbdc159bd42c0d6e","observation_id":"a5acc589-816e-4212-9a41-05c03a5b85cc","resolution":{"observed_at":"2026-08-07T10:28:11.573027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:11.641278Z","title":"Lawrence Zitnick, and Piotr Dollár","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.641278Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:a952f867d62e2a0c5d81e7ae6fb40790648ad6df8eb2acdd0847926849d95225","observation_id":"77c18e82-2491-4fb3-a229-e0f3415129a5","resolution":{"observed_at":"2026-08-07T10:28:11.641278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-07-06T17:53:06.519891Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-08-07T10:28:11.715662Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.715662Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:8699ee0cb0d2d54579477fceb56da6f4d49ee4468948103d6b9a1a8644377bf3","observation_id":"734693f4-cdbe-494d-a851-2c3d8c130a24","resolution":{"observed_at":"2026-08-07T10:28:11.715662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-07T10:28:11.759586Z","title":"Deepseek-v3 technical report.arXiv preprint arXiv:2412.19437, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.759586Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:b72d44b0e52495169c557943e47360647fa1c8c0c09fbf259c0be3597e66c23f","observation_id":"af9be36d-cf66-4a61-ae61-4bb17b033d4f","resolution":{"observed_at":"2026-08-07T10:28:11.759586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:11.842353Z","title":"Gres: Generalized referring expression segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.842353Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:0f0d38f8ebfbb9c4a2a4c96015f4a9793e7088a951568b8a521155169e697868","observation_id":"9b80ee9d-3b88-48a8-aa56-1bae4df04a53","resolution":{"observed_at":"2026-08-07T10:28:11.842353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05935","last_updated":"2025-03-21T10:19:01Z","snapshot_observed_at":"2026-08-06T11:32:00.537531Z","submitted_at":"2024-02-08T18:59:48Z","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05935","snapshot_observed_at":"2026-08-07T10:28:11.917393Z","title":"Sphinx-x: Scaling data and parameters for a family of multi-modal large language models.arXiv preprint arXiv:2402.05935, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.917393Z"},"links":{"cited_paper":"/paper/2402.05935","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:43f126992cfdc3ac1d86d890f701ec0460ff00c4377c29d8ef8dfe3713f25b55","observation_id":"9626591a-cf6b-4f0e-85b6-962987fec048","resolution":{"observed_at":"2026-08-07T10:28:11.917393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:12.028467Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.028467Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:d2ff7e1aa6272acf2b3cce889bfc6b5110f1093feba067cbfeca49cc2b4bbf3e","observation_id":"d60893a0-c1ee-47ba-8b40-24254752769b","resolution":{"observed_at":"2026-08-07T10:28:12.028467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14824","snapshot_observed_at":"2026-08-07T10:28:12.090612Z","title":"Kosmos-2: Grounding multimodal large language models to the world.arXiv preprint arXiv:2306.14824, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.090612Z"},"links":{"cited_paper":"/paper/2306.14824","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:ba9af2504ff63471d55efd88a0142620097e085b168eae1b3799c304fb2de056","observation_id":"2f5a9006-e068-4a88-9059-5b2f4069a2bb","resolution":{"observed_at":"2026-08-07T10:28:12.090612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:19.133837Z","title":"The 2017 davis challenge on video object segmentation, 2018","venue":null,"work_id":"d603845d-f0db-4bf6-a0da-45985af796fd","year":2017},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.161483Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:0d563b3fb42e24ee97e952b648b70904b8de66ef55d6393aefd54c9d2a4b5f7d","observation_id":"40857708-b9b1-43ac-8f63-68cc2bbdbcff","resolution":{"observed_at":"2026-08-07T10:28:19.218820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.936646Z","title":"Artemis: Towards referential understanding in complex videos","venue":null,"work_id":"a400922d-ae93-469e-90df-99ef04627912","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.252891Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:47bddd5d1c24398cec4d9be04f0fa2690f64aa8432ad8129c758a4c26cce45b3","observation_id":"dca6c900-6108-4828-b8be-4fda60a8bde5","resolution":{"observed_at":"2026-08-07T10:28:19.045839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.765741Z","title":"Artemis: Towards referential understanding in complex videos, 2024","venue":null,"work_id":"b2a008eb-7923-4ad4-b294-ecc9bd06fea9","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.318885Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:2e361c9719f0669d43e969a5fb96533d47784d5da4ea0d492bd11e104bc1851a","observation_id":"0411d034-bbf6-4397-8a7b-2527829345fb","resolution":{"observed_at":"2026-08-07T10:28:18.841719Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.606462Z","title":"Paco: Parts and attributes of common objects, 2023","venue":null,"work_id":"5cd31bd2-06df-477a-a395-a5070ad0bba9","year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.398651Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:c4cc388db0d813ae86d08518c0070777f0a4b01267735820f8e1d23f94e421b3","observation_id":"615de475-4912-4b52-843a-076e6b94c72f","resolution":{"observed_at":"2026-08-07T10:28:18.669927Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:12.438893Z","title":"Anwer, Erix Xing, Ming-Hsuan Yang, and Fahad S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.438893Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:34968b30944830f40950f744b90a0f356860da222a32cef9bffaeae2aa21af8b","observation_id":"3b52654c-8e87-4765-9961-b934d46bc038","resolution":{"observed_at":"2026-08-07T10:28:12.438893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-07T10:28:12.549714Z","title":"Sam 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.549714Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:ef71cbe09e5faea3779a263bb71842b8369fe534fea5337cbdec4fcbaa58d626","observation_id":"80503a7d-ce6a-4a73-abf5-43bb8717d469","resolution":{"observed_at":"2026-08-07T10:28:12.549714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:12.604590Z","title":"Sam 2: Segment anything in images and videos, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.604590Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:72aeea4aa14eedf61a7b630a8f2d45cffacf078117e553c7207353d1a295854f","observation_id":"77d3f82b-7679-4289-a451-10f39f150192","resolution":{"observed_at":"2026-08-07T10:28:12.604590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14159","last_updated":"2024-01-25T13:12:09Z","snapshot_observed_at":"2026-07-06T17:20:25.138890Z","submitted_at":"2024-01-25T13:12:09Z","title":"Grounded SAM: Assembling Open-World Models for Diverse Visual Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14159","snapshot_observed_at":"2026-08-07T10:28:12.695638Z","title":"Grounded sam: Assembling open-world models for diverse visual tasks.arXiv preprint arXiv:2401.14159, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.695638Z"},"links":{"cited_paper":"/paper/2401.14159","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:5c990704a2ef4950ec875ce8f5bf525c3d45c1f6e89e9fb02147a734067ecb4b","observation_id":"dc7a87ef-9de9-4a0f-a155-21f4c3590af6","resolution":{"observed_at":"2026-08-07T10:28:12.695638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:12.752865Z","title":"Objects365: A large-scale, high-quality dataset for object detection","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.752865Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:3dc9d9c1ee2e7f05824f0c554a353e6917d2b16b6bc8f6a83b8a6386f185bcd6","observation_id":"53fdc3f9-85b0-41cd-9010-687b9512139e","resolution":{"observed_at":"2026-08-07T10:28:12.752865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.433001Z","title":"Textocr: Towards large-scale end-to-end reasoning for arbitrary-shaped scene text, 2021","venue":null,"work_id":"1d599850-450a-4a0d-be75-bc955840ffcb","year":2021},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.830981Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:9b6694f6924208d25d87e2b43050a8a6954e501fe861e90dc692744891810ccc","observation_id":"bd84cbc5-8476-4f71-9d32-a5ca9a7245e6","resolution":{"observed_at":"2026-08-07T10:28:18.501128Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.286911Z","title":"Icdar 2019 competition on large-scale street view text with partial labeling – rrc-lsvt, 2019","venue":null,"work_id":"74117015-d431-445a-b4f5-a171cd857834","year":2019},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.899875Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:66c8bd86d4422bf642c8f3105e743c40a3eaeaf2e523e473cf50857b9d5c960f","observation_id":"003298e2-5e3e-4286-a1da-025dd32bb251","resolution":{"observed_at":"2026-08-07T10:28:18.357067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:18.116563Z","title":"Human-centric spatio-temporal video grounding with visual transformers, 2021","venue":null,"work_id":"ca513442-32e7-4c40-af91-f7bb06e4638f","year":2021},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:12.974590Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:4fad73acc511c52a3ba0cd7e2b8ab3729d215770529ee13bdfa3264208b177af","observation_id":"97ae4dcd-4b1e-4f1b-90fd-60373548ef85","resolution":{"observed_at":"2026-08-07T10:28:18.211229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.951510Z","title":"Human- centric spatio-temporal video grounding with visual transformers.IEEE Transactions on Circuits and Systems for Video Technology, 32(12):8238–8249, 2021","venue":null,"work_id":"d65ec784-2fdf-4b46-bf8b-301e2a2fdf08","year":2021},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.075724Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:a521fc076a45d3dfb7c4f704093d9622459941930344b4e328ceb1483ced0d7a","observation_id":"f81eebbf-4a2b-4617-958f-4e2a0ea98be1","resolution":{"observed_at":"2026-08-07T10:28:18.042690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:13.106006Z","title":"Cider: Consensus-based image description evaluation","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.106006Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:16ce6e5024c427c1f21274d06fb31c3b44d1f5d75a946d85d90e86e7a26c1f8d","observation_id":"a6c7bb8a-736d-4403-b5a3-847e4dbdc50a","resolution":{"observed_at":"2026-08-07T10:28:13.106006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.797415Z","title":"Coco-text: Dataset and benchmark for text detection and recognition in natural images, 2016","venue":null,"work_id":"2a8f37c0-03de-4cf4-a928-81cb1d7a7a15","year":2016},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.197387Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:ad7a074cb7adb350e4645aaaa79db43f485a5d71033c2c920aa6b5b2bef9b63a","observation_id":"6538d8ce-016e-4755-9409-05c67de2a623","resolution":{"observed_at":"2026-08-07T10:28:17.871637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.624767Z","title":"Elysium: Exploring object-level perception in videos via mllm, 2024","venue":null,"work_id":"4b7741c9-b5a3-40b8-97d0-01e2b1fa78bc","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.272446Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:16970aa24a16936450e6048a8f1b4ff84313078aaeac91b09a39a0a9cc073e77","observation_id":"2be0f5f2-0406-4073-b936-c8820e72116c","resolution":{"observed_at":"2026-08-07T10:28:17.712210Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.462594Z","title":"Elysium: Exploring object-level perception in videos via mllm","venue":null,"work_id":"1bc77669-60ab-4e66-b07b-c6764de5ecf0","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.376288Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:11b3776439a72167161235a986a52470140c6809f5f1106e7d10d919a92782d7","observation_id":"86242f7d-353d-4fce-9401-a217cb7e31d1","resolution":{"observed_at":"2026-08-07T10:28:17.544629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.316954Z","title":"Towards open-vocabulary video instance segmentation, 2023","venue":null,"work_id":"dae86712-c075-4f92-b220-cb7eb5c853a8","year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.408456Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:912ea97febe3e9f4a94ffd0b7a2fee3b2750751a86851df089cf9e9912263b71","observation_id":"d27a146f-4d23-4b7a-8f8a-50063baddd19","resolution":{"observed_at":"2026-08-07T10:28:17.380342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:13.430957Z","title":"Git: A generative image-to-text transformer for vision and language, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.430957Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:79778f296f49d7a15df2ec15beafdf7fbc4fcedd7ef9beff855492c3d67ceddd","observation_id":"24ea961e-be69-4b03-8336-d84a91f841d6","resolution":{"observed_at":"2026-08-07T10:28:13.430957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:17.170260Z","title":"V3det: Vast vocabulary visual detection dataset, 2023","venue":null,"work_id":"09b59949-2c9b-4b86-b272-062425adf1e3","year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.449740Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:8dfc08b277a97d2e3db4303df51a9652407535c690b444a29379584498826f1b","observation_id":"b4f51c45-574e-48be-87e2-b80093e76b7c","resolution":{"observed_at":"2026-08-07T10:28:17.243065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.02677","last_updated":"2023-07-06T13:47:21Z","snapshot_observed_at":"2026-07-06T15:23:10.114682Z","submitted_at":"2023-05-04T09:48:22Z","title":"Caption Anything: Interactive Image Description with Diverse Multimodal Controls","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.02677","snapshot_observed_at":"2026-08-07T10:28:13.474745Z","title":"Caption anything: Interactive image description with diverse multimodal controls.arXiv preprint arXiv:2305.02677, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.474745Z"},"links":{"cited_paper":"/paper/2305.02677","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:3a93c9be1baf1cde859e5dd936c7c9d9c8204003624fd1ea8db71e2d3461cae0","observation_id":"64243a39-bee3-4aca-b170-7f18f7ea9eea","resolution":{"observed_at":"2026-08-07T10:28:13.474745Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01907","last_updated":"2023-08-03T17:59:47Z","snapshot_observed_at":"2026-07-06T16:02:15.266899Z","submitted_at":"2023-08-03T17:59:47Z","title":"The All-Seeing Project: Towards Panoptic Visual Recognition and Understanding of the Open World","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.01907","snapshot_observed_at":"2026-08-07T10:28:13.496607Z","title":"The all-seeing project: Towards panoptic visual recognition and understanding of the open world.arXiv preprint arXiv:2308.01907, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.496607Z"},"links":{"cited_paper":"/paper/2308.01907","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:821aba0a1f97e379cbfb8c2ec37137a298f33799e8cb31e3ac13ce8fb2004901","observation_id":"7b4824b5-1247-46f0-9acb-6b6f537991e6","resolution":{"observed_at":"2026-08-07T10:28:13.496607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.973287Z","title":"Grit: A generative region-to-text transformer for object understanding","venue":null,"work_id":"d2c5e61b-a93f-465a-a2c1-88f264021de9","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.564740Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:714f0acd9547d1adb16c7b4f4f6920cfa350df8519353bcad913371e246f9d67","observation_id":"ce98c32a-3bee-4a17-b336-b03bff860f17","resolution":{"observed_at":"2026-08-07T10:28:17.065248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:13.586867Z","title":"Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks.Advances in Neural Information Processing Systems, 37:69925–69975, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.586867Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:68651879ed1c222ac51991053252fd59cffd853be8c19743a57aac8a4d461b73","observation_id":"feb50aa7-36ce-4526-9e51-e569d7c557cb","resolution":{"observed_at":"2026-08-07T10:28:13.586867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.789844Z","title":"Youtube-vos: A large-scale video object segmentation benchmark, 2018","venue":null,"work_id":"edfab5fb-0b7b-429c-bace-0f8638b8b022","year":2018},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.605063Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:919a12d78ef0225a32f45718fc32b7108b95ba08040c3fc751de0e4dc60790e7","observation_id":"4e4340d2-d0db-4de9-ba17-cf974d152d9f","resolution":{"observed_at":"2026-08-07T10:28:16.863847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T10:28:13.643014Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.643014Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:fe093bb3222bef8b679eae49232d701be5239cbe224afcc1d8cf250dd3a12a95","observation_id":"6306acd4-37f1-40c5-8523-43cfcc258b31","resolution":{"observed_at":"2026-08-07T10:28:13.643014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.509176Z","title":"Vid2seq: Large-scale pretraining of a visual language model for dense video captioning, 2023","venue":null,"work_id":"f4162725-7b4c-4fdf-a5bf-622e5a972704","year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.676752Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:ce2c53cafcdf980d11d5682c1388832f13c557ee793c0c8a1a0328fa0b477ef1","observation_id":"bba35fb9-1d5f-43a0-9e47-1b2befb8714e","resolution":{"observed_at":"2026-08-07T10:28:16.586598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.352292Z","title":"Samurai: Adapting segment anything model for zero-shot visual tracking with motion-aware memory, 2024","venue":null,"work_id":"6b3b498b-a18c-44a0-a1b0-bbd98a4627fa","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.714771Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:03c9af5bcb547b6dd671f1c65bf50375c9bdb823b5951ca05f656ddbe991d64c","observation_id":"038aa238-4d5e-452e-897b-8f9fd415f96a","resolution":{"observed_at":"2026-08-07T10:28:16.410525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:13.738804Z","title":"Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.738804Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:4e0af1a3305d45e4dfc21a3db008dc75428e726240bc0ad74e48834597f857e1","observation_id":"6f24cf1c-e96e-498b-94ec-f38cf6b5d0a6","resolution":{"observed_at":"2026-08-07T10:28:13.738804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.203152Z","title":"Detecting texts of arbitrary orientations in natural images","venue":null,"work_id":"ac7158e4-9d3c-43f1-aad3-c04b04fc377a","year":2012},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.788564Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:0abf7c7c9a66b160e26b3bf28e496eebae6c03f29b00a2817890f2f04a721aa2","observation_id":"7b581093-f66b-4a18-8762-233b527d2abf","resolution":{"observed_at":"2026-08-07T10:28:16.269910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07704","last_updated":"2023-10-11T17:55:15Z","snapshot_observed_at":"2026-07-06T16:31:25.350087Z","submitted_at":"2023-10-11T17:55:15Z","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07704","snapshot_observed_at":"2026-08-07T10:28:13.842739Z","title":"Ferret: Refer and ground anything anywhere at any granularity.arXiv preprint arXiv:2310.07704, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.842739Z"},"links":{"cited_paper":"/paper/2310.07704","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:0e7245317b468bc051e9ed55d020021fcb2bf0f00118125d03541cf0137b7ff6","observation_id":"15915057-4c0d-42e2-9ead-e8bb545702d5","resolution":{"observed_at":"2026-08-07T10:28:13.842739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:13.904324Z","title":"Merlin: Empowering multimodal llms with foresight minds","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.904324Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:a4ed9fdffef4398e83537982548f6e4e98268163663c84143a9b6a4bd885bc12","observation_id":"54f1de86-fd8f-4e15-ac4e-e612520875a4","resolution":{"observed_at":"2026-08-07T10:28:13.904324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:16.007221Z","title":"Sa2va: Marrying sam2 with llava for dense grounded understanding of images and videos, 2025","venue":null,"work_id":"0ad46f26-a4fc-487d-8483-5195049792cb","year":2025},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:13.964588Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:ce22c98bb0766127fbdb022a2c441ad0a724ff705ae661090eede63b0e7225dc","observation_id":"04667787-33fe-4d7a-b681-83be6aea276a","resolution":{"observed_at":"2026-08-07T10:28:16.098427Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:14.049510Z","title":"Osprey: Pixel understanding with visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.049510Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:4c3acb7062bba5ea040fde5ef362d635db78aeef87496faf808c274dd4633279","observation_id":"3532d1b8-250b-4955-bf83-c62da59b9e9e","resolution":{"observed_at":"2026-08-07T10:28:14.049510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00599","last_updated":"2025-03-25T08:10:15Z","snapshot_observed_at":"2026-08-02T21:54:09.307199Z","submitted_at":"2024-12-31T18:56:46Z","title":"VideoRefer Suite: Advancing Spatial-Temporal Object Understanding with Video LLM","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00599","snapshot_observed_at":"2026-08-07T10:28:14.117492Z","title":"Videorefer suite: Advancing spatial-temporal object understanding with video llm.arXiv preprint arXiv:2501.00599, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.117492Z"},"links":{"cited_paper":"/paper/2501.00599","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:09b625ea2b7edaa5b29cecce095192d1bcd7f610bd39fa1733fd105c78e53813","observation_id":"60a624db-a6fa-43c6-aedc-ac269ea708c2","resolution":{"observed_at":"2026-08-07T10:28:14.117492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14289","last_updated":"2023-07-01T07:26:22Z","snapshot_observed_at":"2026-07-06T15:46:29.060519Z","submitted_at":"2023-06-25T16:37:25Z","title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14289","snapshot_observed_at":"2026-08-07T10:28:14.194914Z","title":"Faster segment anything: Towards lightweight sam for mobile applications.arXiv preprint arXiv:2306.14289, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.194914Z"},"links":{"cited_paper":"/paper/2306.14289","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:c83f26b72a7338d9edacbd40f8959a5a013016f4b17752e7232a008dfc5a4918","observation_id":"86547f57-af2f-49a4-aa87-4cd8e57a6fa5","resolution":{"observed_at":"2026-08-07T10:28:14.194914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07973","last_updated":"2024-04-11T17:56:05Z","snapshot_observed_at":"2026-08-07T09:16:48.366534Z","submitted_at":"2024-04-11T17:56:05Z","title":"Ferret-v2: An Improved Baseline for Referring and Grounding with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07973","snapshot_observed_at":"2026-08-07T10:28:14.279027Z","title":"Ferret-v2: An improved baseline for referring and grounding with large language models.arXiv preprint arXiv:2404.07973, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.279027Z"},"links":{"cited_paper":"/paper/2404.07973","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:b1483f25f8a9f2e4b4f08dcf47b9b2b55ea587c25f3f5a5ee21a4ef13de05b74","observation_id":"ae642888-c3de-4cbe-aa24-3161505b0301","resolution":{"observed_at":"2026-08-07T10:28:14.279027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:15.817591Z","title":"Gpt4roi: Instruction tuning large language model on region-of-interest, 2025","venue":null,"work_id":"d50fe2a2-c302-4c6a-8846-53f71aa8c351","year":2025},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.380692Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:60622be2cd527d6e7d03902000810b734505a87c8a196b35a933a765270c9cd0","observation_id":"e377b8a5-dc62-43a2-b95e-5499da6488b8","resolution":{"observed_at":"2026-08-07T10:28:15.913568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:15.675110Z","title":"Where does it exist: Spatio-temporal video grounding for multi-form sentences","venue":null,"work_id":"f0c31e77-381f-4875-8b03-6580131db20d","year":2020},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.500909Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:202c07772fd24b79fc90dd1da38532e3d7440e0e1298dde173265b99fb1f627b","observation_id":"614da7e3-d07b-487c-875c-d939bf00e4fa","resolution":{"observed_at":"2026-08-07T10:28:15.744187Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09474","last_updated":"2023-07-18T17:56:06Z","snapshot_observed_at":"2026-07-06T15:55:28.974661Z","submitted_at":"2023-07-18T17:56:06Z","title":"ChatSpot: Bootstrapping Multimodal LLMs via Precise Referring Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09474","snapshot_observed_at":"2026-08-07T10:28:14.565949Z","title":"Chatspot: Bootstrapping multimodal llms via precise referring instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.565949Z"},"links":{"cited_paper":"/paper/2307.09474","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:61f4f2cf79e07ff1a071e901232b47a4d45b0e32138fa35320b52d85b3e2bc30","observation_id":"ccdc906f-94a4-4d76-bf34-91859924d5ba","resolution":{"observed_at":"2026-08-07T10:28:14.565949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.12156","last_updated":"2023-06-21T10:08:29Z","snapshot_observed_at":"2026-07-06T15:45:01.961896Z","submitted_at":"2023-06-21T10:08:29Z","title":"Fast Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.12156","snapshot_observed_at":"2026-08-07T10:28:14.704495Z","title":"Fast segment anything.arXiv preprint arXiv:2306.12156, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.704495Z"},"links":{"cited_paper":"/paper/2306.12156","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:4dbe8552dd49e1ea23a4eac909bc2c8b3ae03c4b06766d0fad6c5013bb066413","observation_id":"037adc13-83fb-4e76-a27e-08af8718a73b","resolution":{"observed_at":"2026-08-07T10:28:14.704495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:15.491527Z","title":"Controlcap: Controllable region-level captioning","venue":null,"work_id":"6bc19424-1ffe-4ac9-806f-accdf1cc1fb9","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.776221Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:7d8a07c306a539401aa4ac5b8a8f4dd4a8ca01d613f0ba38200434b668ccb4c9","observation_id":"1b5a5567-5bb8-4335-a8ee-4977daff2dd2","resolution":{"observed_at":"2026-08-07T10:28:15.590718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:28:15.323383Z","title":"Streaming dense video captioning","venue":null,"work_id":"5bab4f21-a0ce-4e4b-8ca1-e4ee75dcc2ad","year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:14.907661Z"},"links":{"citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:4f6a7cd627587b8ed563b4db5ec7856335cf8e8c0d4dca5fbfd400a690c2aadc","observation_id":"103c9006-1579-4b31-b1f1-df970f5937e1","resolution":{"observed_at":"2026-08-07T10:28:15.393823Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos"},"reference_resolution":{"displayed":89,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":51,"verified_exact":0,"verified_fuzzy":38},"total_outbound_references":89},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 89 of 89 outbound references and 15 inbound Pith citation observations for arXiv:2506.05302."}