{"as_of":"2026-08-08T06:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6a9398dfc4592c0351954bad64be9b0bd2a97aa3149a72fc5f176eb7be659e13","coverage":[{"denominator":33,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:22:38.056203Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T04:26:51.521243Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T11:28:04.115693Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"cited_work":{"arxiv_id":"2506.21873","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.21873","snapshot_observed_at":"2026-07-03T11:28:04.115693Z","title":"Grounding-aware token pruning: Recovering from drastic performance drops in visual grounding caused by pruning, 2025","venue":null,"work_id":"2e1fc42d-eac9-4215-b5ea-a3c62cfc02f8","year":2025},"citing_paper":{"arxiv_id":"2606.12412","last_updated":"2026-06-10T17:59:57Z","snapshot_observed_at":"2026-08-03T05:45:43.148184Z","submitted_at":"2026-06-10T17:59:57Z","title":"Reroute, Don't Remove: Recoverable Visual Token Routing for Vision-Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T09:35:24.118536Z"},"links":{"cited_paper":"/paper/2506.21873","citing_paper":"/paper/2606.12412"},"observation_digest":"sha256:b9d4251c103839c4c86382f8220e585af678ff8b4768d25f72090e898c93145e","observation_id":"1d10623c-83fb-4f45-95cd-23b3146a6d3d","resolution":{"observed_at":"2026-07-03T11:28:04.117867Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"cited_work":{"arxiv_id":"2506.21873","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.21873","snapshot_observed_at":"2026-07-03T11:28:04.115693Z","title":"Grounding-aware token pruning: Recovering from drastic performance drops in visual grounding caused by pruning, 2025","venue":null,"work_id":"2e1fc42d-eac9-4215-b5ea-a3c62cfc02f8","year":2025},"citing_paper":{"arxiv_id":"2606.31599","last_updated":"2026-06-30T12:47:30Z","snapshot_observed_at":"2026-07-07T00:05:17.259491Z","submitted_at":"2026-06-30T12:47:30Z","title":"Token-Sparse Medical Multimodal Reasoning via Dual-Stream Reinforcement Learning","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-07-01T05:36:43.609602Z"},"links":{"cited_paper":"/paper/2506.21873","citing_paper":"/paper/2606.31599"},"observation_digest":"sha256:93f13a6f5ad19b60fb4092fea108c392b7bb75d80d0d71157c4086433d307982","observation_id":"dd37fb2c-b09e-4a78-b543-4a85336b2d2f","resolution":{"observed_at":"2026-07-01T10:15:45.167041Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.21873","snapshot_observed_at":"2026-08-04T01:29:54.348125Z","title":"arXiv preprint arXiv:2506.21873 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00077","last_updated":"2026-08-04T10:46:13Z","snapshot_observed_at":"2026-08-07T23:11:47.320726Z","submitted_at":"2026-07-29T13:45:31Z","title":"Beyond Accuracy: Auditing Spatial Provenance in Visual Token Pruning for OCR-Critical MLLM Inference","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-04T01:29:54.348125Z"},"links":{"cited_paper":"/paper/2506.21873","citing_paper":"/paper/2608.00077"},"observation_digest":"sha256:64a8826b8c9fdf428544bcc4a06d1b715ac508ca0cf3b9fb386ed6dbf5596f32","observation_id":"9580cabb-5320-4a7c-ba02-205e08d07265","resolution":{"observed_at":"2026-08-04T01:29:54.348125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.21873","snapshot_observed_at":"2026-08-05T04:26:51.521243Z","title":"arXiv preprint arXiv:2506.21873 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00077","last_updated":"2026-08-04T10:46:13Z","snapshot_observed_at":"2026-08-07T23:11:47.320726Z","submitted_at":"2026-07-29T13:45:31Z","title":"Beyond Accuracy: Auditing Spatial Provenance in Visual Token Pruning for OCR-Critical MLLM Inference","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-05T04:26:51.521243Z"},"links":{"cited_paper":"/paper/2506.21873","citing_paper":"/paper/2608.00077"},"observation_digest":"sha256:fa16e8361b667f4ec4b60ea6d7387caf68573651ea22d4705fe8f6411353ed7c","observation_id":"1208ac5a-4bec-4880-8f3e-70d05e81322a","resolution":{"observed_at":"2026-08-05T04:26:51.521243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.21873/citation-record","integrity":"/paper/2506.21873/integrity","json":"/paper/2506.21873/citation-record.json","paper":"/paper/2506.21873"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2103.00020","last_updated":"2021-02-26T19:04:58Z","snapshot_observed_at":"2026-07-06T10:45:03.059688Z","submitted_at":"2021-02-26T19:04:58Z","title":"Learning Transferable Visual Models From Natural Language Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.00020","snapshot_observed_at":"2026-08-06T22:22:34.222152Z","title":"Learning transferable visual models from natu- ral language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:34.222152Z"},"links":{"cited_paper":"/paper/2103.00020","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:e10f8eda8e16ed7b08b62e230cdaf4e197d7bfb7c55873125a5f4fd963c9c7e6","observation_id":"87ff0b8a-ca3b-4887-b17f-c9c0cec093d5","resolution":{"observed_at":"2026-08-06T22:22:34.222152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:40.719527Z","title":"Lmms-eval: Accelerating the development of large multimoal models, 2024","venue":null,"work_id":"99a46a9f-19b5-4402-93d7-9a8778506457","year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:34.319536Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:d29075630da764f878e11804093516cc83e0451411b4efb8c150d6ace7e54518","observation_id":"6a4d00e1-d3d4-4eff-8769-671d4229e0c0","resolution":{"observed_at":"2026-08-06T22:22:40.797436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:34.453915Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:34.453915Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:5bb7a962a98c54c8bd555d54ec1a03cc1bec9bb80aa0c1951d44c42502dc73b4","observation_id":"13662c94-8ae8-41ee-ae62-f8865c5086aa","resolution":{"observed_at":"2026-08-06T22:22:34.453915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:40.565158Z","title":"PuMer: Pruning and merging tokens for efficient vision language models","venue":null,"work_id":"43aae28a-9e85-4547-8cdd-15055b81257e","year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:34.502806Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:faf6f57d093a26fd92d7071a27a28ba27fff3b4e7f7ad2aecdf0014ae0c1e787","observation_id":"ed13873d-7303-41c2-887f-f70ad36352f5","resolution":{"observed_at":"2026-08-06T22:22:40.638759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-07-06T15:47:07.545213Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-06T22:22:34.600551Z","title":"Shikra: Unleashing multi- modal llm’s referential dialogue magic","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:34.600551Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:72ce0952f18dfc2c682f4273b83961e69f2017f668f7f0713860be81b2b99ecc","observation_id":"3f9feba8-b212-40bb-b00d-6e25488e19e9","resolution":{"observed_at":"2026-08-06T22:22:34.600551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:40.406044Z","title":"Chasing sparsity in vision transform- ers: An end-to-end exploration","venue":null,"work_id":"0cd49533-3beb-4872-abb5-283707a491fa","year":2021},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:36.588326Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:b5545ad45ff316a38d0e14cee7745f49b571601b463013799232a828da419d62","observation_id":"5af470b7-381b-4a2d-9d38-e97d8fc76450","resolution":{"observed_at":"2026-08-06T22:22:40.493296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01179","last_updated":"2024-12-19T06:26:04Z","snapshot_observed_at":"2026-07-06T19:09:14.628371Z","submitted_at":"2024-09-02T11:19:54Z","title":"Recoverable Compression: A Multimodal Vision Token Recovery Mechanism Guided by Text Information","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01179","snapshot_observed_at":"2026-08-06T22:22:36.892337Z","title":"Recoverable compression: A mul- timodal vision token recovery mechanism guided by text in- formation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:36.892337Z"},"links":{"cited_paper":"/paper/2409.01179","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:9dade1c11d0c93a5089bd5229cdd123128e520c5013488a95bfaed3083bf5936","observation_id":"7977b23f-3ab8-4e3b-a190-dbfb2a7e998a","resolution":{"observed_at":"2026-08-06T22:22:36.892337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:40.226085Z","title":"Bert: Pre-training of deep bidirectional trans- formers for language understanding, 2019","venue":null,"work_id":"9b33d571-8167-4275-9463-39d6415fb6e3","year":2019},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:36.947010Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:bb4ddda9fcde5af8769162c531c7e2c65304ed76d27050736028cc9710e00085","observation_id":"fc0b8107-ea83-4856-9830-afdc21830e16","resolution":{"observed_at":"2026-08-06T22:22:40.317527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-06T22:22:37.063835Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.063835Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:af3b3462fac72ebadc8dc06abd05b7ee65e9b579e56d4ce95a6e43afb5ffc6f6","observation_id":"9d0203be-a9ae-4267-b819-47f2116e220d","resolution":{"observed_at":"2026-08-06T22:22:37.063835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:40.051463Z","title":"Mme: A compre- hensive evaluation benchmark for multimodal large language models, 2024","venue":null,"work_id":"591fe45c-7102-495c-82a7-79ab4a5547af","year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.158952Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:1620b0f89aca9dbd60f2f6b96e38f92e7e8cec49962087a96742436416125a2f","observation_id":"80d3a7ef-ddd5-48d5-8a6d-db11f437122b","resolution":{"observed_at":"2026-08-06T22:22:40.112200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:39.845194Z","title":"Stangl, Anhong Guo, Chi Lin, Kristen Grauman, Jiebo Luo, and Jeffrey P","venue":null,"work_id":"7f038059-1292-4ccb-bf8b-35e1e00d4ae6","year":2018},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.240849Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:594bc591a1a981e41ad289542fa7576fb5aab85d9be8f3c2ba0d97d188da1d37","observation_id":"6252318a-6096-4ab7-8ba4-8f2660f663eb","resolution":{"observed_at":"2026-08-06T22:22:39.929819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07704","last_updated":"2023-10-11T17:55:15Z","snapshot_observed_at":"2026-07-06T16:31:25.350087Z","submitted_at":"2023-10-11T17:55:15Z","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07704","snapshot_observed_at":"2026-08-06T22:22:37.288321Z","title":"Ferret: Refer and ground anything anywhere at any granularity","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.288321Z"},"links":{"cited_paper":"/paper/2310.07704","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:48c66ebf32197b2d6822e04dbe3230766444ae14fb7f6d3239fc2c5cea3080b4","observation_id":"96308886-7034-4e99-b06c-ec9193ca46fc","resolution":{"observed_at":"2026-08-06T22:22:37.288321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:39.654091Z","title":"Hudson and Christopher D","venue":null,"work_id":"7b66f5e5-cf26-4379-9ede-22b3514985dc","year":2019},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.322374Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:cee6d287212f2bba1306976217e9d2f1634417ff5b4425e62e3556db2c3060f4","observation_id":"94758521-e0ad-4fef-9396-4cb57a0b7a2c","resolution":{"observed_at":"2026-08-06T22:22:39.759716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.09478","last_updated":"2023-11-07T18:25:48Z","snapshot_observed_at":"2026-08-06T12:38:03.720232Z","submitted_at":"2023-10-14T03:22:07Z","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.09478","snapshot_observed_at":"2026-08-06T22:22:37.354640Z","title":"Minigpt- v2: Large language model as a unified interface for vision- language multi-task learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.354640Z"},"links":{"cited_paper":"/paper/2310.09478","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:57a05b8aab178255913acfed861df5821c0838a484e4ab318b507080f2ad46cb","observation_id":"7ffded63-3567-47ec-910b-4f75f5548ef0","resolution":{"observed_at":"2026-08-06T22:22:37.354640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.12763","last_updated":"2021-10-12T00:49:54Z","snapshot_observed_at":"2026-08-03T17:40:24.925899Z","submitted_at":"2021-04-26T17:55:33Z","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.12763","snapshot_observed_at":"2026-08-06T22:22:37.398957Z","title":"Mdetr– modulated detection for end-to-end multi-modal understand- ing","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.398957Z"},"links":{"cited_paper":"/paper/2104.12763","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:b719af86aa6efdffc038d930c72f126d8a0a36343776689b10c1f6ef52790667","observation_id":"e5177977-f620-44b3-af16-da781b15a257","resolution":{"observed_at":"2026-08-06T22:22:37.398957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:39.492458Z","title":"ReferItGame: Referring to objects in pho- tographs of natural scenes","venue":null,"work_id":"83f6d142-9d5b-4e20-88d7-95a80ba9b511","year":2014},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.439860Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:b35d55ced8085bc86983d3665314f1c1efc508933587a5ced8611c423af8236d","observation_id":"1047d42b-4035-4686-8f2f-34b0c62f05e4","resolution":{"observed_at":"2026-08-06T22:22:39.581887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-06T22:22:37.486139Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.486139Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:9900db23f444733e8b15f2d06e5b2c84f9a90a2f745652c611186df8c411faa5","observation_id":"6aaf2107-1780-41a2-9582-3bfcd228cdca","resolution":{"observed_at":"2026-08-06T22:22:37.486139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:39.300918Z","title":"Improved baselines with visual instruction tuning, 2023","venue":null,"work_id":"7b0014f3-3bee-4c7b-92eb-399dff8e7a2a","year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.532926Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:0b3fa4820bacf832ca6a7eb2dde59d1fa6d753e9bced3e82aea30fc891a56b97","observation_id":"6b850b5a-68b8-422d-8942-b13c090b933f","resolution":{"observed_at":"2026-08-06T22:22:39.392107Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:39.124690Z","title":"Visual instruction tuning","venue":null,"work_id":"6a7aaff5-c77a-461b-a2cc-5d8239b5f914","year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.566606Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:905ddfafc6c4bfa58c53a422a6cffb107930b2ef8b62b411328e1535df6b00d1","observation_id":"68059e97-b956-4043-b2dd-cd8c386fdc96","resolution":{"observed_at":"2026-08-06T22:22:39.211131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:38.986948Z","title":"Llava-next: Im- proved reasoning, ocr, and world knowledge, 2024","venue":null,"work_id":"5e965e34-1315-4503-a7a6-b4899a310554","year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.595557Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:f6b598aafa30f269d32dc292ed0999a61006a801cee802278f00c0c855200fdd","observation_id":"a98f1f51-a427-4ee7-8008-049e208faa8f","resolution":{"observed_at":"2026-08-06T22:22:39.048593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:38.776008Z","title":"Ok-vqa: A visual question answer- ing benchmark requiring external knowledge","venue":null,"work_id":"d271bd48-0ca9-4e18-b137-00a6a6fc3489","year":2019},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.636569Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:5f3ca4af9c3f3be9c62696d5994cc145426f6bfb30286442dda3d4950ba84d44","observation_id":"56f3c6af-f16a-4ade-a470-9398eb3fad83","resolution":{"observed_at":"2026-08-06T22:22:38.868981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:37.656404Z","title":"Llava-prumerge: Adaptive token reduction for efficient large multimodal models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.656404Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:678df7d9f4e0cabb18dc117f95bc7b8f72828b02feb7327c4181e09b8d0637de","observation_id":"54baaea2-b91c-4979-b628-4e52c11da981","resolution":{"observed_at":"2026-08-06T22:22:37.656404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-06T22:22:37.694996Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.694996Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:77ae2002e93527d87ba9385272d64deef2acdcda3b5c91b188c96b8c47449f86","observation_id":"503b3c1d-7e40-4cf9-b445-c27e3b6e1c85","resolution":{"observed_at":"2026-08-06T22:22:37.694996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10994","last_updated":"2024-12-17T02:05:27Z","snapshot_observed_at":"2026-07-06T19:16:36.688367Z","submitted_at":"2024-09-17T08:56:27Z","title":"Less is More: A Simple yet Effective Token Reduction Method for Efficient Multi-modal LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.10994","snapshot_observed_at":"2026-08-06T22:22:37.715782Z","title":"Less is more: A sim- ple yet effective token reduction method for efficient multi- modal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.715782Z"},"links":{"cited_paper":"/paper/2409.10994","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:73d3a5291712fa8ec30f858341b8853fe7fbdd460de94eb77ff17cff9dea5990","observation_id":"8632f363-36a1-41db-8844-8a1c699956a1","resolution":{"observed_at":"2026-08-06T22:22:37.715782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.15033","last_updated":"2024-02-26T15:26:20Z","snapshot_observed_at":"2026-08-05T16:55:19.542695Z","submitted_at":"2023-05-24T11:18:00Z","title":"SmartTrim: Adaptive Tokens and Attention Pruning for Efficient Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2305.15033","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.15033","snapshot_observed_at":"2026-08-06T22:22:38.192026Z","title":"SmartTrim: Adaptive Tokens and Attention Pruning for Efficient Vision-Language Models","venue":"cs.CL","work_id":"c57fed02-5dc9-471f-afe7-8e1d85f74346","year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.743642Z"},"links":{"cited_paper":"/paper/2305.15033","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:0f2e6f4abcbaff1013cbbca92fc61090637d1179c325aae8165885018831236d","observation_id":"7f131960-fa44-40a5-9d6e-9b4f5b04fd33","resolution":{"observed_at":"2026-08-06T22:22:38.253638Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.03079","last_updated":"2024-02-04T08:23:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-06T13:04:39Z","title":"CogVLM: Visual Expert for Pretrained Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.03079","snapshot_observed_at":"2026-08-06T22:22:37.779534Z","title":"Cogvlm: Visual expert for pretrained language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.779534Z"},"links":{"cited_paper":"/paper/2311.03079","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:20eaa3925e0eb9008987cb5c355ba66929fb0b672bb80b60475707e707fa5f6d","observation_id":"9006af8c-53da-4f81-ac21-12430f38a059","resolution":{"observed_at":"2026-08-06T22:22:37.779534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15343","last_updated":"2023-09-27T12:05:41Z","snapshot_observed_at":"2026-07-06T15:08:30.190912Z","submitted_at":"2023-03-27T15:53:01Z","title":"Sigmoid Loss for Language Image Pre-Training","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.15343","snapshot_observed_at":"2026-08-06T22:22:37.804075Z","title":"Sigmoid loss for language image pre- training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.804075Z"},"links":{"cited_paper":"/paper/2303.15343","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:b72ba277ca60b0312ecb919f3dc50ba2a8bea46ec9f125074f762d521255ad24","observation_id":"740a8bef-e1ff-46b1-89b5-71d8a266bd68","resolution":{"observed_at":"2026-08-06T22:22:37.804075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.09554","last_updated":"2020-12-07T04:56:24Z","snapshot_observed_at":"2026-07-06T09:39:55.355752Z","submitted_at":"2020-07-19T01:45:02Z","title":"Referring Expression Comprehension: A Survey of Methods and Datasets","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.09554","snapshot_observed_at":"2026-08-06T22:22:37.839061Z","title":"Referring expression comprehension: A survey of methods and datasets","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.839061Z"},"links":{"cited_paper":"/paper/2007.09554","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:0ec4b53f5c297cb05e62f2c59556c91e529355dfa2198628b65d3c827820e722","observation_id":"f0abc5a8-7da7-4a9b-b116-4f52f17b07b3","resolution":{"observed_at":"2026-08-06T22:22:37.839061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01162","last_updated":"2025-02-12T04:35:28Z","snapshot_observed_at":"2026-07-06T19:09:14.628371Z","submitted_at":"2024-09-02T10:49:10Z","title":"Sparsity Meets Similarity: Leveraging Long-Tail Distribution for Dynamic Optimized Token Representation in Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01162","snapshot_observed_at":"2026-08-06T22:22:37.891208Z","title":"Balancing performance and efficiency: A multimodal large language model pruning method based image text interaction.ArXiv, abs/2409.01162,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.891208Z"},"links":{"cited_paper":"/paper/2409.01162","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:b8c724535b7cbdbc8566c63b6976ec2161d7778f10e2937e9277155a44727aef","observation_id":"00fd528a-580c-476a-9dff-37e14e2d1699","resolution":{"observed_at":"2026-08-06T22:22:37.891208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:38.627571Z","title":"Mm-vet: Evaluating large multimodal models for integrated capabilities, 2023","venue":null,"work_id":"f469aaa2-cde0-4d86-a9cf-fd5c4a90ee8f","year":2023},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.932357Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:7901993f6648d1b7e0ba23e5296da4f3da86cad924a2e891859b1f5166f0b390","observation_id":"f291d98b-bdaa-49bb-bf10-0afed6d4abc4","resolution":{"observed_at":"2026-08-06T22:22:38.683766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.07636","last_updated":"2022-12-05T13:53:51Z","snapshot_observed_at":"2026-07-06T14:18:10.647862Z","submitted_at":"2022-11-14T18:59:52Z","title":"EVA: Exploring the Limits of Masked Visual Representation Learning at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.07636","snapshot_observed_at":"2026-08-06T22:22:37.974207Z","title":"Eva: Exploring the limits of masked visual representation learning at scale","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:37.974207Z"},"links":{"cited_paper":"/paper/2211.07636","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:42faacd7d5a99c526b1be0de9acbca31423e2347fff4e97f82364c654f165930","observation_id":"a7432697-617b-4d76-9b68-62afca620f25","resolution":{"observed_at":"2026-08-06T22:22:37.974207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01818","last_updated":"2025-05-11T17:45:02Z","snapshot_observed_at":"2026-08-08T00:45:04.947316Z","submitted_at":"2024-12-02T18:57:40Z","title":"Beyond Text-Visual Attention: Exploiting Visual Cues for Effective Token Pruning in VLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.01818","snapshot_observed_at":"2026-08-06T22:22:38.007759Z","title":"[cls] attention is all you need for training- free visual token pruning: Make vlm inference faster","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:38.007759Z"},"links":{"cited_paper":"/paper/2412.01818","citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:ec643081764541793751904fbb0401b68ead832f73f2c9f2a77375fec218050b","observation_id":"9769bad8-5242-44f0-8796-3fb811209059","resolution":{"observed_at":"2026-08-06T22:22:38.007759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:22:38.460845Z","title":"Sparsevlm: Visual token sparsification for efficient vision- language model inference","venue":null,"work_id":"d6f96e8c-f04c-480a-932c-7ee1152e0a65","year":2024},"citing_paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:38.056203Z"},"links":{"citing_paper":"/paper/2506.21873"},"observation_digest":"sha256:7d14f1ef21d98fb20646f23851aa583ea4caa6bac06af6ed78483f841e49ef01","observation_id":"8406bbc7-9c0c-4615-b864-0bf5b81d436a","resolution":{"observed_at":"2026-08-06T22:22:38.533842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.21873","last_updated":"2025-06-27T03:11:22Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T00:45:25.154379Z","submitted_at":"2025-06-27T03:11:22Z","title":"Grounding-Aware Token Pruning: Recovering from Drastic Performance Drops in Visual Grounding Caused by Pruning"},"reference_resolution":{"displayed":33,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":1,"verified_fuzzy":14},"total_outbound_references":33},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 33 of 33 outbound references and 4 inbound Pith citation observations for arXiv:2506.21873."}