{"as_of":"2026-08-09T19:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a41439fe98e73cd90409c4fc482fa5113263a03e84c8f98febbadb582bed595d","coverage":[{"denominator":64,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":64,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:45:32.061891Z","state":"measured"},{"denominator":64,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":64,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.17812/citation-record","integrity":"/paper/2505.17812/integrity","json":"/paper/2505.17812/citation-record.json","paper":"/paper/2505.17812"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-07T14:45:26.861270Z","title":"Llama: Open and efficient foundation language models.arXiv preprint arXiv:2302.13971, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:26.861270Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:980dc5af0f9d055380e7ddb99d334cdec483c8b6a61034b1b8d03807c158179f","observation_id":"882aa6cd-9854-4dfc-8100-a9006e0ca4f8","resolution":{"observed_at":"2026-08-07T14:45:26.861270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T14:45:26.929012Z","title":"Llama 2: Open foundation and fine-tuned chat models.arXiv preprint arXiv:2307.09288, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:26.929012Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:67fe137d7e6c4459339af656fa52cdf4a6caf525ed557006c35cdfa47d3fa42f","observation_id":"737a10df-2d22-40cb-bfce-2668d47766e6","resolution":{"observed_at":"2026-08-07T14:45:26.929012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:39.164872Z","title":"Visual instruction tuning.Adv","venue":null,"work_id":"6ca83564-7f5a-4dbd-ad4f-f364175f4a91","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.001968Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:63e256b3255db32004305720c87c74f25a18f7f86802c147a509d9d9227128b2","observation_id":"81c64c90-6c7e-4622-bfbb-60dfd6c91d37","resolution":{"observed_at":"2026-08-07T14:45:39.249729Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03744","last_updated":"2024-05-15T19:22:44Z","snapshot_observed_at":"2026-07-06T16:28:22.350574Z","submitted_at":"2023-10-05T17:59:56Z","title":"Improved Baselines with Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03744","snapshot_observed_at":"2026-08-07T14:45:27.094656Z","title":"Improved baselines with visual instruction tuning.arXiv preprint arXiv:2310.03744, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.094656Z"},"links":{"cited_paper":"/paper/2310.03744","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:e1c97c053b2233edfb250a95d6734e3193d41fef3772dc8951bba09a2bf5c3cd","observation_id":"4455e4e8-22fe-4da9-847f-c24901bf3ee4","resolution":{"observed_at":"2026-08-07T14:45:27.094656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06500","last_updated":"2023-06-15T08:00:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-11T00:38:10Z","title":"InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06500","snapshot_observed_at":"2026-08-07T14:45:27.170843Z","title":"Instructblip: Towards general-purpose vision-language models with instruction tuning.arXiv preprint arXiv:2305.06500, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.170843Z"},"links":{"cited_paper":"/paper/2305.06500","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:1eb7b902b02c7644c131e6067604859ccdf9bc578284152b6492f8e55bc7b9a5","observation_id":"c8a21e09-834d-476c-9bdd-064d5f45d427","resolution":{"observed_at":"2026-08-07T14:45:27.170843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-07T14:45:27.251904Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models.arXiv preprint arXiv:2304.10592, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.251904Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:d629dd5cd728cea40e3b102ec6be31e2f0d86dd80453af4f843d5d14994911ff","observation_id":"8f887f16-e771-46b9-8839-be6850db0e26","resolution":{"observed_at":"2026-08-07T14:45:27.251904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-07T14:45:27.325359Z","title":"Qwen technical report.arXiv preprint arXiv:2309.16609, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.325359Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:8ac8819be8224b4f60251964ef7bda2a0ed11e4715fcf5a2628a69416cfbbbdf","observation_id":"9e209bd6-89d7-4769-830f-56caefcb9c6f","resolution":{"observed_at":"2026-08-07T14:45:27.325359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-07T14:45:27.396913Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution.arXiv preprint arXiv:2409.12191, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.396913Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:f82020a3ca7bb7016afd1ce9d5c4d70c1e9dd05b7186ddf1fc5580c1f47b52f7","observation_id":"cfe75358-ff73-4c7d-8cbe-b2cfef5b7b6e","resolution":{"observed_at":"2026-08-07T14:45:27.396913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.18930","last_updated":"2025-04-01T18:36:08Z","snapshot_observed_at":"2026-08-06T19:08:14.800394Z","submitted_at":"2024-04-29T17:59:41Z","title":"Hallucination of Multimodal Large Language Models: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.18930","snapshot_observed_at":"2026-08-07T14:45:27.497184Z","title":"Hallucination of multimodal large language models: A survey.arXiv preprint arXiv:2404.18930, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.497184Z"},"links":{"cited_paper":"/paper/2404.18930","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:6820509f86e666dc30f1acaf7d38f1b760c1c302ce488c88c908c34504e9dd0b","observation_id":"f1d4b41e-4c28-4897-82a9-ea75f1ccdfe5","resolution":{"observed_at":"2026-08-07T14:45:27.497184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:38.965099Z","title":"Nullu: Mitigating object hallucinations in large vision-language models via halluspace projection","venue":null,"work_id":"b7d4ec7d-0199-4c61-a00d-64e58389c73a","year":2025},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.582588Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:27652973b0681066cc11d5bb6ad11148fc4d7f2d281e915522c2782012be3fcd","observation_id":"66e2ef9a-7aaf-41da-9155-14405f3ba299","resolution":{"observed_at":"2026-08-07T14:45:39.061415Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:27.675017Z","title":"Truthprint: Mitigating lvlm object hallucination via latent truthful-guided pre-intervention.arXiv preprint arXiv:2503.10602, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.675017Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:85ac24915f12a588536f05d3a1938a6af75afa4b8f2208947c7d53a1dd1b86ce","observation_id":"f12fe377-086e-4152-a5f1-8a0c152760c5","resolution":{"observed_at":"2026-08-07T14:45:27.675017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:38.822045Z","title":"Analyzing and mitigating object hallucination in large vision-language models","venue":null,"work_id":"24e3ed51-e0a2-4961-b706-ea3954871d7b","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.767172Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:55a8df5ad937d50c9bca94ece89fd4fe9865aa591d1a65381cd8eafd4a96294d","observation_id":"0473397b-fec5-45da-b3c0-13bccfee3515","resolution":{"observed_at":"2026-08-07T14:45:38.889223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14565","last_updated":"2024-03-19T22:53:25Z","snapshot_observed_at":"2026-08-06T22:33:34.254048Z","submitted_at":"2023-06-26T10:26:33Z","title":"Mitigating Hallucination in Large Multi-Modal Models via Robust Instruction Tuning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14565","snapshot_observed_at":"2026-08-07T14:45:27.838292Z","title":"Mitigating hallucination in large multi-modal models via robust instruction tuning.arXiv preprint arXiv:2306.14565, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.838292Z"},"links":{"cited_paper":"/paper/2306.14565","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:0228c2df025538b500c4f7df3512b64d514c37b8306ac93d5f7c0b3ed5ff32ee","observation_id":"7bd6d8df-457b-42e7-abf7-ea2e01419e83","resolution":{"observed_at":"2026-08-07T14:45:27.838292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:38.627244Z","title":"Hallucination augmented contrastive learning for multimodal large language model","venue":null,"work_id":"f9bbeb39-15ae-455c-a059-4b58bc3f0341","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:27.945183Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:37d68e0ef49f166c1b8f18eaaffa54dfada4c71808f7d1709553f7f96b446ff9","observation_id":"c1a8a98d-b379-472b-9222-2525fcf3a5c7","resolution":{"observed_at":"2026-08-07T14:45:38.723563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:38.453564Z","title":"Exposing and mitigating spurious correlations for cross-modal retrieval","venue":null,"work_id":"9ff1b74d-7c5f-4809-b5ed-0bb1e57dd41f","year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.024239Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:ffd05d9e89cba0c88e7056294495bed883f6ca175cd3dfe66df1b0e2f9475fbe","observation_id":"4c94a0d3-e0d6-4f2f-b22b-47cf3c77c1a1","resolution":{"observed_at":"2026-08-07T14:45:38.528588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:38.268940Z","title":"Mitigating object hallucinations in large vision-language models through visual contrastive decoding","venue":null,"work_id":"80c34643-335c-4a9d-8309-5c5b0e7577fd","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.107255Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:7ba09291875a195c98f03fccd9c29258e0de668a687f680659ce3f8543104eaf","observation_id":"5671c12d-a4aa-4613-a229-587492166ecc","resolution":{"observed_at":"2026-08-07T14:45:38.339332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:38.044229Z","title":"Debiasing large visual language models","venue":null,"work_id":"77aafbe6-2777-44c0-949f-29ef990aacd6","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.181188Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:9d1bf8818067630d70af0edce2377c355b9a260f2a6c0e18a968050f744494d9","observation_id":"196a4003-abf4-40d3-8f5e-25063d5a3327","resolution":{"observed_at":"2026-08-07T14:45:38.147222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:37.822854Z","title":"Halc: Object hallucination reduction via adaptive focal-contrast decoding","venue":null,"work_id":"27d4bb6d-5b3e-4e2b-af98-41fad3b7431e","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.261744Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:035117a5579e1dc0221faf233456df3e9d68611159736ff482850b735347afa7","observation_id":"65bf3d5c-5958-41c4-ae01-28e195126a20","resolution":{"observed_at":"2026-08-07T14:45:37.936158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15268","last_updated":"2024-11-22T12:22:21Z","snapshot_observed_at":"2026-08-08T10:38:10.039848Z","submitted_at":"2024-11-22T12:22:21Z","title":"ICT: Image-Object Cross-Level Trusted Intervention for Mitigating Object Hallucination in Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15268","snapshot_observed_at":"2026-08-07T14:45:28.344201Z","title":"Ict: Image-object cross-level trusted intervention for mitigating object hallucination in large vision-language models.arXiv preprint arXiv:2411.15268, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.344201Z"},"links":{"cited_paper":"/paper/2411.15268","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:791a28c11e4acbe83abe7bf8be47c2f724a93e940f81bca6db7159dc9c78d40d","observation_id":"ad860591-dd20-4f99-9944-9c9f1db318f6","resolution":{"observed_at":"2026-08-07T14:45:28.344201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:37.636188Z","title":"Reducing hallucinations in large vision-language models via latent space steering","venue":null,"work_id":"a4b625b7-26c9-47a9-9ecc-83b62dad753c","year":2025},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.404847Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:d8750aa3d33f8b082bc65a9462616664ae0392cbef11c9dac75d5a6ef0f3df3e","observation_id":"25ab535f-ea73-4a22-8646-e069e69f639a","resolution":{"observed_at":"2026-08-07T14:45:37.709424Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13891","last_updated":"2025-03-18T04:34:43Z","snapshot_observed_at":"2026-08-07T16:55:58.912014Z","submitted_at":"2025-03-18T04:34:43Z","title":"Where do Large Vision-Language Models Look at when Answering Questions?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.13891","snapshot_observed_at":"2026-08-07T14:45:28.465442Z","title":"Where do large vision-language models look at when answering questions?arXiv preprint arXiv:2503.13891, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.465442Z"},"links":{"cited_paper":"/paper/2503.13891","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:afaae7dca6db0cc249ab58d14a993656c0034de0a004734f04debad2e442deb3","observation_id":"0cfe0c03-bf00-4cff-95a9-3fae186b43fb","resolution":{"observed_at":"2026-08-07T14:45:28.465442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:37.408817Z","title":"Lvlm-intrepret: An interpretability tool for large vision-language models, 2024","venue":null,"work_id":"f984ae6a-fa08-4c61-be50-83db49fa5a63","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.517311Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:b0d4db04cafcbdc6448742f2799f02606a3489ece0259faea17d3ffc4b37a76b","observation_id":"49f3903b-7a7a-4ab0-9e11-feaee20efb4a","resolution":{"observed_at":"2026-08-07T14:45:37.516328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.03321","last_updated":"2025-03-05T09:55:07Z","snapshot_observed_at":"2026-08-07T17:28:11.499465Z","submitted_at":"2025-03-05T09:55:07Z","title":"See What You Are Told: Visual Attention Sink in Large Multimodal Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.03321","snapshot_observed_at":"2026-08-07T14:45:28.622401Z","title":"See what you are told: Visual attention sink in large multimodal models.arXiv preprint arXiv:2503.03321, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.622401Z"},"links":{"cited_paper":"/paper/2503.03321","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:7ffc16e4b9c98b565de656f6bb8def90345055e214fc3373512ccfd10c560e51","observation_id":"26f19368-69be-4297-ad00-dc9f67a84f3a","resolution":{"observed_at":"2026-08-07T14:45:28.622401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16588","last_updated":"2024-04-12T09:38:33Z","snapshot_observed_at":"2026-08-07T09:40:32.614733Z","submitted_at":"2023-09-28T16:45:46Z","title":"Vision Transformers Need Registers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16588","snapshot_observed_at":"2026-08-07T14:45:28.704062Z","title":"Vision transformers need registers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.704062Z"},"links":{"cited_paper":"/paper/2309.16588","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:16e8c8788dced8b3ba55476eb0f49a056de2da61b475f8e62a245838b31f7d74","observation_id":"3f9dcb33-210b-4d5c-bbad-5ea4f33e9034","resolution":{"observed_at":"2026-08-07T14:45:28.704062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.17762","last_updated":"2024-08-14T16:00:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-27T18:55:17Z","title":"Massive Activations in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.17762","snapshot_observed_at":"2026-08-07T14:45:28.759191Z","title":"Massive activations in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.759191Z"},"links":{"cited_paper":"/paper/2402.17762","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:0d3b228e082f6e4a9b2f0428ee6a5fa1c191fac2d8b17cf20057057c1585dc75","observation_id":"bb0f94bf-7c2a-4275-9518-11fbd28bcacf","resolution":{"observed_at":"2026-08-07T14:45:28.759191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.02156","last_updated":"2019-03-29T23:48:52Z","snapshot_observed_at":"2026-08-04T05:57:07.163978Z","submitted_at":"2018-09-06T18:25:18Z","title":"Object Hallucination in Image Captioning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.02156","snapshot_observed_at":"2026-08-07T14:45:28.823154Z","title":"Object hallucination in image captioning.arXiv preprint arXiv:1809.02156, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.823154Z"},"links":{"cited_paper":"/paper/1809.02156","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:f02bf0579902f0a05ce3dd63dbb0622bac07d2e0de44d8aa2d3648047c9e35c7","observation_id":"ac162876-881f-4b81-8f13-701f6b5eeb37","resolution":{"observed_at":"2026-08-07T14:45:28.823154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:37.189345Z","title":"mplug-owl2: Revolutionizing multi-modal large language model with modality collaboration","venue":null,"work_id":"e03b73c7-1956-4dcf-8287-78e475734bd3","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.894823Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:91c4102b46217dc0a1887549aa64be1dd7845421078f2bffe7b3414b9c4a8cb4","observation_id":"d6fcddd8-1821-40b3-b54b-4f0175019e62","resolution":{"observed_at":"2026-08-07T14:45:37.289552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-07T14:45:28.975925Z","title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:28.975925Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:d3311602ee3eb60cf7aaea876608eb400962c1eb146f566cfd86b14ebc61df70","observation_id":"de3e5d8d-fea9-4b4c-892f-d99344a2f540","resolution":{"observed_at":"2026-08-07T14:45:28.975925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:37.003297Z","title":"Llava-phi: Efficient multi-modal assistant with small language model","venue":null,"work_id":"bcffa188-487b-4487-854f-ff59e96ea6ca","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.029305Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:954026659ae78a505e733150e0506d0f7570850d7dbc7e4b1a89fb5e3befbcb3","observation_id":"9ab16929-c20f-4955-b6bc-ff8ad009c7d1","resolution":{"observed_at":"2026-08-07T14:45:37.107790Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05525","last_updated":"2024-03-11T16:47:41Z","snapshot_observed_at":"2026-08-05T16:51:32.094151Z","submitted_at":"2024-03-08T18:46:00Z","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05525","snapshot_observed_at":"2026-08-07T14:45:29.111032Z","title":"Deepseek-vl: towards real-world vision-language understanding.arXiv preprint arXiv:2403.05525, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.111032Z"},"links":{"cited_paper":"/paper/2403.05525","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:814fc632ce8481d0dc3dff044276bec7fd53c3a2e539b57944dd2ff7a1346309","observation_id":"fc8a6a40-c744-42de-9ae0-c7cf38e76412","resolution":{"observed_at":"2026-08-07T14:45:29.111032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:36.829472Z","title":"Detecting and preventing hallucinations in large vision language models","venue":null,"work_id":"ca67cf20-1315-478e-a423-b5e9e8799b9f","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.202872Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:2e7c0e1ed0cf97f8a5772d6ebb4802d48ba0e6320b54c98864a1790f49ad1070","observation_id":"cdd349ec-b545-4bf8-9229-43c1bfdc52b9","resolution":{"observed_at":"2026-08-07T14:45:36.895414Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14525","last_updated":"2023-09-25T20:59:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-25T20:59:33Z","title":"Aligning Large Multimodal Models with Factually Augmented RLHF","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14525","snapshot_observed_at":"2026-08-07T14:45:29.287602Z","title":"Aligning large multimodal models with factually augmented rlhf.arXiv preprint arXiv:2309.14525, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.287602Z"},"links":{"cited_paper":"/paper/2309.14525","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:3238d517f0c6007d09af3e80eb55079531b7daff3798c43bd6d9345f42be2f57","observation_id":"125f43ae-65de-4b25-be32-532fb592a6ec","resolution":{"observed_at":"2026-08-07T14:45:29.287602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:36.640212Z","title":"Dress: Instructing large vision-language models to align and interact with humans via natural language feedback","venue":null,"work_id":"7984a78d-72b8-4d50-975f-87bc3d8a2a3f","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.363088Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:d5ee9bdbb0aa6bf1f613726851d82319cf308e41c584555d85d3d791a8b63e88","observation_id":"85fa9ad1-866f-4ebb-b530-46f4fd57c7f6","resolution":{"observed_at":"2026-08-07T14:45:36.712636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16045","last_updated":"2024-12-11T02:46:29Z","snapshot_observed_at":"2026-08-08T00:11:31.801569Z","submitted_at":"2023-10-24T17:58:07Z","title":"Woodpecker: Hallucination Correction for Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16045","snapshot_observed_at":"2026-08-07T14:45:29.439784Z","title":"Woodpecker: Hallucination correction for multimodal large language models.arXiv preprint arXiv:2310.16045, 2023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.439784Z"},"links":{"cited_paper":"/paper/2310.16045","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:428c0e3f74ffdcdfae359eb4f6984e6805fa95c6794dc411a8224cef52470bb3","observation_id":"7f9b0304-50f3-4f2e-b0ec-d8cabe019404","resolution":{"observed_at":"2026-08-07T14:45:29.439784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:36.423142Z","title":"Mitigating object hallucination in large vision-language models via image-grounded guidance","venue":null,"work_id":"e1018798-deae-4653-88a9-25ff9175df93","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.524121Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:180c763df68243c89203b2b6782c358d726a9881a768b04f91f89514256086ab","observation_id":"a0f958ef-718f-482d-b0f9-39b5dd6e23b3","resolution":{"observed_at":"2026-08-07T14:45:36.500425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:36.225215Z","title":"Paying more attention to image: A training-free method for alleviating hallucination in lvlms","venue":null,"work_id":"330862ea-3e4f-420e-904e-769e2bbfedd8","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.593024Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:9a8dc48e35e535f9745282c6ba99f6ef5c75f7ea75053db52a065d4372599de2","observation_id":"ff3e8d54-95f9-4d80-b404-ff620b82690d","resolution":{"observed_at":"2026-08-07T14:45:36.337308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18476","last_updated":"2024-02-28T16:57:22Z","snapshot_observed_at":"2026-07-06T17:36:56.979448Z","submitted_at":"2024-02-28T16:57:22Z","title":"IBD: Alleviating Hallucinations in Large Vision-Language Models via Image-Biased Decoding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18476","snapshot_observed_at":"2026-08-07T14:45:29.672408Z","title":"Ibd: Alleviating hallucinations in large vision-language models via image-biased decoding.arXiv preprint arXiv:2402.18476, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.672408Z"},"links":{"cited_paper":"/paper/2402.18476","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:635c4ab9f7ab766dc6c6e0b28bac08d40d35fa29ad3d918723853d7f74f2f30e","observation_id":"b6698032-96ff-40d5-a3aa-b68b7600a43a","resolution":{"observed_at":"2026-08-07T14:45:29.672408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:35.993649Z","title":"Opera: Alleviating hallucination in multi-modal large language models via over-trust penalty and retrospection-allocation","venue":null,"work_id":"1b071a87-a549-45fc-b6a6-e3907fa63c36","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.758823Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:98b3c1f057e134655e2d5b697359db67bb420c9201f1b5d82f8a8db97dfeaffb","observation_id":"a24c3d4f-df58-4cf0-bcd3-cb50f4e2da2d","resolution":{"observed_at":"2026-08-07T14:45:36.102530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:35.803501Z","title":"Multi-modal hallucination control by visual information grounding","venue":null,"work_id":"c5f47232-e459-4518-a067-990c222d8a15","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.857326Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:0af756ce4e1405fc25cd1bd1e33dfe7e4074b3d8188a39321673f6ae8a86031b","observation_id":"77a176db-9161-4150-9360-879d127009a8","resolution":{"observed_at":"2026-08-07T14:45:35.874270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02355","last_updated":"2025-04-22T16:15:47Z","snapshot_observed_at":"2026-08-08T23:26:11.950951Z","submitted_at":"2024-10-03T10:06:27Z","title":"AlphaEdit: Null-Space Constrained Knowledge Editing for Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02355","snapshot_observed_at":"2026-08-07T14:45:29.943737Z","title":"Alphaedit: Null-space constrained knowledge editing for language models.arXiv preprint arXiv:2410.02355, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:29.943737Z"},"links":{"cited_paper":"/paper/2410.02355","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:cff4f458a310c18b966e158dbb4f854d52afac4d7f8f12bffe0e8e9e95c7f4c4","observation_id":"a06fcb51-2132-4696-a792-026aa2d63c38","resolution":{"observed_at":"2026-08-07T14:45:29.943737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12718","last_updated":"2025-03-14T04:38:44Z","snapshot_observed_at":"2026-07-06T18:33:02.286642Z","submitted_at":"2024-06-18T15:38:41Z","title":"Mitigating Object Hallucinations in Large Vision-Language Models with Assembly of Global and Local Attention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12718","snapshot_observed_at":"2026-08-07T14:45:30.042461Z","title":"Agla: Mitigating object hallucinations in large vision-language models with assembly of global and local attention.arXiv preprint arXiv:2406.12718, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.042461Z"},"links":{"cited_paper":"/paper/2406.12718","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:4794af0cbbbdbd05186e0d9210d930053948e326f69474c900cd057fd9837b1c","observation_id":"b2eb7736-2804-407e-9555-9576ea099f9f","resolution":{"observed_at":"2026-08-07T14:45:30.042461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:35.576709Z","title":"Grad-cam: Visual explanations from deep networks via gradient-based localization","venue":null,"work_id":"e3e66d6d-cc93-458e-8074-5719b26f96ef","year":2017},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.110923Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:4629cf757d1c29d6cc3e2d5935de639049b75cee3bffecafe5f59a3110a99ca9","observation_id":"b4fefc28-d1a6-4b60-ab67-38a9e25413b6","resolution":{"observed_at":"2026-08-07T14:45:35.675866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:35.349141Z","title":"Grad-cam++: Generalized gradient-based visual explanations for deep convolutional networks","venue":null,"work_id":"f02203ef-3be5-47b2-9484-049127882a27","year":2018},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.219446Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:a4364dda90ed591c430a05ffac91493ea3256b612e3fb7beed27683f0b0e245b","observation_id":"847e83b9-e414-4ccd-9632-e2aecb5eb7ab","resolution":{"observed_at":"2026-08-07T14:45:35.484548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:35.165458Z","title":"Generic attention-model explainability for interpreting bi-modal and encoder-decoder transformers","venue":null,"work_id":"d42ebcce-ae0d-4989-bbd4-23b2f763755e","year":2021},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.320579Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:51f65eef25d104f74dbadc21896ba616c840f66e01c09efcb804227ec1b9e989","observation_id":"3284945c-49ca-4b6d-9391-ef63178ae38a","resolution":{"observed_at":"2026-08-07T14:45:35.230180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:34.941083Z","title":"Transformer interpretability beyond attention visualization","venue":null,"work_id":"9578935f-d099-4435-acf9-5cd505e3b42a","year":2021},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.399480Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:bda7c7dcee23e8d876785b3709c41f6f941ff8ba61476754c6a57e40c00d0424","observation_id":"8272d023-731e-4d73-b974-2645efdb56d7","resolution":{"observed_at":"2026-08-07T14:45:35.077456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:34.745867Z","title":"Vl- interpret: An interactive visualization tool for interpreting vision-language transformers","venue":null,"work_id":"4887b767-6fcd-45f9-94f1-ceffc37437a6","year":2022},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.463305Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:a24800084035159e47062b9485ae5ab2e4d78570218a9c3d555949117dedb986","observation_id":"82447b30-d45b-4a50-a20a-83d9ff7a2806","resolution":{"observed_at":"2026-08-07T14:45:34.838376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01487","last_updated":"2025-05-06T14:49:11Z","snapshot_observed_at":"2026-07-06T20:00:09.775334Z","submitted_at":"2024-12-02T13:39:29Z","title":"FastRM: An efficient and automatic explainability framework for multimodal generative models","version":4},"cited_work":{"arxiv_id":"2412.01487","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.01487","snapshot_observed_at":"2026-08-07T14:45:32.737877Z","title":"FastRM: An efficient and automatic explainability framework for multimodal generative models","venue":"cs.AI","work_id":"6959f19c-88ce-4e8e-bf1b-b98338b9b2c0","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.527268Z"},"links":{"cited_paper":"/paper/2412.01487","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:26600d467f7354b30819322b931ad917399facf294990813cc6a6213f5f5b97e","observation_id":"2c82cd78-bd42-4df1-8c17-3405ca1d6741","resolution":{"observed_at":"2026-08-07T14:45:32.849467Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14612","last_updated":"2024-05-28T11:18:20Z","snapshot_observed_at":"2026-07-06T18:18:39.093732Z","submitted_at":"2024-05-23T14:24:23Z","title":"Explaining Multi-modal Large Language Models by Analyzing their Vision Perception","version":2},"cited_work":{"arxiv_id":"2405.14612","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.14612","snapshot_observed_at":"2026-08-07T14:45:32.487571Z","title":"Explaining Multi-modal Large Language Models by Analyzing their Vision Perception","venue":"cs.CV","work_id":"21632da1-a336-4490-bdcb-03a02bddcc70","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.624312Z"},"links":{"cited_paper":"/paper/2405.14612","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:dc5c3a39485a04ffb2c299d31ad3276f72757605a6bfe582f5ae0a45f98894a7","observation_id":"3519c456-6106-4f20-b0df-74c2d34808f7","resolution":{"observed_at":"2026-08-07T14:45:32.614357Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06579","last_updated":"2024-10-17T01:17:25Z","snapshot_observed_at":"2026-07-06T18:28:20.634381Z","submitted_at":"2024-06-04T13:52:54Z","title":"From Redundancy to Relevance: Information Flow in LVLMs Across Reasoning Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06579","snapshot_observed_at":"2026-08-07T14:45:30.728861Z","title":"From redundancy to relevance: Information flow in lvlms across reasoning tasks.arXiv preprint arXiv:2406.06579, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.728861Z"},"links":{"cited_paper":"/paper/2406.06579","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:cbde7e6e29974d98de1c9c8a1302d3fbd26c68337153c2210aa48fddd7931b4a","observation_id":"7089ab6c-d0a4-4247-8caf-b136bf755174","resolution":{"observed_at":"2026-08-07T14:45:30.728861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07470","last_updated":"2024-06-11T12:30:02Z","snapshot_observed_at":"2026-07-06T16:46:42.817599Z","submitted_at":"2023-11-13T17:03:02Z","title":"Finding and Editing Multi-Modal Neurons in Pre-Trained Transformers","version":2},"cited_work":{"arxiv_id":"2311.07470","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.07470","snapshot_observed_at":"2026-08-07T14:45:32.186301Z","title":"Finding and Editing Multi-Modal Neurons in Pre-Trained Transformers","venue":"cs.CL","work_id":"47a256ff-651a-4b55-bcb1-40e67acfdce4","year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.773041Z"},"links":{"cited_paper":"/paper/2311.07470","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:e4c80d40f175c830bfb3b2050e8e4c19f3ac3b02e321fd0e0d52319bb30d920f","observation_id":"bc37ab4e-748e-45aa-aa7f-6b302367a133","resolution":{"observed_at":"2026-08-07T14:45:32.303375Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07397","last_updated":"2024-02-23T07:54:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-13T15:25:42Z","title":"AMBER: An LLM-free Multi-dimensional Benchmark for MLLMs Hallucination Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07397","snapshot_observed_at":"2026-08-07T14:45:30.862604Z","title":"An llm-free multi-dimensional benchmark for mllms hallucination evaluation.arXiv preprint arXiv:2311.07397, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.862604Z"},"links":{"cited_paper":"/paper/2311.07397","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:8f50a25be7802e06eec0e0231397f73012476c51d82195e71f8da1a26e04d387","observation_id":"6ccb7e24-eaa5-4926-accd-cfd30a528366","resolution":{"observed_at":"2026-08-07T14:45:30.862604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:34.562138Z","title":"Evaluating object hallucination in large vision-language models","venue":null,"work_id":"3b26bef6-1846-4f79-a541-6f37d6c548bd","year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:30.949942Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:3c803f218b32f18872ca866486bad7dec99f77731926776bc3e97506145b3b3c","observation_id":"1889192c-40e8-4d30-96c2-56425dc377f8","resolution":{"observed_at":"2026-08-07T14:45:34.654011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:34.381921Z","title":"Aligning large multimodal models with factually augmented rlhf","venue":null,"work_id":"c816e833-1883-48b9-b589-41f3225a9ee4","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.024274Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:1fa90887a1ff1eb13199bda7ca489b3a114fc57151461b255047c38cc4bcf492","observation_id":"3e5446db-90a3-421c-a606-095f94019fde","resolution":{"observed_at":"2026-08-07T14:45:34.471205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:34.202590Z","title":"Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training.Adv","venue":null,"work_id":"773a6253-fc06-44af-9471-312b8c2d4838","year":2022},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.070353Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:88c60328846543fe8f6145aef6da9e25bb9787b5c1a481fbb542409d920048a7","observation_id":"707ed996-fdbb-4892-94d2-a0dcd63b2e9c","resolution":{"observed_at":"2026-08-07T14:45:34.299454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13394","snapshot_observed_at":"2026-08-07T14:45:31.150819Z","title":"Mme: A comprehensive evaluation benchmark for multimodal large language models.arXiv preprint arXiv:2306.13394, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.150819Z"},"links":{"cited_paper":"/paper/2306.13394","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:0fcc04da3158db23da10917da27ef308a45b7654e320c210f4dbf1369c2dffd1","observation_id":"6f82f695-ebfd-409f-8796-bc9795f604ae","resolution":{"observed_at":"2026-08-07T14:45:31.150819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:34.046267Z","title":"Gqa: A new dataset for real-world visual reasoning and compositional question answering","venue":null,"work_id":"3ed788fc-3892-421d-83d2-f0f5dabb67f0","year":2019},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.203608Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:efa358bff1950beb60614b18994e4be812f1695ba0f8f232e5ac2908909c57e3","observation_id":"4fe76543-d92b-4d9d-af14-a3bec28a87d2","resolution":{"observed_at":"2026-08-07T14:45:34.114986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1702.01806","last_updated":"2017-06-14T01:00:18Z","snapshot_observed_at":"2026-07-06T05:29:01.228177Z","submitted_at":"2017-02-06T22:08:46Z","title":"Beam Search Strategies for Neural Machine Translation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1702.01806","snapshot_observed_at":"2026-08-07T14:45:31.289807Z","title":"Beam search strategies for neural machine translation.arXiv preprint arXiv:1702.01806, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.289807Z"},"links":{"cited_paper":"/paper/1702.01806","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:4da429290355cdfd8b8bcb0b93fa3c46a1d0ff098de05e11cabd6fb1f5bd0246","observation_id":"ebdf2516-b466-4eaa-ace5-f91f494677cb","resolution":{"observed_at":"2026-08-07T14:45:31.289807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.03883","last_updated":"2024-03-11T02:01:09Z","snapshot_observed_at":"2026-08-06T11:43:44.726916Z","submitted_at":"2023-09-07T17:45:31Z","title":"DoLa: Decoding by Contrasting Layers Improves Factuality in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.03883","snapshot_observed_at":"2026-08-07T14:45:31.400584Z","title":"Dola: Decoding by contrasting layers improves factuality in large language models.arXiv preprint arXiv:2309.03883, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.400584Z"},"links":{"cited_paper":"/paper/2309.03883","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:5cb011c4b416f2d79974adb5398714feccc9e46d98e783bac999859bf563b03e","observation_id":"4de1a212-ef38-42f1-8a30-cf3c92771f3c","resolution":{"observed_at":"2026-08-07T14:45:31.400584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:33.872078Z","title":"Blended diffusion for text-driven editing of natural images","venue":null,"work_id":"1676116d-1243-4341-afe7-3d95f3bb7679","year":2022},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.506177Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:c6880e8ddb22db158007f35c1717af9f74132a0cdc97fef2b5eb5e15bc0ac705","observation_id":"f8bab591-d094-47bc-931f-2d936769cd25","resolution":{"observed_at":"2026-08-07T14:45:33.960642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:33.644842Z","title":"Unveiling typographic deceptions: Insights of the typographic vulnerability in large vision-language models","venue":null,"work_id":"0411d3c1-7e96-4f1f-a76f-ebbb9d0558d7","year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.619307Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:441e8ee15692603877e2bd66212070a5599c48196e91b4456491548ed05f395f","observation_id":"b87fca78-67d1-42ec-9d67-d2cc6903bcab","resolution":{"observed_at":"2026-08-07T14:45:33.708879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:33.517723Z","title":"Real-time single image and video super-resolution using an efficient sub-pixel convolutional neural network","venue":null,"work_id":"d3524808-e3f7-49c6-99f1-8f830a3684c8","year":2016},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.774717Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:ef7f9588e2149f278a2833c715c119d559369ad5442a4d2f1f6623984e48f4d8","observation_id":"2e495d30-7f69-4a63-92fe-e137a6360e93","resolution":{"observed_at":"2026-08-07T14:45:33.567854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:45:33.370366Z","title":"Towards interpreting visual information processing in vision-language models","venue":null,"work_id":"7976ea96-e580-466a-8dda-118d2807e397","year":2025},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.888895Z"},"links":{"citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:93585ebf8079b85ca59371f90364ff7b4330e8b91ab85ed72b5775dda8168c66","observation_id":"f7264abe-3342-48d1-8953-f240882fb7ca","resolution":{"observed_at":"2026-08-07T14:45:33.440316Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T14:45:31.991900Z","title":"Gpt-4 technical report.arXiv preprint arXiv:2303.08774, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:31.991900Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:49cdf7cf4a58987606e998e11046b46986b4c48baf21f329b2d16b1747edb5c2","observation_id":"bf173a22-c628-4ba6-8068-16487a728e86","resolution":{"observed_at":"2026-08-07T14:45:31.991900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-07T14:45:32.061891Z","title":"Please describe this image in detail","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:32.061891Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:c1e43a9407771666438d754820f179351b361db9dd981bbddd37f37c1ac52060","observation_id":"e4e60157-5b8f-4138-bf26-7e5b1f2d07bd","resolution":{"observed_at":"2026-08-07T14:45:32.061891Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations"},"reference_resolution":{"displayed":64,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":29,"verified_exact":3,"verified_fuzzy":31},"total_outbound_references":64},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 64 of 64 outbound references and 0 inbound Pith citation observations for arXiv:2505.17812."}