{"as_of":"2026-08-08T20:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2ab0377c73c16a632b18d308bb42fba2fc6f8a49d48b3a703dd278da0f9a0eb7","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":20,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:42:32.345547Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":10,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":"2412.02104","doi":"10.48550/arxiv.2412.02104","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey","venue":"arXiv (Cornell University)","work_id":"ca6d8610-ceff-4c4d-beff-d2f876503d3e","year":2024},"citing_paper":{"arxiv_id":"2502.02871","last_updated":"2026-04-20T02:18:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-05T04:05:27Z","title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-23T04:30:38.804702Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2502.02871"},"observation_digest":"sha256:afc1e2197d12fc3f8cfcf3b0aa73a727dcc26bbd22b5458aa247c8e02671b349","observation_id":"b9220300-1571-474a-9f81-64b400bdb17a","resolution":{"observed_at":"2026-05-23T04:32:32.888726Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-07T15:42:32.345547Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13965","last_updated":"2025-05-20T06:05:56Z","snapshot_observed_at":"2026-08-07T15:41:39.208060Z","submitted_at":"2025-05-20T06:05:56Z","title":"CAFES: A Collaborative Multi-Agent Framework for Multi-Granular Multimodal Essay Scoring","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T15:42:32.345547Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2505.13965"},"observation_digest":"sha256:addb194096c991646d0df72bd5116f8032bad7d9f7ec75397c4faf5665ebff07","observation_id":"33d8aa99-94d9-4781-9d5c-7fddd82368c0","resolution":{"observed_at":"2026-08-07T15:42:32.345547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-07T11:19:44.885332Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey.arXiv preprint arXiv:2412.02104, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02867","last_updated":"2025-06-04T15:00:58Z","snapshot_observed_at":"2026-08-08T18:19:35.851376Z","submitted_at":"2025-06-03T13:31:10Z","title":"Demystifying Reasoning Dynamics with Mutual Information: Thinking Tokens are Information Peaks in LLM Reasoning","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:19:44.885332Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2506.02867"},"observation_digest":"sha256:f25ecf821bf3a25e40ab5a1df1e41baf7af2d134fdbd2312d782dc7b42369792","observation_id":"dfe72256-4221-43c3-8a49-7ae6a8a501e9","resolution":{"observed_at":"2026-08-07T11:19:44.885332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-06T23:11:33.412140Z","title":"language-aligned","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2506.19497","last_updated":"2025-06-24T10:35:52Z","snapshot_observed_at":"2026-08-06T23:04:30.571051Z","submitted_at":"2025-06-24T10:35:52Z","title":"The time course of visuo-semantic representations in the human brain is captured by combining vision and language models","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T23:11:33.412140Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2506.19497"},"observation_digest":"sha256:c545a0154db642650f7bb0631edb27b9f4a97e60b3f71e7e92aed988943cdf02","observation_id":"58060430-2b71-43dc-95f0-8289fe674f1f","resolution":{"observed_at":"2026-08-06T23:11:33.412140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-06T22:22:42.220033Z","title":"and et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21812","last_updated":"2025-06-26T23:25:22Z","snapshot_observed_at":"2026-08-06T22:16:14.734980Z","submitted_at":"2025-06-26T23:25:22Z","title":"Towards Transparent AI: A Survey on Explainable Large Language Models","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T22:22:42.220033Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2506.21812"},"observation_digest":"sha256:fd272f25b8e88f4fb949c06347610eca9102aab336e5b8572ccab6f3f6e09d19","observation_id":"139b1daa-47d7-46ab-8b81-4a9e375eafdd","resolution":{"observed_at":"2026-08-06T22:22:42.220033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-06T20:43:59.901023Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01945","last_updated":"2025-07-09T16:30:21Z","snapshot_observed_at":"2026-08-06T20:37:19.917083Z","submitted_at":"2025-07-02T17:55:50Z","title":"LongAnimation: Long Animation Generation with Dynamic Global-Local Memory","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T20:43:59.901023Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2507.01945"},"observation_digest":"sha256:9f9f19fa1be654270091e078befd6d5d551c9f483c9376b8e5efa627eecb4537","observation_id":"6f02e5a7-a4d0-4819-a23e-96ab81ce1ba8","resolution":{"observed_at":"2026-08-06T20:43:59.901023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-06T20:25:59.826071Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02648","last_updated":"2025-07-03T14:12:43Z","snapshot_observed_at":"2026-08-08T08:58:36.818316Z","submitted_at":"2025-07-03T14:12:43Z","title":"Recourse, Repair, Reparation, & Prevention: A Stakeholder Analysis of AI Supply Chains","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T20:25:59.826071Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2507.02648"},"observation_digest":"sha256:798204384be0b88a6cc010c3cb1cecdb646c77a1ff8187b1f5c97e13614807f3","observation_id":"1a94cb14-3b90-4d4a-86b4-5b968eb7dec5","resolution":{"observed_at":"2026-08-06T20:25:59.826071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-06T15:50:15.335365Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14824","last_updated":"2025-07-20T05:08:28Z","snapshot_observed_at":"2026-08-08T17:00:15.188814Z","submitted_at":"2025-07-20T05:08:28Z","title":"Benchmarking Foundation Models with Multimodal Public Electronic Health Records","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T15:50:15.335365Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2507.14824"},"observation_digest":"sha256:bbe6f8e4120b12263c3e8f8f9f23e707832b50caafe0fbf54198fbd5cbc778f4","observation_id":"91c07260-d94d-42fb-9594-9506434b01f1","resolution":{"observed_at":"2026-08-06T15:50:15.335365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-06T14:37:19.033109Z","title":"Explainableandinterpretablemultimodallargelanguage models: Acomprehensivesurvey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18576","last_updated":"2025-08-07T08:02:42Z","snapshot_observed_at":"2026-08-08T02:00:15.105327Z","submitted_at":"2025-07-24T16:49:19Z","title":"SafeWork-R1: Coevolving Safety and Intelligence under the AI-45$^{\\circ}$ Law","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T14:37:19.033109Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2507.18576"},"observation_digest":"sha256:8f89ab5e5f046328b2683902a7ca6bafa87df99d7c22b642dcbaec43e0c4ad7f","observation_id":"a855dbf3-1936-4255-8ffb-2b16a7239d31","resolution":{"observed_at":"2026-08-06T14:37:19.033109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-04T12:44:41.600573Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.02528","last_updated":"2026-05-29T19:41:43Z","snapshot_observed_at":"2026-08-07T21:53:59.199146Z","submitted_at":"2025-10-02T19:55:56Z","title":"Multimodal Function Vectors for Visual Relations","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-04T12:44:41.600573Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2510.02528"},"observation_digest":"sha256:5f63c6024a7690923f978e10af42f9b2e99b98884757891ed41bb875005fa716","observation_id":"67fde6f4-d426-4769-bb32-61f74d561522","resolution":{"observed_at":"2026-08-04T12:44:41.600573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-02T20:19:13.457923Z","title":"arXiv preprint arXiv:2412.02104 (2024) 1","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.23615","last_updated":"2026-07-08T09:59:15Z","snapshot_observed_at":"2026-08-07T23:59:50.152219Z","submitted_at":"2026-02-27T02:43:35Z","title":"HART: High-Resolution Annotation-Free Reasoning Technique through a Closed-loop Framework","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T20:19:13.457923Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2602.23615"},"observation_digest":"sha256:36ffa0f45f9185b6eb4aa9d678c72e508cd5e8d09cf6e678df36441f8b7ad8b9","observation_id":"bb2e2c30-9dc5-43f9-a2d5-8101406fb5b0","resolution":{"observed_at":"2026-08-02T20:19:13.457923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":"2412.02104","doi":"10.48550/arxiv.2412.02104","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey","venue":"arXiv (Cornell University)","work_id":"ca6d8610-ceff-4c4d-beff-d2f876503d3e","year":2024},"citing_paper":{"arxiv_id":"2602.24176","last_updated":"2026-05-25T13:57:02Z","snapshot_observed_at":"2026-08-02T20:04:31.237597Z","submitted_at":"2026-02-27T16:58:27Z","title":"Beyond Explainable AI (XAI): An Overdue Paradigm Shift and Post-XAI Research Directions","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T18:50:52.313363Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2602.24176"},"observation_digest":"sha256:aab7f63095485399f7866aa35dbefc118065e8b9efe2343df6723741a068a6c3","observation_id":"40dc1e0e-3e8d-458a-ab18-7b9aad361e32","resolution":{"observed_at":"2026-05-15T18:51:30.385816Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-02T20:04:32.769786Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey.arXiv preprint arXiv:2412.02104, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.24176","last_updated":"2026-05-25T13:57:02Z","snapshot_observed_at":"2026-08-02T20:04:31.237597Z","submitted_at":"2026-02-27T16:58:27Z","title":"Beyond Explainable AI (XAI): An Overdue Paradigm Shift and Post-XAI Research Directions","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T20:04:32.769786Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2602.24176"},"observation_digest":"sha256:3f09ffeb8f62490c0794125fc4511bacad4265540357e0481b343b9be6a1d87e","observation_id":"b8fdaef4-9574-46bd-9444-b1ee9cac28f8","resolution":{"observed_at":"2026-08-02T20:04:32.769786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":"2412.02104","doi":"10.48550/arxiv.2412.02104","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey","venue":"arXiv (Cornell University)","work_id":"ca6d8610-ceff-4c4d-beff-d2f876503d3e","year":2024},"citing_paper":{"arxiv_id":"2603.11689","last_updated":"2026-07-01T08:11:59Z","snapshot_observed_at":"2026-07-14T22:43:17.174240Z","submitted_at":"2026-03-12T08:56:14Z","title":"Explicit Logic Channel for Validation and Enhancement of MLLMs on Zero-Shot Tasks","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-21T11:09:51.816554Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2603.11689"},"observation_digest":"sha256:9c70cf8d2dc35ec4a55c62da509a3d8c3b84f12f9bc004e0a2f338accf2f88eb","observation_id":"53edac71-202e-4cd9-b8b8-37c348243d79","resolution":{"observed_at":"2026-05-21T11:10:01.961487Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":"2412.02104","doi":"10.48550/arxiv.2412.02104","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey","venue":"arXiv (Cornell University)","work_id":"ca6d8610-ceff-4c4d-beff-d2f876503d3e","year":2024},"citing_paper":{"arxiv_id":"2604.13565","last_updated":"2026-04-15T07:21:37Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T07:21:37Z","title":"UHR-BAT: Budget-Aware Token Compression Vision-Language model for Ultra-High-Resolution Remote Sensing","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T13:53:13.255412Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2604.13565"},"observation_digest":"sha256:899dc3abbdab65f6c192f49618d16ee5ed27f57045cd2a5760aaa5ca8d40ee29","observation_id":"9ada7a52-d94c-42ab-8348-381645d93efc","resolution":{"observed_at":"2026-05-10T13:55:28.864017Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":"2412.02104","doi":"10.48550/arxiv.2412.02104","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey","venue":"arXiv (Cornell University)","work_id":"ca6d8610-ceff-4c4d-beff-d2f876503d3e","year":2024},"citing_paper":{"arxiv_id":"2604.17941","last_updated":"2026-04-20T08:21:06Z","snapshot_observed_at":"2026-07-06T23:04:55.190916Z","submitted_at":"2026-04-20T08:21:06Z","title":"From Heads to Neurons: Causal Attribution and Steering in Multi-Task Vision-Language Models","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-10T05:44:37.891638Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2604.17941"},"observation_digest":"sha256:998f50e7fa9bd4be37b7843c3cdd656977cfa32a8f10eb4635308d65b7bde49e","observation_id":"f4f1e593-6c83-4209-b8a9-127fd4fdbe67","resolution":{"observed_at":"2026-05-10T05:46:09.455482Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":"2412.02104","doi":"10.48550/arxiv.2412.02104","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey","venue":"arXiv (Cornell University)","work_id":"ca6d8610-ceff-4c4d-beff-d2f876503d3e","year":2024},"citing_paper":{"arxiv_id":"2604.19083","last_updated":"2026-04-21T04:52:38Z","snapshot_observed_at":"2026-07-06T23:05:48.885141Z","submitted_at":"2026-04-21T04:52:38Z","title":"ProjLens: Unveiling the Role of Projectors in Multimodal Model Safety","version":1},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-05-10T03:00:34.862711Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2604.19083"},"observation_digest":"sha256:a786e9efbd2b853d0c546e0eaf8916a4c714b5e9531f0b43857eba9ff61e41eb","observation_id":"85b28b53-d8f1-4a66-8350-cfa4c018e985","resolution":{"observed_at":"2026-05-11T12:46:05.779290Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":"2412.02104","doi":"10.48550/arxiv.2412.02104","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Explainable and interpretable multimodal large language models: A comprehensive survey","venue":"arXiv (Cornell University)","work_id":"ca6d8610-ceff-4c4d-beff-d2f876503d3e","year":2024},"citing_paper":{"arxiv_id":"2606.11446","last_updated":"2026-06-09T20:57:03Z","snapshot_observed_at":"2026-08-02T06:28:13.105999Z","submitted_at":"2026-06-09T20:57:03Z","title":"3D-CBM: A Framework for Concept-Based Interpretability in Generative 3D Modeling","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T13:10:42.454041Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2606.11446"},"observation_digest":"sha256:90c3cf366221796d1e8c5dd7ba7dd73caf894e05cdf7aa7ccfc7f07440b336a2","observation_id":"15f7d97c-7a62-4810-a9c1-7923d25e4cb0","resolution":{"observed_at":"2026-07-03T05:27:40.524932Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-04T01:33:50.503472Z","title":"arXiv preprint arXiv:2412.02104","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00076","last_updated":"2026-08-04T08:25:13Z","snapshot_observed_at":"2026-08-07T23:09:40.857736Z","submitted_at":"2026-07-29T13:33:11Z","title":"Which Modality Decides? Counterfactual Modality Attribution for Multimodal LLMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T01:33:50.503472Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2608.00076"},"observation_digest":"sha256:a5b295115d1ecf204a326e7c90fc7a7fc999e6cd126ae3c0f4cb6634b248306d","observation_id":"4ea7972b-7e10-44d2-ab6a-ca94eda19b02","resolution":{"observed_at":"2026-08-04T01:33:50.503472Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02104","snapshot_observed_at":"2026-08-05T04:27:07.859701Z","title":"arXiv preprint arXiv:2412.02104","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00076","last_updated":"2026-08-04T08:25:13Z","snapshot_observed_at":"2026-08-07T23:09:40.857736Z","submitted_at":"2026-07-29T13:33:11Z","title":"Which Modality Decides? Counterfactual Modality Attribution for Multimodal LLMs","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T04:27:07.859701Z"},"links":{"cited_paper":"/paper/2412.02104","citing_paper":"/paper/2608.00076"},"observation_digest":"sha256:930596bffcd8f9e291fefc7e2dfd87e59077012e0cbcf742f90d609fedae50fa","observation_id":"763be5cb-b83e-42dd-b839-0eec196a34bc","resolution":{"observed_at":"2026-08-05T04:27:07.859701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2412.02104/citation-record","integrity":"/paper/2412.02104/integrity","json":"/paper/2412.02104/citation-record.json","paper":"/paper/2412.02104"},"outbound":[],"paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T20:00:37.331726Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 20 inbound Pith citation observations for arXiv:2412.02104."}