{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Q6LIHGCBFY7L3VVAAXBE7NGACT","short_pith_number":"pith:Q6LIHGCB","schema_version":"1.0","canonical_sha256":"87968398412e3ebdd6a005c24fb4c014fa3f5166bf40f09a08f7deb03b82895a","source":{"kind":"arxiv","id":"2305.14882","version":2},"attestation_state":"computed","paper":{"title":"Dynamic Clue Bottlenecks: Towards Interpretable-by-Design Visual Question Answering","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Ben Zhou, Dan Roth, Mark Yatskar, Sihao Chen, Xingyu Fu","submitted_at":"2023-05-24T08:33:15Z","abstract_excerpt":"Recent advances in multimodal large language models (LLMs) have shown extreme effectiveness in visual question answering (VQA). However, the design nature of these end-to-end models prevents them from being interpretable to humans, undermining trust and applicability in critical domains. While post-hoc rationales offer certain insight into understanding model behavior, these explanations are not guaranteed to be faithful to the model. In this paper, we address these shortcomings by introducing an interpretable by design model that factors model decisions into intermediate human-legible explana"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14882","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-24T08:33:15Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"5caa02c88888d9fc64788e14fc365e156e4d005aea1fb15ed89dd4d5d066aa1d","abstract_canon_sha256":"d7cef97f839ff33cacefe205ca26fad7b59ea7748a98e43d937146e1d32261f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:07:27.621809Z","signature_b64":"BD1orXM6NFj533kxotdOx/h9GwTi25iDtyQuIJmx0/vf+MpaazvthLjApuFf0AhBP2HmPZKQW73pgHQ86KyCBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"87968398412e3ebdd6a005c24fb4c014fa3f5166bf40f09a08f7deb03b82895a","last_reissued_at":"2026-07-05T08:07:27.621260Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:07:27.621260Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dynamic Clue Bottlenecks: Towards Interpretable-by-Design Visual Question Answering","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Ben Zhou, Dan Roth, Mark Yatskar, Sihao Chen, Xingyu Fu","submitted_at":"2023-05-24T08:33:15Z","abstract_excerpt":"Recent advances in multimodal large language models (LLMs) have shown extreme effectiveness in visual question answering (VQA). However, the design nature of these end-to-end models prevents them from being interpretable to humans, undermining trust and applicability in critical domains. While post-hoc rationales offer certain insight into understanding model behavior, these explanations are not guaranteed to be faithful to the model. In this paper, we address these shortcomings by introducing an interpretable by design model that factors model decisions into intermediate human-legible explana"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14882","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14882/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14882","created_at":"2026-07-05T08:07:27.621337+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14882v2","created_at":"2026-07-05T08:07:27.621337+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14882","created_at":"2026-07-05T08:07:27.621337+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q6LIHGCBFY7L","created_at":"2026-07-05T08:07:27.621337+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q6LIHGCBFY7L3VVA","created_at":"2026-07-05T08:07:27.621337+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q6LIHGCB","created_at":"2026-07-05T08:07:27.621337+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2404.12390","citing_title":"BLINK: Multimodal Large Language Models Can See but Not Perceive","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q6LIHGCBFY7L3VVAAXBE7NGACT","json":"https://pith.science/pith/Q6LIHGCBFY7L3VVAAXBE7NGACT.json","graph_json":"https://pith.science/api/pith-number/Q6LIHGCBFY7L3VVAAXBE7NGACT/graph.json","events_json":"https://pith.science/api/pith-number/Q6LIHGCBFY7L3VVAAXBE7NGACT/events.json","paper":"https://pith.science/paper/Q6LIHGCB"},"agent_actions":{"view_html":"https://pith.science/pith/Q6LIHGCBFY7L3VVAAXBE7NGACT","download_json":"https://pith.science/pith/Q6LIHGCBFY7L3VVAAXBE7NGACT.json","view_paper":"https://pith.science/paper/Q6LIHGCB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14882&json=true","fetch_graph":"https://pith.science/api/pith-number/Q6LIHGCBFY7L3VVAAXBE7NGACT/graph.json","fetch_events":"https://pith.science/api/pith-number/Q6LIHGCBFY7L3VVAAXBE7NGACT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q6LIHGCBFY7L3VVAAXBE7NGACT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q6LIHGCBFY7L3VVAAXBE7NGACT/action/storage_attestation","attest_author":"https://pith.science/pith/Q6LIHGCBFY7L3VVAAXBE7NGACT/action/author_attestation","sign_citation":"https://pith.science/pith/Q6LIHGCBFY7L3VVAAXBE7NGACT/action/citation_signature","submit_replication":"https://pith.science/pith/Q6LIHGCBFY7L3VVAAXBE7NGACT/action/replication_record"}},"created_at":"2026-07-05T08:07:27.621337+00:00","updated_at":"2026-07-05T08:07:27.621337+00:00"}