{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:Y7UOTBL6B5WCNFFPOL3ZX6LXZP","short_pith_number":"pith:Y7UOTBL6","schema_version":"1.0","canonical_sha256":"c7e8e9857e0f6c2694af72f79bf977cbe7cb3240ca005a581adb6cc853d7faec","source":{"kind":"arxiv","id":"2607.02593","version":1},"attestation_state":"computed","paper":{"title":"Token-level Response-visual Attention Guidance for Multimodal LLMs Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chang D. Yoo, Eunseop Yoon, Hee Suk Yoon, Jaehyun Jang, Mark A. Hasegawa-Johnson, SooHwan Eom","submitted_at":"2026-07-01T08:03:30Z","abstract_excerpt":"While knowledge distillation (KD) is widely adopted for training lightweight models by leveraging supervision from larger teacher models, relying solely on output token distributions has proven insufficient for compressing Multimodal Large Language Models (MLLMs). Since output tokens are a byproduct of the model attending to visual inputs, prior works have explored explicitly distilling attention to provide a direct supervisory signal. While promising, the precise utility of which attention signals to distill remains under-explored. In this work, we challenge the conventional reliance on promp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.02593","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-07-01T08:03:30Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"0bd3b588626b3467a52ed57348a250425d1827396f5f7ea74949a64ba3a86528","abstract_canon_sha256":"0d9c13ce09105ac50036364bacb9bd41f1f0899292affcf2699fa005e02f4c63"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T00:16:02.146136Z","signature_b64":"8wkp2MuZGigdjKj6nlNnAKmQhkgII1z+T287W8V2sDBg0UCv1De/RRUhv6NR1immnfd4NuuJa5MkZCuR+3G4AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c7e8e9857e0f6c2694af72f79bf977cbe7cb3240ca005a581adb6cc853d7faec","last_reissued_at":"2026-07-07T00:16:02.145398Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T00:16:02.145398Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Token-level Response-visual Attention Guidance for Multimodal LLMs Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chang D. Yoo, Eunseop Yoon, Hee Suk Yoon, Jaehyun Jang, Mark A. Hasegawa-Johnson, SooHwan Eom","submitted_at":"2026-07-01T08:03:30Z","abstract_excerpt":"While knowledge distillation (KD) is widely adopted for training lightweight models by leveraging supervision from larger teacher models, relying solely on output token distributions has proven insufficient for compressing Multimodal Large Language Models (MLLMs). Since output tokens are a byproduct of the model attending to visual inputs, prior works have explored explicitly distilling attention to provide a direct supervisory signal. While promising, the precise utility of which attention signals to distill remains under-explored. In this work, we challenge the conventional reliance on promp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.02593","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.02593/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.02593","created_at":"2026-07-07T00:16:02.145518+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.02593v1","created_at":"2026-07-07T00:16:02.145518+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.02593","created_at":"2026-07-07T00:16:02.145518+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y7UOTBL6B5WC","created_at":"2026-07-07T00:16:02.145518+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y7UOTBL6B5WCNFFP","created_at":"2026-07-07T00:16:02.145518+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y7UOTBL6","created_at":"2026-07-07T00:16:02.145518+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y7UOTBL6B5WCNFFPOL3ZX6LXZP","json":"https://pith.science/pith/Y7UOTBL6B5WCNFFPOL3ZX6LXZP.json","graph_json":"https://pith.science/api/pith-number/Y7UOTBL6B5WCNFFPOL3ZX6LXZP/graph.json","events_json":"https://pith.science/api/pith-number/Y7UOTBL6B5WCNFFPOL3ZX6LXZP/events.json","paper":"https://pith.science/paper/Y7UOTBL6"},"agent_actions":{"view_html":"https://pith.science/pith/Y7UOTBL6B5WCNFFPOL3ZX6LXZP","download_json":"https://pith.science/pith/Y7UOTBL6B5WCNFFPOL3ZX6LXZP.json","view_paper":"https://pith.science/paper/Y7UOTBL6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.02593&json=true","fetch_graph":"https://pith.science/api/pith-number/Y7UOTBL6B5WCNFFPOL3ZX6LXZP/graph.json","fetch_events":"https://pith.science/api/pith-number/Y7UOTBL6B5WCNFFPOL3ZX6LXZP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y7UOTBL6B5WCNFFPOL3ZX6LXZP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y7UOTBL6B5WCNFFPOL3ZX6LXZP/action/storage_attestation","attest_author":"https://pith.science/pith/Y7UOTBL6B5WCNFFPOL3ZX6LXZP/action/author_attestation","sign_citation":"https://pith.science/pith/Y7UOTBL6B5WCNFFPOL3ZX6LXZP/action/citation_signature","submit_replication":"https://pith.science/pith/Y7UOTBL6B5WCNFFPOL3ZX6LXZP/action/replication_record"}},"created_at":"2026-07-07T00:16:02.145518+00:00","updated_at":"2026-07-07T00:16:02.145518+00:00"}