{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QCOBOMJMVJ7TQ2ZB2R46P5ZAD6","short_pith_number":"pith:QCOBOMJM","schema_version":"1.0","canonical_sha256":"809c17312caa7f386b21d479e7f7201fa923eb80a37dbdb4973142c68232b87c","source":{"kind":"arxiv","id":"2305.15805","version":3},"attestation_state":"computed","paper":{"title":"Dynamic Context Pruning for Efficient and Interpretable Autoregressive Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aurelien Lucchi, Dario Pavllo, Lorenzo Noci, Luca Biggio, Sotiris Anagnostidis, Thomas Hofmann","submitted_at":"2023-05-25T07:39:41Z","abstract_excerpt":"Autoregressive Transformers adopted in Large Language Models (LLMs) are hard to scale to long sequences. Despite several works trying to reduce their computational cost, most of LLMs still adopt attention layers between all pairs of tokens in the sequence, thus incurring a quadratic cost. In this study, we present a novel approach that dynamically prunes contextual information while preserving the model's expressiveness, resulting in reduced memory and computational requirements during inference. Our method employs a learnable mechanism that determines which uninformative tokens can be dropped"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.15805","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-25T07:39:41Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"06aa1c007bea4d090e91de9e3c4c27104c42f62b57bf8de1cb46aedaaf53f341","abstract_canon_sha256":"606ee6539b5bd7d932ba32bd36d4ea5002e0b11de95f467962a98db9780f380f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:27.662276Z","signature_b64":"oZzQvJ7zSmlEeujgwEuTP7Q6R+qGDueqtBW9uDMpgW3tiPhkFCu12ObiYmIZfoFnhkRrUDM/OevX/+2uJi75Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"809c17312caa7f386b21d479e7f7201fa923eb80a37dbdb4973142c68232b87c","last_reissued_at":"2026-07-05T08:25:27.661786Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:27.661786Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dynamic Context Pruning for Efficient and Interpretable Autoregressive Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aurelien Lucchi, Dario Pavllo, Lorenzo Noci, Luca Biggio, Sotiris Anagnostidis, Thomas Hofmann","submitted_at":"2023-05-25T07:39:41Z","abstract_excerpt":"Autoregressive Transformers adopted in Large Language Models (LLMs) are hard to scale to long sequences. Despite several works trying to reduce their computational cost, most of LLMs still adopt attention layers between all pairs of tokens in the sequence, thus incurring a quadratic cost. In this study, we present a novel approach that dynamically prunes contextual information while preserving the model's expressiveness, resulting in reduced memory and computational requirements during inference. Our method employs a learnable mechanism that determines which uninformative tokens can be dropped"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.15805","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.15805/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.15805","created_at":"2026-07-05T08:25:27.661846+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.15805v3","created_at":"2026-07-05T08:25:27.661846+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.15805","created_at":"2026-07-05T08:25:27.661846+00:00"},{"alias_kind":"pith_short_12","alias_value":"QCOBOMJMVJ7T","created_at":"2026-07-05T08:25:27.661846+00:00"},{"alias_kind":"pith_short_16","alias_value":"QCOBOMJMVJ7TQ2ZB","created_at":"2026-07-05T08:25:27.661846+00:00"},{"alias_kind":"pith_short_8","alias_value":"QCOBOMJM","created_at":"2026-07-05T08:25:27.661846+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2306.14048","citing_title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05459","citing_title":"Personal LLM Agents: Insights and Survey about the Capability, Efficiency and Security","ref_index":265,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":158,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6","json":"https://pith.science/pith/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6.json","graph_json":"https://pith.science/api/pith-number/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6/graph.json","events_json":"https://pith.science/api/pith-number/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6/events.json","paper":"https://pith.science/paper/QCOBOMJM"},"agent_actions":{"view_html":"https://pith.science/pith/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6","download_json":"https://pith.science/pith/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6.json","view_paper":"https://pith.science/paper/QCOBOMJM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.15805&json=true","fetch_graph":"https://pith.science/api/pith-number/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6/graph.json","fetch_events":"https://pith.science/api/pith-number/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6/action/storage_attestation","attest_author":"https://pith.science/pith/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6/action/author_attestation","sign_citation":"https://pith.science/pith/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6/action/citation_signature","submit_replication":"https://pith.science/pith/QCOBOMJMVJ7TQ2ZB2R46P5ZAD6/action/replication_record"}},"created_at":"2026-07-05T08:25:27.661846+00:00","updated_at":"2026-07-05T08:25:27.661846+00:00"}