{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2FQ3UCLMZHAA7WWDFMZM4X3CLL","short_pith_number":"pith:2FQ3UCLM","schema_version":"1.0","canonical_sha256":"d161ba096cc9c00fdac32b32ce5f625ac55462c95eafbb8ca1b11c3a18d80e72","source":{"kind":"arxiv","id":"2310.18780","version":1},"attestation_state":"computed","paper":{"title":"Laughing Hyena Distillery: Extracting Compact Recurrences From Convolutions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.SP"],"primary_cat":"cs.LG","authors_text":"Aman Timalsina, Atri Rudra, Beidi Chen, Ce Zhang, Christopher Re, Daniel Y. Fu, David W. Romero, Hermann Kumbong, Michael Poli, Quinn McIntyre, Rom N. Parnichkun, Stefano Ermon, Stefano Massaroli, Yoshua Bengio","submitted_at":"2023-10-28T18:40:03Z","abstract_excerpt":"Recent advances in attention-free sequence models rely on convolutions as alternatives to the attention operator at the core of Transformers. In particular, long convolution sequence models have achieved state-of-the-art performance in many domains, but incur a significant cost during auto-regressive inference workloads -- naively requiring a full pass (or caching of activations) over the input sequence for each generated token -- similarly to attention-based models. In this paper, we seek to enable $\\mathcal O(1)$ compute and memory cost per token in any pre-trained long convolution architect"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.18780","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-28T18:40:03Z","cross_cats_sorted":["cs.AI","eess.SP"],"title_canon_sha256":"f9bdae3ccc4c90f53fbe7b4745487dd96017bf23aacd159c4d8c26ec3ec005d2","abstract_canon_sha256":"6c6f403e2b8cb846dad1c208a4fe92e04a4301e4a796ccf3a26b6a5f19cd5735"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:41.460751Z","signature_b64":"86wrL9qm42g+4RzBFzNS9RSEIgF68aGVD98Wq3cTSfYzQMoNMuGb/HTCbssCN0TPM/fX5KdzM8V2m/xqOBFUAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d161ba096cc9c00fdac32b32ce5f625ac55462c95eafbb8ca1b11c3a18d80e72","last_reissued_at":"2026-07-05T07:06:41.460224Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:41.460224Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Laughing Hyena Distillery: Extracting Compact Recurrences From Convolutions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.SP"],"primary_cat":"cs.LG","authors_text":"Aman Timalsina, Atri Rudra, Beidi Chen, Ce Zhang, Christopher Re, Daniel Y. Fu, David W. Romero, Hermann Kumbong, Michael Poli, Quinn McIntyre, Rom N. Parnichkun, Stefano Ermon, Stefano Massaroli, Yoshua Bengio","submitted_at":"2023-10-28T18:40:03Z","abstract_excerpt":"Recent advances in attention-free sequence models rely on convolutions as alternatives to the attention operator at the core of Transformers. In particular, long convolution sequence models have achieved state-of-the-art performance in many domains, but incur a significant cost during auto-regressive inference workloads -- naively requiring a full pass (or caching of activations) over the input sequence for each generated token -- similarly to attention-based models. In this paper, we seek to enable $\\mathcal O(1)$ compute and memory cost per token in any pre-trained long convolution architect"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.18780","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.18780/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.18780","created_at":"2026-07-05T07:06:41.460284+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.18780v1","created_at":"2026-07-05T07:06:41.460284+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.18780","created_at":"2026-07-05T07:06:41.460284+00:00"},{"alias_kind":"pith_short_12","alias_value":"2FQ3UCLMZHAA","created_at":"2026-07-05T07:06:41.460284+00:00"},{"alias_kind":"pith_short_16","alias_value":"2FQ3UCLMZHAA7WWD","created_at":"2026-07-05T07:06:41.460284+00:00"},{"alias_kind":"pith_short_8","alias_value":"2FQ3UCLM","created_at":"2026-07-05T07:06:41.460284+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2312.06635","citing_title":"Gated Linear Attention Transformers with Hardware-Efficient Training","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2FQ3UCLMZHAA7WWDFMZM4X3CLL","json":"https://pith.science/pith/2FQ3UCLMZHAA7WWDFMZM4X3CLL.json","graph_json":"https://pith.science/api/pith-number/2FQ3UCLMZHAA7WWDFMZM4X3CLL/graph.json","events_json":"https://pith.science/api/pith-number/2FQ3UCLMZHAA7WWDFMZM4X3CLL/events.json","paper":"https://pith.science/paper/2FQ3UCLM"},"agent_actions":{"view_html":"https://pith.science/pith/2FQ3UCLMZHAA7WWDFMZM4X3CLL","download_json":"https://pith.science/pith/2FQ3UCLMZHAA7WWDFMZM4X3CLL.json","view_paper":"https://pith.science/paper/2FQ3UCLM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.18780&json=true","fetch_graph":"https://pith.science/api/pith-number/2FQ3UCLMZHAA7WWDFMZM4X3CLL/graph.json","fetch_events":"https://pith.science/api/pith-number/2FQ3UCLMZHAA7WWDFMZM4X3CLL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2FQ3UCLMZHAA7WWDFMZM4X3CLL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2FQ3UCLMZHAA7WWDFMZM4X3CLL/action/storage_attestation","attest_author":"https://pith.science/pith/2FQ3UCLMZHAA7WWDFMZM4X3CLL/action/author_attestation","sign_citation":"https://pith.science/pith/2FQ3UCLMZHAA7WWDFMZM4X3CLL/action/citation_signature","submit_replication":"https://pith.science/pith/2FQ3UCLMZHAA7WWDFMZM4X3CLL/action/replication_record"}},"created_at":"2026-07-05T07:06:41.460284+00:00","updated_at":"2026-07-05T07:06:41.460284+00:00"}