{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:H7HTAYD7N63QBI6KIVTULRXQSG","short_pith_number":"pith:H7HTAYD7","schema_version":"1.0","canonical_sha256":"3fcf30607f6fb700a3ca456745c6f091b0c8f89df1515f1f7296322a198ca6b9","source":{"kind":"arxiv","id":"2504.04308","version":1},"attestation_state":"computed","paper":{"title":"Gating is Weighting: Understanding Gated Linear Attention through In-context Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","math.OC"],"primary_cat":"cs.LG","authors_text":"Ankit Singh Rawat, Davoud Ataee Tarzanagh, Maryam Fazel, Samet Oymak, Yingcong Li","submitted_at":"2025-04-06T00:37:36Z","abstract_excerpt":"Linear attention methods offer a compelling alternative to softmax attention due to their efficiency in recurrent decoding. Recent research has focused on enhancing standard linear attention by incorporating gating while retaining its computational benefits. Such Gated Linear Attention (GLA) architectures include competitive models such as Mamba and RWKV. In this work, we investigate the in-context learning capabilities of the GLA model and make the following contributions. We show that a multilayer GLA can implement a general class of Weighted Preconditioned Gradient Descent (WPGD) algorithms"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.04308","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-06T00:37:36Z","cross_cats_sorted":["cs.AI","cs.CL","math.OC"],"title_canon_sha256":"7e5741099e665cfb9d1650bc4355bc239e583e7723d5006b426960c4a21a117a","abstract_canon_sha256":"c9d47525c7f837c3c8da6b5aaa8b0651da831722916eb41ee62451ec0188ab04"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:22.183098Z","signature_b64":"+yZaYmvxa8pwY8cWGaXoLSb3AvC7FZHlpxLGmKKnnObZyuJVRpY/CRVwkf5wO96YBy7VoiUAshZj/PZgNv2vAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3fcf30607f6fb700a3ca456745c6f091b0c8f89df1515f1f7296322a198ca6b9","last_reissued_at":"2026-07-05T10:45:22.182634Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:22.182634Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Gating is Weighting: Understanding Gated Linear Attention through In-context Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","math.OC"],"primary_cat":"cs.LG","authors_text":"Ankit Singh Rawat, Davoud Ataee Tarzanagh, Maryam Fazel, Samet Oymak, Yingcong Li","submitted_at":"2025-04-06T00:37:36Z","abstract_excerpt":"Linear attention methods offer a compelling alternative to softmax attention due to their efficiency in recurrent decoding. Recent research has focused on enhancing standard linear attention by incorporating gating while retaining its computational benefits. Such Gated Linear Attention (GLA) architectures include competitive models such as Mamba and RWKV. In this work, we investigate the in-context learning capabilities of the GLA model and make the following contributions. We show that a multilayer GLA can implement a general class of Weighted Preconditioned Gradient Descent (WPGD) algorithms"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.04308","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.04308/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.04308","created_at":"2026-07-05T10:45:22.182690+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.04308v1","created_at":"2026-07-05T10:45:22.182690+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.04308","created_at":"2026-07-05T10:45:22.182690+00:00"},{"alias_kind":"pith_short_12","alias_value":"H7HTAYD7N63Q","created_at":"2026-07-05T10:45:22.182690+00:00"},{"alias_kind":"pith_short_16","alias_value":"H7HTAYD7N63QBI6K","created_at":"2026-07-05T10:45:22.182690+00:00"},{"alias_kind":"pith_short_8","alias_value":"H7HTAYD7","created_at":"2026-07-05T10:45:22.182690+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.10946","citing_title":"Learning to Adapt: In-Context Learning Beyond Stationarity","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H7HTAYD7N63QBI6KIVTULRXQSG","json":"https://pith.science/pith/H7HTAYD7N63QBI6KIVTULRXQSG.json","graph_json":"https://pith.science/api/pith-number/H7HTAYD7N63QBI6KIVTULRXQSG/graph.json","events_json":"https://pith.science/api/pith-number/H7HTAYD7N63QBI6KIVTULRXQSG/events.json","paper":"https://pith.science/paper/H7HTAYD7"},"agent_actions":{"view_html":"https://pith.science/pith/H7HTAYD7N63QBI6KIVTULRXQSG","download_json":"https://pith.science/pith/H7HTAYD7N63QBI6KIVTULRXQSG.json","view_paper":"https://pith.science/paper/H7HTAYD7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.04308&json=true","fetch_graph":"https://pith.science/api/pith-number/H7HTAYD7N63QBI6KIVTULRXQSG/graph.json","fetch_events":"https://pith.science/api/pith-number/H7HTAYD7N63QBI6KIVTULRXQSG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H7HTAYD7N63QBI6KIVTULRXQSG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H7HTAYD7N63QBI6KIVTULRXQSG/action/storage_attestation","attest_author":"https://pith.science/pith/H7HTAYD7N63QBI6KIVTULRXQSG/action/author_attestation","sign_citation":"https://pith.science/pith/H7HTAYD7N63QBI6KIVTULRXQSG/action/citation_signature","submit_replication":"https://pith.science/pith/H7HTAYD7N63QBI6KIVTULRXQSG/action/replication_record"}},"created_at":"2026-07-05T10:45:22.182690+00:00","updated_at":"2026-07-05T10:45:22.182690+00:00"}