{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:I3T3OEVFPJFOTQFBJTYRXFE43M","short_pith_number":"pith:I3T3OEVF","schema_version":"1.0","canonical_sha256":"46e7b712a57a4ae9c0a14cf11b949cdb047a7a1b9637b41098a915a30868f5fd","source":{"kind":"arxiv","id":"2411.01030","version":5},"attestation_state":"computed","paper":{"title":"Birdie: Advancing State Space Models with Reward-Driven Objectives and Curricula","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Amarda Shehu, Antonios Anastasopoulos, Jimmy T.H. Smith, Sam Blouir","submitted_at":"2024-11-01T21:01:13Z","abstract_excerpt":"Efficient state space models (SSMs), such as linear recurrent neural networks and linear attention variants, offer computational advantages over Transformers but struggle with tasks requiring long-range in-context retrieval-like text copying, associative recall, and question answering over long contexts. Previous efforts to address these challenges have focused on architectural modifications, often reintroducing computational inefficiencies. In this paper, we propose a novel training procedure, Birdie, that significantly enhances the in-context retrieval capabilities of SSMs without altering t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.01030","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-01T21:01:13Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"376c427206c895f1a43697cd6c6d504e83201e8dad80a409053d04f6836d50ff","abstract_canon_sha256":"d5c1415776b85f6f678f237649f09ec6cfeda5d51d2cbc42894624348d230e91"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:10.803303Z","signature_b64":"n3jhJd1O8tJ3W6D6eBxsHs61FX2L5f5IxM2CbKAnZil842MGm6aT1KyTVYkWA0uV+0zPSMDzEahWwQypahvLCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"46e7b712a57a4ae9c0a14cf11b949cdb047a7a1b9637b41098a915a30868f5fd","last_reissued_at":"2026-07-05T10:18:10.802831Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:10.802831Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Birdie: Advancing State Space Models with Reward-Driven Objectives and Curricula","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Amarda Shehu, Antonios Anastasopoulos, Jimmy T.H. Smith, Sam Blouir","submitted_at":"2024-11-01T21:01:13Z","abstract_excerpt":"Efficient state space models (SSMs), such as linear recurrent neural networks and linear attention variants, offer computational advantages over Transformers but struggle with tasks requiring long-range in-context retrieval-like text copying, associative recall, and question answering over long contexts. Previous efforts to address these challenges have focused on architectural modifications, often reintroducing computational inefficiencies. In this paper, we propose a novel training procedure, Birdie, that significantly enhances the in-context retrieval capabilities of SSMs without altering t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01030","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.01030/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.01030","created_at":"2026-07-05T10:18:10.802887+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.01030v5","created_at":"2026-07-05T10:18:10.802887+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01030","created_at":"2026-07-05T10:18:10.802887+00:00"},{"alias_kind":"pith_short_12","alias_value":"I3T3OEVFPJFO","created_at":"2026-07-05T10:18:10.802887+00:00"},{"alias_kind":"pith_short_16","alias_value":"I3T3OEVFPJFOTQFB","created_at":"2026-07-05T10:18:10.802887+00:00"},{"alias_kind":"pith_short_8","alias_value":"I3T3OEVF","created_at":"2026-07-05T10:18:10.802887+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.18574","citing_title":"Understanding the Skill Gap in Recurrent Language Models: The Role of the Gather-and-Aggregate Mechanism","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I3T3OEVFPJFOTQFBJTYRXFE43M","json":"https://pith.science/pith/I3T3OEVFPJFOTQFBJTYRXFE43M.json","graph_json":"https://pith.science/api/pith-number/I3T3OEVFPJFOTQFBJTYRXFE43M/graph.json","events_json":"https://pith.science/api/pith-number/I3T3OEVFPJFOTQFBJTYRXFE43M/events.json","paper":"https://pith.science/paper/I3T3OEVF"},"agent_actions":{"view_html":"https://pith.science/pith/I3T3OEVFPJFOTQFBJTYRXFE43M","download_json":"https://pith.science/pith/I3T3OEVFPJFOTQFBJTYRXFE43M.json","view_paper":"https://pith.science/paper/I3T3OEVF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.01030&json=true","fetch_graph":"https://pith.science/api/pith-number/I3T3OEVFPJFOTQFBJTYRXFE43M/graph.json","fetch_events":"https://pith.science/api/pith-number/I3T3OEVFPJFOTQFBJTYRXFE43M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I3T3OEVFPJFOTQFBJTYRXFE43M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I3T3OEVFPJFOTQFBJTYRXFE43M/action/storage_attestation","attest_author":"https://pith.science/pith/I3T3OEVFPJFOTQFBJTYRXFE43M/action/author_attestation","sign_citation":"https://pith.science/pith/I3T3OEVFPJFOTQFBJTYRXFE43M/action/citation_signature","submit_replication":"https://pith.science/pith/I3T3OEVFPJFOTQFBJTYRXFE43M/action/replication_record"}},"created_at":"2026-07-05T10:18:10.802887+00:00","updated_at":"2026-07-05T10:18:10.802887+00:00"}