{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IUMLCBZZDARUSZMLVXUC4PZ6FU","short_pith_number":"pith:IUMLCBZZ","schema_version":"1.0","canonical_sha256":"4518b10739182349658bade82e3f3e2d279ce1a7754445899f3b7b453181ae72","source":{"kind":"arxiv","id":"2310.02980","version":4},"attestation_state":"computed","paper":{"title":"Never Train from Scratch: Fair Comparison of Long-Sequence Models Requires Data-Driven Priors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ankit Gupta, Ido Amos, Jonathan Berant","submitted_at":"2023-10-04T17:17:06Z","abstract_excerpt":"Modeling long-range dependencies across sequences is a longstanding goal in machine learning and has led to architectures, such as state space models, that dramatically outperform Transformers on long sequences. However, these impressive empirical gains have been by and large demonstrated on benchmarks (e.g. Long Range Arena), where models are randomly initialized and trained to predict a target label from an input sequence. In this work, we show that random initialization leads to gross overestimation of the differences between architectures and that pretraining with standard denoising object"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.02980","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-04T17:17:06Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"41f8a2e0b73a814cf5e40ce9fcba02ddbc5ee005449c47267b6cd92e40f569eb","abstract_canon_sha256":"5cb038bab948404cdc1b0fe4ce4c035884d25c5072a82467357fb1652bcb9c4a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:12:45.243416Z","signature_b64":"Ql436xdE+tk0aMLDlAuj26vCamvpyV0cBGqKwWI4cPQ0UAQ47z6voggcZSv7tJHR3zqGvgWDvyIVotkSy5HuDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4518b10739182349658bade82e3f3e2d279ce1a7754445899f3b7b453181ae72","last_reissued_at":"2026-07-05T08:12:45.242923Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:12:45.242923Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Never Train from Scratch: Fair Comparison of Long-Sequence Models Requires Data-Driven Priors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ankit Gupta, Ido Amos, Jonathan Berant","submitted_at":"2023-10-04T17:17:06Z","abstract_excerpt":"Modeling long-range dependencies across sequences is a longstanding goal in machine learning and has led to architectures, such as state space models, that dramatically outperform Transformers on long sequences. However, these impressive empirical gains have been by and large demonstrated on benchmarks (e.g. Long Range Arena), where models are randomly initialized and trained to predict a target label from an input sequence. In this work, we show that random initialization leads to gross overestimation of the differences between architectures and that pretraining with standard denoising object"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.02980","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.02980/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.02980","created_at":"2026-07-05T08:12:45.242982+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.02980v4","created_at":"2026-07-05T08:12:45.242982+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.02980","created_at":"2026-07-05T08:12:45.242982+00:00"},{"alias_kind":"pith_short_12","alias_value":"IUMLCBZZDARU","created_at":"2026-07-05T08:12:45.242982+00:00"},{"alias_kind":"pith_short_16","alias_value":"IUMLCBZZDARUSZML","created_at":"2026-07-05T08:12:45.242982+00:00"},{"alias_kind":"pith_short_8","alias_value":"IUMLCBZZ","created_at":"2026-07-05T08:12:45.242982+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07756","citing_title":"The Importance of Encoder Choice:A Tabular-Image Study","ref_index":156,"is_internal_anchor":true},{"citing_arxiv_id":"2604.00754","citing_title":"Stochastic Attention: Connectome-Inspired Randomized Routing for Expressive Linear-Time Attention","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03688","citing_title":"Fusion and Alignment Enhancement with Large Language Models for Tail-item Sequential Recommendation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08539","citing_title":"Continuity Laws for Sequential Models","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IUMLCBZZDARUSZMLVXUC4PZ6FU","json":"https://pith.science/pith/IUMLCBZZDARUSZMLVXUC4PZ6FU.json","graph_json":"https://pith.science/api/pith-number/IUMLCBZZDARUSZMLVXUC4PZ6FU/graph.json","events_json":"https://pith.science/api/pith-number/IUMLCBZZDARUSZMLVXUC4PZ6FU/events.json","paper":"https://pith.science/paper/IUMLCBZZ"},"agent_actions":{"view_html":"https://pith.science/pith/IUMLCBZZDARUSZMLVXUC4PZ6FU","download_json":"https://pith.science/pith/IUMLCBZZDARUSZMLVXUC4PZ6FU.json","view_paper":"https://pith.science/paper/IUMLCBZZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.02980&json=true","fetch_graph":"https://pith.science/api/pith-number/IUMLCBZZDARUSZMLVXUC4PZ6FU/graph.json","fetch_events":"https://pith.science/api/pith-number/IUMLCBZZDARUSZMLVXUC4PZ6FU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IUMLCBZZDARUSZMLVXUC4PZ6FU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IUMLCBZZDARUSZMLVXUC4PZ6FU/action/storage_attestation","attest_author":"https://pith.science/pith/IUMLCBZZDARUSZMLVXUC4PZ6FU/action/author_attestation","sign_citation":"https://pith.science/pith/IUMLCBZZDARUSZMLVXUC4PZ6FU/action/citation_signature","submit_replication":"https://pith.science/pith/IUMLCBZZDARUSZMLVXUC4PZ6FU/action/replication_record"}},"created_at":"2026-07-05T08:12:45.242982+00:00","updated_at":"2026-07-05T08:12:45.242982+00:00"}