{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:SVE6YRE4FOXSWXTFMY2IJHX555","short_pith_number":"pith:SVE6YRE4","schema_version":"1.0","canonical_sha256":"9549ec449c2baf2b5e656634849efdef4a3f6c70ae61abf2678d6ca080bc418e","source":{"kind":"arxiv","id":"1803.01271","version":2},"attestation_state":"computed","paper":{"title":"An Empirical Evaluation of Generic Convolutional and Recurrent Networks for Sequence Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"A simple convolutional architecture outperforms LSTMs on diverse sequence tasks while showing longer effective memory.","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"J. Zico Kolter, Shaojie Bai, Vladlen Koltun","submitted_at":"2018-03-04T00:20:29Z","abstract_excerpt":"For most deep learning practitioners, sequence modeling is synonymous with recurrent networks. Yet recent results indicate that convolutional architectures can outperform recurrent networks on tasks such as audio synthesis and machine translation. Given a new sequence modeling task or dataset, which architecture should one use? We conduct a systematic evaluation of generic convolutional and recurrent architectures for sequence modeling. The models are evaluated across a broad range of standard tasks that are commonly used to benchmark recurrent networks. Our results indicate that a simple conv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":true},"canonical_record":{"source":{"id":"1803.01271","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-03-04T00:20:29Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"e2a459f1ec375a791fed5430baab2cdfc8d318616e337d8f91f080b3b9eb5371","abstract_canon_sha256":"0a29688397528447abf9132de26d3bfc15cb75eb3f9cf67ad8c0217fbc15b766"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T22:42:36.847004Z","signature_b64":"G0we/8YLUxCwSixQBArY1f50lbMrCyUYoykE8mjVMsy6INSsojIPCvl7Ss8ZZD72lFlJD7n/cppWRq7uVGG5DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9549ec449c2baf2b5e656634849efdef4a3f6c70ae61abf2678d6ca080bc418e","last_reissued_at":"2026-07-04T22:42:36.846473Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T22:42:36.846473Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Evaluation of Generic Convolutional and Recurrent Networks for Sequence Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"A simple convolutional architecture outperforms LSTMs on diverse sequence tasks while showing longer effective memory.","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"J. Zico Kolter, Shaojie Bai, Vladlen Koltun","submitted_at":"2018-03-04T00:20:29Z","abstract_excerpt":"For most deep learning practitioners, sequence modeling is synonymous with recurrent networks. Yet recent results indicate that convolutional architectures can outperform recurrent networks on tasks such as audio synthesis and machine translation. Given a new sequence modeling task or dataset, which architecture should one use? We conduct a systematic evaluation of generic convolutional and recurrent architectures for sequence modeling. The models are evaluated across a broad range of standard tasks that are commonly used to benchmark recurrent networks. Our results indicate that a simple conv"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Our results indicate that a simple convolutional architecture outperforms canonical recurrent networks such as LSTMs across a diverse range of tasks and datasets, while demonstrating longer effective memory.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the chosen tasks and datasets are representative of general sequence modeling challenges and that the generic convolutional and recurrent architectures are implemented and compared fairly without hidden advantages.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"A simple convolutional network outperforms LSTMs across diverse sequence tasks while showing longer effective memory.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"A simple convolutional architecture outperforms LSTMs on diverse sequence tasks while showing longer effective memory.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"a7893b8bc1fc49f86d22a6cf05b90ebac13add52188df11b8a7ef149f4b40cf6"},"source":{"id":"1803.01271","kind":"arxiv","version":2},"verdict":{"id":"0866af40-487f-48c5-ae7b-68eab1b2d011","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-11T19:30:04.145631Z","strongest_claim":"Our results indicate that a simple convolutional architecture outperforms canonical recurrent networks such as LSTMs across a diverse range of tasks and datasets, while demonstrating longer effective memory.","one_line_summary":"A simple convolutional network outperforms LSTMs across diverse sequence tasks while showing longer effective memory.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the chosen tasks and datasets are representative of general sequence modeling challenges and that the generic convolutional and recurrent architectures are implemented and compared fairly without hidden advantages.","pith_extraction_headline":"A simple convolutional architecture outperforms LSTMs on diverse sequence tasks while showing longer effective memory."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1803.01271/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":1,"snapshot_sha256":"bc59c0cce00c67ebdb6fa9de07055e0c5c77d51c7c2633f1f5a6e1977e3b54cf"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1803.01271","created_at":"2026-07-04T22:42:36.846537+00:00"},{"alias_kind":"arxiv_version","alias_value":"1803.01271v2","created_at":"2026-07-04T22:42:36.846537+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1803.01271","created_at":"2026-07-04T22:42:36.846537+00:00"},{"alias_kind":"pith_short_12","alias_value":"SVE6YRE4FOXS","created_at":"2026-07-04T22:42:36.846537+00:00"},{"alias_kind":"pith_short_16","alias_value":"SVE6YRE4FOXSWXTF","created_at":"2026-07-04T22:42:36.846537+00:00"},{"alias_kind":"pith_short_8","alias_value":"SVE6YRE4","created_at":"2026-07-04T22:42:36.846537+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":140,"internal_anchor_count":140,"sample":[{"citing_arxiv_id":"2607.07756","citing_title":"The Importance of Encoder Choice:A Tabular-Image Study","ref_index":264,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08234","citing_title":"RhyMix: A Lightweight Adaptive Multi-Rhythm Network for Long-Term Time Series Forecasting","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08273","citing_title":"Empirical Calibration and Conditional-Reliability Diagnostics for Bearing RUL Prediction under Operating-Regime Shift","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08044","citing_title":"Catching Disguised Transients with ASTRANet: Anomaly-Aware Spectroscopic Classification and Conformal Calibration","ref_index":55,"is_internal_anchor":true},{"citing_arxiv_id":"2607.07574","citing_title":"Context-Aware Force Estimation for Deformable Tool Manipulation in Robotic Environmental Swabbing via Few-Shot Continual Adaptation","ref_index":37,"is_internal_anchor":true},{"citing_arxiv_id":"2604.17616","citing_title":"Conditional Attribution for Root Cause Analysis in Time-Series Anomaly Detection","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25864","citing_title":"Sequential and Generative Models for Vehicular Distributed MIMO Channel Prediction","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24347","citing_title":"MVG-KAN: Multi-View Geo-Wind Guided KAN for PM$_{2.5}$ Forecasting","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22450","citing_title":"Cross-Layer Intrusion Detection in 5G O-RAN: Gains and Limits of Fusing Radio Telemetry with Network Flow Records","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22662","citing_title":"LSTM Variants for Chaotic Dynamical Systems: An Empirical Study on the Lorenz Attractor","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22261","citing_title":"Learning a Normal World Model for Few-Shot Boundary-Calibrated Abnormality Detection","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21018","citing_title":"LK Jam: System Architecture and Implementation of a Real-Time Human-AI Interactive Music Generation System using Role-Aware GRU","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18946","citing_title":"SenFlow: Inter-Sentence Flow Modeling for AI-Generated Text Detection in Hybrid Documents","ref_index":62,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19560","citing_title":"Understanding Key Features of Time Series Foundation Models from Epidemic Forecasting","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19138","citing_title":"INDEQS: Informed Neural controlled Differential EQuationS","ref_index":151,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18539","citing_title":"TS-Fault: Benchmarking Time Series Forecasters Against Structural Faults","ref_index":65,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18049","citing_title":"ConTex: Reformulating Counterfactual Generation For Time Series Forecasting","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01918","citing_title":"Zeus: Towards Tuning-Free Foundation Model for Time Series Analysis","ref_index":93,"is_internal_anchor":true},{"citing_arxiv_id":"2606.13285","citing_title":"Once-for-All: Scalable Simultaneous Forecasting via Equilibrium State Estimation","ref_index":162,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11746","citing_title":"Time Series Analysis in Machine Learning","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2606.10972","citing_title":"Optimizing 2D Input Representations and Sub-phase Fusion Strategies for Differential Diagnosis of Asthma and COPD Using CNN- and GRU-Based Networks","ref_index":31,"is_internal_anchor":true},{"citing_arxiv_id":"2606.10511","citing_title":"Simplified Temporal Convolutional-Based Channel Estimation for a WiFi Vehicular Communication Channel","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.08930","citing_title":"RankGLU: Residual Gated Score Formation for Cross-Sectional Stock Prediction","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09392","citing_title":"From Coarse to Fine: Managing Temporal Granularity in Spatio-Temporal Data for Fine-Grained Traffic Prediction","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2606.10084","citing_title":"Divide-and-Conquer Modeling for the CTF-4-Science Lorenz Benchmark","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":1,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SVE6YRE4FOXSWXTFMY2IJHX555","json":"https://pith.science/pith/SVE6YRE4FOXSWXTFMY2IJHX555.json","graph_json":"https://pith.science/api/pith-number/SVE6YRE4FOXSWXTFMY2IJHX555/graph.json","events_json":"https://pith.science/api/pith-number/SVE6YRE4FOXSWXTFMY2IJHX555/events.json","paper":"https://pith.science/paper/SVE6YRE4"},"agent_actions":{"view_html":"https://pith.science/pith/SVE6YRE4FOXSWXTFMY2IJHX555","download_json":"https://pith.science/pith/SVE6YRE4FOXSWXTFMY2IJHX555.json","view_paper":"https://pith.science/paper/SVE6YRE4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1803.01271&json=true","fetch_graph":"https://pith.science/api/pith-number/SVE6YRE4FOXSWXTFMY2IJHX555/graph.json","fetch_events":"https://pith.science/api/pith-number/SVE6YRE4FOXSWXTFMY2IJHX555/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SVE6YRE4FOXSWXTFMY2IJHX555/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SVE6YRE4FOXSWXTFMY2IJHX555/action/storage_attestation","attest_author":"https://pith.science/pith/SVE6YRE4FOXSWXTFMY2IJHX555/action/author_attestation","sign_citation":"https://pith.science/pith/SVE6YRE4FOXSWXTFMY2IJHX555/action/citation_signature","submit_replication":"https://pith.science/pith/SVE6YRE4FOXSWXTFMY2IJHX555/action/replication_record"}},"created_at":"2026-07-04T22:42:36.846537+00:00","updated_at":"2026-07-04T22:42:36.846537+00:00"}