{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2016:7X3X64MDHLOUBSGKHO43PWRLAE","short_pith_number":"pith:7X3X64MD","schema_version":"1.0","canonical_sha256":"fdf77f71833add40c8ca3bb9b7da2b010abbbb1aeedaad7f8bcd271c93baaaeb","source":{"kind":"arxiv","id":"1612.07771","version":3},"attestation_state":"computed","paper":{"title":"Highway and Residual Networks learn Unrolled Iterative Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.NE","authors_text":"J\\\"urgen Schmidhuber, Klaus Greff, Rupesh K. Srivastava","submitted_at":"2016-12-22T19:57:35Z","abstract_excerpt":"The past year saw the introduction of new architectures such as Highway networks and Residual networks which, for the first time, enabled the training of feedforward networks with dozens to hundreds of layers using simple gradient descent. While depth of representation has been posited as a primary reason for their success, there are indications that these architectures defy a popular view of deep learning as a hierarchical computation of increasingly abstract features at each layer.\n  In this report, we argue that this view is incomplete and does not adequately explain several recent findings"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1612.07771","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.NE","submitted_at":"2016-12-22T19:57:35Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9d037c9e8138f19bc5ae0391aa762b67bcc1f7aa1e894d85ead66b5ac69c61e8","abstract_canon_sha256":"aa24f65834ee6993e2c9f4a9d6434d54ac41f6a39d62c096d41aedb0c101afe0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:48:41.927200Z","signature_b64":"7ooS/oFab2AgJx6Ky/7CEXGeY4itw4H9sq1/CEAECBGn9vjMA3r4/THtFFzTDoS7v62vTthz40buCghF1xr0Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fdf77f71833add40c8ca3bb9b7da2b010abbbb1aeedaad7f8bcd271c93baaaeb","last_reissued_at":"2026-05-18T00:48:41.926271Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:48:41.926271Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Highway and Residual Networks learn Unrolled Iterative Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.NE","authors_text":"J\\\"urgen Schmidhuber, Klaus Greff, Rupesh K. Srivastava","submitted_at":"2016-12-22T19:57:35Z","abstract_excerpt":"The past year saw the introduction of new architectures such as Highway networks and Residual networks which, for the first time, enabled the training of feedforward networks with dozens to hundreds of layers using simple gradient descent. While depth of representation has been posited as a primary reason for their success, there are indications that these architectures defy a popular view of deep learning as a hierarchical computation of increasingly abstract features at each layer.\n  In this report, we argue that this view is incomplete and does not adequately explain several recent findings"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1612.07771","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1612.07771","created_at":"2026-05-18T00:48:41.926585+00:00"},{"alias_kind":"arxiv_version","alias_value":"1612.07771v3","created_at":"2026-05-18T00:48:41.926585+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1612.07771","created_at":"2026-05-18T00:48:41.926585+00:00"},{"alias_kind":"pith_short_12","alias_value":"7X3X64MDHLOU","created_at":"2026-05-18T12:30:04.600751+00:00"},{"alias_kind":"pith_short_16","alias_value":"7X3X64MDHLOUBSGK","created_at":"2026-05-18T12:30:04.600751+00:00"},{"alias_kind":"pith_short_8","alias_value":"7X3X64MD","created_at":"2026-05-18T12:30:04.600751+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"1906.12284","citing_title":"Widening the Representation Bottleneck in Neural Machine Translation with Lexical Shortcuts","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"1804.03999","citing_title":"Attention U-Net: Learning Where to Look for the Pancreas","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2303.08112","citing_title":"Eliciting Latent Predictions from Transformers with the Tuned Lens","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7X3X64MDHLOUBSGKHO43PWRLAE","json":"https://pith.science/pith/7X3X64MDHLOUBSGKHO43PWRLAE.json","graph_json":"https://pith.science/api/pith-number/7X3X64MDHLOUBSGKHO43PWRLAE/graph.json","events_json":"https://pith.science/api/pith-number/7X3X64MDHLOUBSGKHO43PWRLAE/events.json","paper":"https://pith.science/paper/7X3X64MD"},"agent_actions":{"view_html":"https://pith.science/pith/7X3X64MDHLOUBSGKHO43PWRLAE","download_json":"https://pith.science/pith/7X3X64MDHLOUBSGKHO43PWRLAE.json","view_paper":"https://pith.science/paper/7X3X64MD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1612.07771&json=true","fetch_graph":"https://pith.science/api/pith-number/7X3X64MDHLOUBSGKHO43PWRLAE/graph.json","fetch_events":"https://pith.science/api/pith-number/7X3X64MDHLOUBSGKHO43PWRLAE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7X3X64MDHLOUBSGKHO43PWRLAE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7X3X64MDHLOUBSGKHO43PWRLAE/action/storage_attestation","attest_author":"https://pith.science/pith/7X3X64MDHLOUBSGKHO43PWRLAE/action/author_attestation","sign_citation":"https://pith.science/pith/7X3X64MDHLOUBSGKHO43PWRLAE/action/citation_signature","submit_replication":"https://pith.science/pith/7X3X64MDHLOUBSGKHO43PWRLAE/action/replication_record"}},"created_at":"2026-05-18T00:48:41.926585+00:00","updated_at":"2026-05-18T00:48:41.926585+00:00"}