{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:66CPJOCBILCOOG2F4VB5JIIL7O","short_pith_number":"pith:66CPJOCB","schema_version":"1.0","canonical_sha256":"f784f4b84142c4e71b45e543d4a10bfb815e8841f047c87aeb6b12e96ebcef8d","source":{"kind":"arxiv","id":"2107.05407","version":2},"attestation_state":"computed","paper":{"title":"PonderNet: Learning to Ponder","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CC"],"primary_cat":"cs.LG","authors_text":"Andrea Banino, Charles Blundell, Jan Balaguer","submitted_at":"2021-07-12T13:24:03Z","abstract_excerpt":"In standard neural networks the amount of computation used grows with the size of the inputs, but not with the complexity of the problem being learnt. To overcome this limitation we introduce PonderNet, a new algorithm that learns to adapt the amount of computation based on the complexity of the problem at hand. PonderNet learns end-to-end the number of computational steps to achieve an effective compromise between training prediction accuracy, computational cost and generalization. On a complex synthetic problem, PonderNet dramatically improves performance over previous adaptive computation m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.05407","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-07-12T13:24:03Z","cross_cats_sorted":["cs.AI","cs.CC"],"title_canon_sha256":"efa076c20e665e9b4f66c430fba148bb4a2cfb5d69e6de085eba33e18974018f","abstract_canon_sha256":"57a0ef7ef8cfca77a2495cb83fe02897521eeb6cf980373287568f0d01dd9048"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:10:54.618133Z","signature_b64":"3q8ByPGydsPKhBob6qWQTDwopHC/LBkihmbksfbmbbxyDlcvGfUkyjea8BpYYbhrhIhzRRIzXgfayFgxi7HiDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f784f4b84142c4e71b45e543d4a10bfb815e8841f047c87aeb6b12e96ebcef8d","last_reissued_at":"2026-07-05T03:10:54.617592Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:10:54.617592Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PonderNet: Learning to Ponder","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CC"],"primary_cat":"cs.LG","authors_text":"Andrea Banino, Charles Blundell, Jan Balaguer","submitted_at":"2021-07-12T13:24:03Z","abstract_excerpt":"In standard neural networks the amount of computation used grows with the size of the inputs, but not with the complexity of the problem being learnt. To overcome this limitation we introduce PonderNet, a new algorithm that learns to adapt the amount of computation based on the complexity of the problem at hand. PonderNet learns end-to-end the number of computational steps to achieve an effective compromise between training prediction accuracy, computational cost and generalization. On a complex synthetic problem, PonderNet dramatically improves performance over previous adaptive computation m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.05407","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.05407/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.05407","created_at":"2026-07-05T03:10:54.617654+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.05407v2","created_at":"2026-07-05T03:10:54.617654+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.05407","created_at":"2026-07-05T03:10:54.617654+00:00"},{"alias_kind":"pith_short_12","alias_value":"66CPJOCBILCO","created_at":"2026-07-05T03:10:54.617654+00:00"},{"alias_kind":"pith_short_16","alias_value":"66CPJOCBILCOOG2F","created_at":"2026-07-05T03:10:54.617654+00:00"},{"alias_kind":"pith_short_8","alias_value":"66CPJOCB","created_at":"2026-07-05T03:10:54.617654+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26463","citing_title":"Finding the Time to Think: Learning Planning Budgets in Real-Time RL","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18206","citing_title":"Fixed-Point Reasoners: Stable and Adaptive Deep Looped Transformers","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24898","citing_title":"Dense Supervision Is Not Enough: The Readout Blind Spot in Looped Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31859","citing_title":"Review Residuals: Update-Conditioned Residual Gating for Transformers","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26463","citing_title":"Finding the Time to Think: Learning Planning Budgets in Real-Time RL","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29983","citing_title":"Stabilizing Extrapolation in Looped Transformers via Learned Stochastic Stopping","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27696","citing_title":"Structure over Pixels: Learning Variable-Length Visual Programs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28919","citing_title":"CosmicFish-HRM: Adaptive Reasoning via Hierarchical Recurrent Mechanisms in Compact Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30202","citing_title":"A Dual-Path Architecture for Scaling Compute and Capacity in LLMs","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23872","citing_title":"Training-Free Looped Transformers","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13215","citing_title":"When to Think Fast and Slow? AMOR: Adaptive Entropy Gate for Hybrid Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2510.25741","citing_title":"Scaling Latent Reasoning via Looped Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21734","citing_title":"Hierarchical Reasoning Model","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2211.09085","citing_title":"Galactica: A Large Language Model for Science","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25299","citing_title":"The Thinking Pixel: Recursive Sparse Reasoning in Multimodal Diffusion Latents","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22110","citing_title":"Do Not Imitate, Reinforce: Iterative Classification via Belief Refinement","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11791","citing_title":"A Mechanistic Analysis of Looped Reasoning Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15259","citing_title":"Stability and Generalization in Looped Transformers","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21999","citing_title":"Universal Transformers Need Memory: Depth-State Trade-offs in Adaptive Recursive Reasoning","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/66CPJOCBILCOOG2F4VB5JIIL7O","json":"https://pith.science/pith/66CPJOCBILCOOG2F4VB5JIIL7O.json","graph_json":"https://pith.science/api/pith-number/66CPJOCBILCOOG2F4VB5JIIL7O/graph.json","events_json":"https://pith.science/api/pith-number/66CPJOCBILCOOG2F4VB5JIIL7O/events.json","paper":"https://pith.science/paper/66CPJOCB"},"agent_actions":{"view_html":"https://pith.science/pith/66CPJOCBILCOOG2F4VB5JIIL7O","download_json":"https://pith.science/pith/66CPJOCBILCOOG2F4VB5JIIL7O.json","view_paper":"https://pith.science/paper/66CPJOCB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.05407&json=true","fetch_graph":"https://pith.science/api/pith-number/66CPJOCBILCOOG2F4VB5JIIL7O/graph.json","fetch_events":"https://pith.science/api/pith-number/66CPJOCBILCOOG2F4VB5JIIL7O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/66CPJOCBILCOOG2F4VB5JIIL7O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/66CPJOCBILCOOG2F4VB5JIIL7O/action/storage_attestation","attest_author":"https://pith.science/pith/66CPJOCBILCOOG2F4VB5JIIL7O/action/author_attestation","sign_citation":"https://pith.science/pith/66CPJOCBILCOOG2F4VB5JIIL7O/action/citation_signature","submit_replication":"https://pith.science/pith/66CPJOCBILCOOG2F4VB5JIIL7O/action/replication_record"}},"created_at":"2026-07-05T03:10:54.617654+00:00","updated_at":"2026-07-05T03:10:54.617654+00:00"}