{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:Y5POH3PWYHZDIQUBU322ZF56RX","short_pith_number":"pith:Y5POH3PW","schema_version":"1.0","canonical_sha256":"c75ee3edf6c1f2344281a6f5ac97be8dfef244f3beae6961b037c2f487b496db","source":{"kind":"arxiv","id":"2108.13264","version":4},"attestation_state":"computed","paper":{"title":"Deep Reinforcement Learning at the Edge of the Statistical Precipice","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ME","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Marc G. Bellemare, Max Schwarzer, Pablo Samuel Castro, Rishabh Agarwal","submitted_at":"2021-08-30T14:23:48Z","abstract_excerpt":"Deep reinforcement learning (RL) algorithms are predominantly evaluated by comparing their relative performance on a large suite of tasks. Most published results on deep RL benchmarks compare point estimates of aggregate performance such as mean and median scores across tasks, ignoring the statistical uncertainty implied by the use of a finite number of training runs. Beginning with the Arcade Learning Environment (ALE), the shift towards computationally-demanding benchmarks has led to the practice of evaluating only a small number of runs per task, exacerbating the statistical uncertainty in "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.13264","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-08-30T14:23:48Z","cross_cats_sorted":["cs.AI","stat.ME","stat.ML"],"title_canon_sha256":"6c829853fdb66010aa5af3f0979e859c178ef5ed6a9f2d7ad54348463f3e830e","abstract_canon_sha256":"4782fa0b4ff7d664b122209fee880c4fbf362ff4512eaa11b296853d77177c49"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:46:13.827012Z","signature_b64":"UDtK+duKTvJk746gq/EfTOAn0al7wVzArmRkLbmoJ07pL/9vrvQAPBRZrTj/C4AbAPvFNgtcpVoCmITzDl68Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c75ee3edf6c1f2344281a6f5ac97be8dfef244f3beae6961b037c2f487b496db","last_reissued_at":"2026-07-05T03:46:13.826451Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:46:13.826451Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Reinforcement Learning at the Edge of the Statistical Precipice","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ME","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Marc G. Bellemare, Max Schwarzer, Pablo Samuel Castro, Rishabh Agarwal","submitted_at":"2021-08-30T14:23:48Z","abstract_excerpt":"Deep reinforcement learning (RL) algorithms are predominantly evaluated by comparing their relative performance on a large suite of tasks. Most published results on deep RL benchmarks compare point estimates of aggregate performance such as mean and median scores across tasks, ignoring the statistical uncertainty implied by the use of a finite number of training runs. Beginning with the Arcade Learning Environment (ALE), the shift towards computationally-demanding benchmarks has led to the practice of evaluating only a small number of runs per task, exacerbating the statistical uncertainty in "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.13264","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.13264/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.13264","created_at":"2026-07-05T03:46:13.826512+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.13264v4","created_at":"2026-07-05T03:46:13.826512+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.13264","created_at":"2026-07-05T03:46:13.826512+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y5POH3PWYHZD","created_at":"2026-07-05T03:46:13.826512+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y5POH3PWYHZDIQUB","created_at":"2026-07-05T03:46:13.826512+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y5POH3PW","created_at":"2026-07-05T03:46:13.826512+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07460","citing_title":"Social-spatial dependencies for learning visual navigation","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20394","citing_title":"Agentic AutoResearch forSpace Autonomy: An Auditable, LLM-Driven Research Agent for Aerospace Control Problems","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19328","citing_title":"UBP2: Uncertainty-Balanced Preference Planning for Efficient Preference-based Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02292","citing_title":"One More Time: Revisiting Neural Quantum States from a Reinforcement Learning Perspective","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06603","citing_title":"The hidden risks of temporal resampling in clinical reinforcement learning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05812","citing_title":"Long-Horizon Q-Learning: Accurate Value Learning via n-Step Inequalities","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05812","citing_title":"Long-Horizon Q-Learning: Accurate Value Learning via n-Step Inequalities","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00428","citing_title":"How to Do Statistical Evaluations in ECE/CS Papers: A Practical Playbook for Defensible Results","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09028","citing_title":"Plasticity-Enhanced Multi-Agent Mixture of Experts for Dynamic Objective Adaptation in UAVs-Assisted Emergency Communication Networks","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y5POH3PWYHZDIQUBU322ZF56RX","json":"https://pith.science/pith/Y5POH3PWYHZDIQUBU322ZF56RX.json","graph_json":"https://pith.science/api/pith-number/Y5POH3PWYHZDIQUBU322ZF56RX/graph.json","events_json":"https://pith.science/api/pith-number/Y5POH3PWYHZDIQUBU322ZF56RX/events.json","paper":"https://pith.science/paper/Y5POH3PW"},"agent_actions":{"view_html":"https://pith.science/pith/Y5POH3PWYHZDIQUBU322ZF56RX","download_json":"https://pith.science/pith/Y5POH3PWYHZDIQUBU322ZF56RX.json","view_paper":"https://pith.science/paper/Y5POH3PW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.13264&json=true","fetch_graph":"https://pith.science/api/pith-number/Y5POH3PWYHZDIQUBU322ZF56RX/graph.json","fetch_events":"https://pith.science/api/pith-number/Y5POH3PWYHZDIQUBU322ZF56RX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y5POH3PWYHZDIQUBU322ZF56RX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y5POH3PWYHZDIQUBU322ZF56RX/action/storage_attestation","attest_author":"https://pith.science/pith/Y5POH3PWYHZDIQUBU322ZF56RX/action/author_attestation","sign_citation":"https://pith.science/pith/Y5POH3PWYHZDIQUBU322ZF56RX/action/citation_signature","submit_replication":"https://pith.science/pith/Y5POH3PWYHZDIQUBU322ZF56RX/action/replication_record"}},"created_at":"2026-07-05T03:46:13.826512+00:00","updated_at":"2026-07-05T03:46:13.826512+00:00"}