{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6RBFIKPYUD3D7RZNN4ROZ2A7FL","short_pith_number":"pith:6RBFIKPY","schema_version":"1.0","canonical_sha256":"f4425429f8a0f63fc72d6f22ece81f2acb11abd180e9e377cb15a03fcbabfc4f","source":{"kind":"arxiv","id":"2506.13672","version":1},"attestation_state":"computed","paper":{"title":"The Courage to Stop: Overcoming Sunk Cost Fallacy in Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Jiashun Liu, Johan Obando-Ceron, Ling Pan, Pablo Samuel Castro","submitted_at":"2025-06-16T16:30:00Z","abstract_excerpt":"Off-policy deep reinforcement learning (RL) typically leverages replay buffers for reusing past experiences during learning. This can help improve sample efficiency when the collected data is informative and aligned with the learning objectives; when that is not the case, it can have the effect of \"polluting\" the replay buffer with data which can exacerbate optimization challenges in addition to wasting environment interactions due to wasteful sampling. We argue that sampling these uninformative and wasteful transitions can be avoided by addressing the sunk cost fallacy, which, in the context "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.13672","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-16T16:30:00Z","cross_cats_sorted":[],"title_canon_sha256":"c87eb952fd5c316c81b5d527b3a99aff5311f9cae5622717e0dea8c9edd2ec04","abstract_canon_sha256":"a8375f75560f6b177759a07cab813bbdc94104afe5ecb49c2aadee148286a368"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:24.738786Z","signature_b64":"SDYjhwfjUBAgVye2vtRcdNmbJ1IXqIR630kATJ6FWLh5BTTojaJsHi2WoWhNVX15UEBQF6W0tZ7D4sdiJmgPDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f4425429f8a0f63fc72d6f22ece81f2acb11abd180e9e377cb15a03fcbabfc4f","last_reissued_at":"2026-07-05T11:22:24.738195Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:24.738195Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Courage to Stop: Overcoming Sunk Cost Fallacy in Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Jiashun Liu, Johan Obando-Ceron, Ling Pan, Pablo Samuel Castro","submitted_at":"2025-06-16T16:30:00Z","abstract_excerpt":"Off-policy deep reinforcement learning (RL) typically leverages replay buffers for reusing past experiences during learning. This can help improve sample efficiency when the collected data is informative and aligned with the learning objectives; when that is not the case, it can have the effect of \"polluting\" the replay buffer with data which can exacerbate optimization challenges in addition to wasting environment interactions due to wasteful sampling. We argue that sampling these uninformative and wasteful transitions can be avoided by addressing the sunk cost fallacy, which, in the context "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13672","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13672/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.13672","created_at":"2026-07-05T11:22:24.738266+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.13672v1","created_at":"2026-07-05T11:22:24.738266+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13672","created_at":"2026-07-05T11:22:24.738266+00:00"},{"alias_kind":"pith_short_12","alias_value":"6RBFIKPYUD3D","created_at":"2026-07-05T11:22:24.738266+00:00"},{"alias_kind":"pith_short_16","alias_value":"6RBFIKPYUD3D7RZN","created_at":"2026-07-05T11:22:24.738266+00:00"},{"alias_kind":"pith_short_8","alias_value":"6RBFIKPY","created_at":"2026-07-05T11:22:24.738266+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6RBFIKPYUD3D7RZNN4ROZ2A7FL","json":"https://pith.science/pith/6RBFIKPYUD3D7RZNN4ROZ2A7FL.json","graph_json":"https://pith.science/api/pith-number/6RBFIKPYUD3D7RZNN4ROZ2A7FL/graph.json","events_json":"https://pith.science/api/pith-number/6RBFIKPYUD3D7RZNN4ROZ2A7FL/events.json","paper":"https://pith.science/paper/6RBFIKPY"},"agent_actions":{"view_html":"https://pith.science/pith/6RBFIKPYUD3D7RZNN4ROZ2A7FL","download_json":"https://pith.science/pith/6RBFIKPYUD3D7RZNN4ROZ2A7FL.json","view_paper":"https://pith.science/paper/6RBFIKPY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.13672&json=true","fetch_graph":"https://pith.science/api/pith-number/6RBFIKPYUD3D7RZNN4ROZ2A7FL/graph.json","fetch_events":"https://pith.science/api/pith-number/6RBFIKPYUD3D7RZNN4ROZ2A7FL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6RBFIKPYUD3D7RZNN4ROZ2A7FL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6RBFIKPYUD3D7RZNN4ROZ2A7FL/action/storage_attestation","attest_author":"https://pith.science/pith/6RBFIKPYUD3D7RZNN4ROZ2A7FL/action/author_attestation","sign_citation":"https://pith.science/pith/6RBFIKPYUD3D7RZNN4ROZ2A7FL/action/citation_signature","submit_replication":"https://pith.science/pith/6RBFIKPYUD3D7RZNN4ROZ2A7FL/action/replication_record"}},"created_at":"2026-07-05T11:22:24.738266+00:00","updated_at":"2026-07-05T11:22:24.738266+00:00"}