{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:SMXVIWIHO3MSQI6PMWBT3YYPIU","short_pith_number":"pith:SMXVIWIH","schema_version":"1.0","canonical_sha256":"932f54590776d92823cf65833de30f4510afcf094d0a2458d7fdcfe6796a65a6","source":{"kind":"arxiv","id":"2103.06257","version":2},"attestation_state":"computed","paper":{"title":"Maximum Entropy RL (Provably) Solves Some Robust RL Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Benjamin Eysenbach, Sergey Levine","submitted_at":"2021-03-10T18:45:48Z","abstract_excerpt":"Many potential applications of reinforcement learning (RL) require guarantees that the agent will perform well in the face of disturbances to the dynamics or reward function. In this paper, we prove theoretically that maximum entropy (MaxEnt) RL maximizes a lower bound on a robust RL objective, and thus can be used to learn policies that are robust to some disturbances in the dynamics and the reward function. While this capability of MaxEnt RL has been observed empirically in prior work, to the best of our knowledge our work provides the first rigorous proof and theoretical characterization of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.06257","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-03-10T18:45:48Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"aa5b3f2ea07ae0e77a8345dc7b702d583010d53793cb47c9c9f550727fc26c64","abstract_canon_sha256":"fe0bb454224db28e6d4fb594df92cd007efde427993a4276c7ae55067e47ee0e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:20:28.053207Z","signature_b64":"z5nXqfqz0Gvd236Mtb7We26ePjm4LPKl4LCnWJtLaLfmUYUhkYH1mYeJ/pM01zHmyXczSk2u43g798y5tKv1BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"932f54590776d92823cf65833de30f4510afcf094d0a2458d7fdcfe6796a65a6","last_reissued_at":"2026-07-05T04:20:28.052747Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:20:28.052747Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Maximum Entropy RL (Provably) Solves Some Robust RL Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Benjamin Eysenbach, Sergey Levine","submitted_at":"2021-03-10T18:45:48Z","abstract_excerpt":"Many potential applications of reinforcement learning (RL) require guarantees that the agent will perform well in the face of disturbances to the dynamics or reward function. In this paper, we prove theoretically that maximum entropy (MaxEnt) RL maximizes a lower bound on a robust RL objective, and thus can be used to learn policies that are robust to some disturbances in the dynamics and the reward function. While this capability of MaxEnt RL has been observed empirically in prior work, to the best of our knowledge our work provides the first rigorous proof and theoretical characterization of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.06257","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.06257/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.06257","created_at":"2026-07-05T04:20:28.052804+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.06257v2","created_at":"2026-07-05T04:20:28.052804+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.06257","created_at":"2026-07-05T04:20:28.052804+00:00"},{"alias_kind":"pith_short_12","alias_value":"SMXVIWIHO3MS","created_at":"2026-07-05T04:20:28.052804+00:00"},{"alias_kind":"pith_short_16","alias_value":"SMXVIWIHO3MSQI6P","created_at":"2026-07-05T04:20:28.052804+00:00"},{"alias_kind":"pith_short_8","alias_value":"SMXVIWIH","created_at":"2026-07-05T04:20:28.052804+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":223,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24005","citing_title":"LC-ERD: Mining Latent Logic for Self-Evolving Reasoning via Consistency-Regulated Reward Decomposition","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07602","citing_title":"Sample-Efficient Post-Training for LEGO Spatial-Physics Reasoning","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24957","citing_title":"Compute Aligned Training: Optimizing for Test Time Inference","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19461","citing_title":"Beyond Mode Collapse: Distribution Matching for Diverse Reasoning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15134","citing_title":"The Unreasonable Effectiveness of Entropy Minimization in LLM Reasoning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20265","citing_title":"Failure Modes of Maximum Entropy RLHF","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08813","citing_title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09349","citing_title":"Mutual Information Optimal Density Control of Linear Systems and Generalized Schr\\\"{o}dinger Bridges with Reference Refinement","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24957","citing_title":"Compute Aligned Training: Optimizing for Test Time Inference","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10974","citing_title":"Robust Adversarial Policy Optimization Under Dynamics Uncertainty","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14265","citing_title":"Reinforcement Learning via Value Gradient Flow","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SMXVIWIHO3MSQI6PMWBT3YYPIU","json":"https://pith.science/pith/SMXVIWIHO3MSQI6PMWBT3YYPIU.json","graph_json":"https://pith.science/api/pith-number/SMXVIWIHO3MSQI6PMWBT3YYPIU/graph.json","events_json":"https://pith.science/api/pith-number/SMXVIWIHO3MSQI6PMWBT3YYPIU/events.json","paper":"https://pith.science/paper/SMXVIWIH"},"agent_actions":{"view_html":"https://pith.science/pith/SMXVIWIHO3MSQI6PMWBT3YYPIU","download_json":"https://pith.science/pith/SMXVIWIHO3MSQI6PMWBT3YYPIU.json","view_paper":"https://pith.science/paper/SMXVIWIH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.06257&json=true","fetch_graph":"https://pith.science/api/pith-number/SMXVIWIHO3MSQI6PMWBT3YYPIU/graph.json","fetch_events":"https://pith.science/api/pith-number/SMXVIWIHO3MSQI6PMWBT3YYPIU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SMXVIWIHO3MSQI6PMWBT3YYPIU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SMXVIWIHO3MSQI6PMWBT3YYPIU/action/storage_attestation","attest_author":"https://pith.science/pith/SMXVIWIHO3MSQI6PMWBT3YYPIU/action/author_attestation","sign_citation":"https://pith.science/pith/SMXVIWIHO3MSQI6PMWBT3YYPIU/action/citation_signature","submit_replication":"https://pith.science/pith/SMXVIWIHO3MSQI6PMWBT3YYPIU/action/replication_record"}},"created_at":"2026-07-05T04:20:28.052804+00:00","updated_at":"2026-07-05T04:20:28.052804+00:00"}