{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:X3O5THQBIYM4TP3JI6VRS5WKUW","short_pith_number":"pith:X3O5THQB","schema_version":"1.0","canonical_sha256":"beddd99e014619c9bf6947ab1976caa58ac73eee9ea4167cbbfac1f3a5909341","source":{"kind":"arxiv","id":"2108.05828","version":5},"attestation_state":"computed","paper":{"title":"A general class of surrogate functions for stable and efficient reinforcement learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Marlos C. Machado, Matthieu Geist, Nicolas Le Roux, Olivier Bachem, Pablo Samuel Castro, Robert Mueller, Sharan Vaswani, Shivam Garg, Simone Totaro","submitted_at":"2021-08-12T16:19:19Z","abstract_excerpt":"Common policy gradient methods rely on the maximization of a sequence of surrogate functions. In recent years, many such surrogate functions have been proposed, most without strong theoretical guarantees, leading to algorithms such as TRPO, PPO or MPO. Rather than design yet another surrogate function, we instead propose a general framework (FMA-PG) based on functional mirror ascent that gives rise to an entire family of surrogate functions. We construct surrogate functions that enable policy improvement guarantees, a property not shared by most existing surrogate functions. Crucially, these g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.05828","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-08-12T16:19:19Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"6222cb7cc9e9886d41c817738d2c3b8eed032e3e3ed65880b9545de9250952f3","abstract_canon_sha256":"3b150658428b655ddf96aae1db4b2bee84b99c7ca86cccdb9a36b7cc28102d86"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:07:02.677029Z","signature_b64":"x7KVGnwOLYe1P4O2QeLa12kIQL6R5dyd4oyKJJ5rI0UwbsxSpVvouAtDBVcJ7ZWHhMZJhjIlnAO8f462/xqQCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"beddd99e014619c9bf6947ab1976caa58ac73eee9ea4167cbbfac1f3a5909341","last_reissued_at":"2026-07-05T07:07:02.676509Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:07:02.676509Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A general class of surrogate functions for stable and efficient reinforcement learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Marlos C. Machado, Matthieu Geist, Nicolas Le Roux, Olivier Bachem, Pablo Samuel Castro, Robert Mueller, Sharan Vaswani, Shivam Garg, Simone Totaro","submitted_at":"2021-08-12T16:19:19Z","abstract_excerpt":"Common policy gradient methods rely on the maximization of a sequence of surrogate functions. In recent years, many such surrogate functions have been proposed, most without strong theoretical guarantees, leading to algorithms such as TRPO, PPO or MPO. Rather than design yet another surrogate function, we instead propose a general framework (FMA-PG) based on functional mirror ascent that gives rise to an entire family of surrogate functions. We construct surrogate functions that enable policy improvement guarantees, a property not shared by most existing surrogate functions. Crucially, these g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.05828","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.05828/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.05828","created_at":"2026-07-05T07:07:02.676568+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.05828v5","created_at":"2026-07-05T07:07:02.676568+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.05828","created_at":"2026-07-05T07:07:02.676568+00:00"},{"alias_kind":"pith_short_12","alias_value":"X3O5THQBIYM4","created_at":"2026-07-05T07:07:02.676568+00:00"},{"alias_kind":"pith_short_16","alias_value":"X3O5THQBIYM4TP3J","created_at":"2026-07-05T07:07:02.676568+00:00"},{"alias_kind":"pith_short_8","alias_value":"X3O5THQB","created_at":"2026-07-05T07:07:02.676568+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18591","citing_title":"Randomized Advantage Transformation (RAT): Computing Natural Policy Gradients via Direct Backpropagation","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09838","citing_title":"Dissecting Discrete Soft Actor-Critic: Limitations and Principled Alternatives","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11694","citing_title":"Augmented Lagrangian Method for Last-Iterate Convergence for Constrained MDPs","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X3O5THQBIYM4TP3JI6VRS5WKUW","json":"https://pith.science/pith/X3O5THQBIYM4TP3JI6VRS5WKUW.json","graph_json":"https://pith.science/api/pith-number/X3O5THQBIYM4TP3JI6VRS5WKUW/graph.json","events_json":"https://pith.science/api/pith-number/X3O5THQBIYM4TP3JI6VRS5WKUW/events.json","paper":"https://pith.science/paper/X3O5THQB"},"agent_actions":{"view_html":"https://pith.science/pith/X3O5THQBIYM4TP3JI6VRS5WKUW","download_json":"https://pith.science/pith/X3O5THQBIYM4TP3JI6VRS5WKUW.json","view_paper":"https://pith.science/paper/X3O5THQB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.05828&json=true","fetch_graph":"https://pith.science/api/pith-number/X3O5THQBIYM4TP3JI6VRS5WKUW/graph.json","fetch_events":"https://pith.science/api/pith-number/X3O5THQBIYM4TP3JI6VRS5WKUW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X3O5THQBIYM4TP3JI6VRS5WKUW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X3O5THQBIYM4TP3JI6VRS5WKUW/action/storage_attestation","attest_author":"https://pith.science/pith/X3O5THQBIYM4TP3JI6VRS5WKUW/action/author_attestation","sign_citation":"https://pith.science/pith/X3O5THQBIYM4TP3JI6VRS5WKUW/action/citation_signature","submit_replication":"https://pith.science/pith/X3O5THQBIYM4TP3JI6VRS5WKUW/action/replication_record"}},"created_at":"2026-07-05T07:07:02.676568+00:00","updated_at":"2026-07-05T07:07:02.676568+00:00"}