{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SD7WNCF56IN2CB2TKGETLTC7FI","short_pith_number":"pith:SD7WNCF5","schema_version":"1.0","canonical_sha256":"90ff6688bdf21ba10753518935cc5f2a21015338db8c4e19e905da77e6723bc6","source":{"kind":"arxiv","id":"2501.09080","version":2},"attestation_state":"computed","paper":{"title":"Average-Reward Soft Actor-Critic","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jacob Adamczyk, Rahul V. Kulkarni, Stas Tiomkin, Volodymyr Makarenko","submitted_at":"2025-01-15T19:00:46Z","abstract_excerpt":"The average-reward formulation of reinforcement learning (RL) has drawn increased interest in recent years for its ability to solve temporally-extended problems without relying on discounting. Meanwhile, in the discounted setting, algorithms with entropy regularization have been developed, leading to improvements over deterministic methods. Despite the distinct benefits of these approaches, deep RL algorithms for the entropy-regularized average-reward objective have not been developed. While policy-gradient based approaches have recently been presented for the average-reward literature, the co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.09080","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-15T19:00:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"63034f36de7feb19850d800e2ddc058451bcaa153d5ad4bf9ac7131ef5263438","abstract_canon_sha256":"f0c4027534a4d9ddc37368252e4becd5c190bb137784e46769ee2c06093c2509"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:47.164454Z","signature_b64":"utFsVfDodk9N0D7h+SI98zvQhVPMQg0vxkK5fzVeOhNih1QBwhMj+J0yR5qVKNepWS74hkZFNmDPDCQ0ixyHCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"90ff6688bdf21ba10753518935cc5f2a21015338db8c4e19e905da77e6723bc6","last_reissued_at":"2026-07-05T11:48:47.163983Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:47.163983Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Average-Reward Soft Actor-Critic","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jacob Adamczyk, Rahul V. Kulkarni, Stas Tiomkin, Volodymyr Makarenko","submitted_at":"2025-01-15T19:00:46Z","abstract_excerpt":"The average-reward formulation of reinforcement learning (RL) has drawn increased interest in recent years for its ability to solve temporally-extended problems without relying on discounting. Meanwhile, in the discounted setting, algorithms with entropy regularization have been developed, leading to improvements over deterministic methods. Despite the distinct benefits of these approaches, deep RL algorithms for the entropy-regularized average-reward objective have not been developed. While policy-gradient based approaches have recently been presented for the average-reward literature, the co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.09080","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.09080/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.09080","created_at":"2026-07-05T11:48:47.164037+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.09080v2","created_at":"2026-07-05T11:48:47.164037+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.09080","created_at":"2026-07-05T11:48:47.164037+00:00"},{"alias_kind":"pith_short_12","alias_value":"SD7WNCF56IN2","created_at":"2026-07-05T11:48:47.164037+00:00"},{"alias_kind":"pith_short_16","alias_value":"SD7WNCF56IN2CB2T","created_at":"2026-07-05T11:48:47.164037+00:00"},{"alias_kind":"pith_short_8","alias_value":"SD7WNCF5","created_at":"2026-07-05T11:48:47.164037+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.19910","citing_title":"Learning Adaptive Parameter Policies for Nonlinear Bayesian Filtering","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12112","citing_title":"When Policy Entropy Constraint Fails: Preserving Diversity in Flow-based RLHF via Perceptual Entropy","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SD7WNCF56IN2CB2TKGETLTC7FI","json":"https://pith.science/pith/SD7WNCF56IN2CB2TKGETLTC7FI.json","graph_json":"https://pith.science/api/pith-number/SD7WNCF56IN2CB2TKGETLTC7FI/graph.json","events_json":"https://pith.science/api/pith-number/SD7WNCF56IN2CB2TKGETLTC7FI/events.json","paper":"https://pith.science/paper/SD7WNCF5"},"agent_actions":{"view_html":"https://pith.science/pith/SD7WNCF56IN2CB2TKGETLTC7FI","download_json":"https://pith.science/pith/SD7WNCF56IN2CB2TKGETLTC7FI.json","view_paper":"https://pith.science/paper/SD7WNCF5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.09080&json=true","fetch_graph":"https://pith.science/api/pith-number/SD7WNCF56IN2CB2TKGETLTC7FI/graph.json","fetch_events":"https://pith.science/api/pith-number/SD7WNCF56IN2CB2TKGETLTC7FI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SD7WNCF56IN2CB2TKGETLTC7FI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SD7WNCF56IN2CB2TKGETLTC7FI/action/storage_attestation","attest_author":"https://pith.science/pith/SD7WNCF56IN2CB2TKGETLTC7FI/action/author_attestation","sign_citation":"https://pith.science/pith/SD7WNCF56IN2CB2TKGETLTC7FI/action/citation_signature","submit_replication":"https://pith.science/pith/SD7WNCF56IN2CB2TKGETLTC7FI/action/replication_record"}},"created_at":"2026-07-05T11:48:47.164037+00:00","updated_at":"2026-07-05T11:48:47.164037+00:00"}