{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:T5SRXTLROFNWXQUAUMMIT2MXTA","short_pith_number":"pith:T5SRXTLR","schema_version":"1.0","canonical_sha256":"9f651bcd71715b6bc280a31889e997981a9db5d3429cb233652a9a7130b97c7d","source":{"kind":"arxiv","id":"2201.12434","version":1},"attestation_state":"computed","paper":{"title":"Do You Need the Entropy Reward (in Practice)?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Haichao Zhang, Haonan Yu, Wei Xu","submitted_at":"2022-01-28T21:43:21Z","abstract_excerpt":"Maximum entropy (MaxEnt) RL maximizes a combination of the original task reward and an entropy reward. It is believed that the regularization imposed by entropy, on both policy improvement and policy evaluation, together contributes to good exploration, training convergence, and robustness of learned policies. This paper takes a closer look at entropy as an intrinsic reward, by conducting various ablation studies on soft actor-critic (SAC), a popular representative of MaxEnt RL. Our findings reveal that in general, entropy rewards should be applied with caution to policy evaluation. On one han"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.12434","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-28T21:43:21Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"91ff2a00ccac90c414dd5a80127b01017e46133aa2fbc86b3af62db3a06b3540","abstract_canon_sha256":"cd3224b13723ebb085e53a1399e2ddb1d10c4ef11027b8e094c6f8695e6444a5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:52:36.460635Z","signature_b64":"aTkx4y10sFQYwAXSayZrCGzj+ieynyXt+PIyIeesJzO9vnIUUbF+/hlGp+VpMtc0D7HkTpx815gTvXCGV1KKAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f651bcd71715b6bc280a31889e997981a9db5d3429cb233652a9a7130b97c7d","last_reissued_at":"2026-07-05T03:52:36.460241Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:52:36.460241Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do You Need the Entropy Reward (in Practice)?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Haichao Zhang, Haonan Yu, Wei Xu","submitted_at":"2022-01-28T21:43:21Z","abstract_excerpt":"Maximum entropy (MaxEnt) RL maximizes a combination of the original task reward and an entropy reward. It is believed that the regularization imposed by entropy, on both policy improvement and policy evaluation, together contributes to good exploration, training convergence, and robustness of learned policies. This paper takes a closer look at entropy as an intrinsic reward, by conducting various ablation studies on soft actor-critic (SAC), a popular representative of MaxEnt RL. Our findings reveal that in general, entropy rewards should be applied with caution to policy evaluation. On one han"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.12434","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.12434/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.12434","created_at":"2026-07-05T03:52:36.460295+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.12434v1","created_at":"2026-07-05T03:52:36.460295+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.12434","created_at":"2026-07-05T03:52:36.460295+00:00"},{"alias_kind":"pith_short_12","alias_value":"T5SRXTLROFNW","created_at":"2026-07-05T03:52:36.460295+00:00"},{"alias_kind":"pith_short_16","alias_value":"T5SRXTLROFNWXQUA","created_at":"2026-07-05T03:52:36.460295+00:00"},{"alias_kind":"pith_short_8","alias_value":"T5SRXTLR","created_at":"2026-07-05T03:52:36.460295+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.03859","citing_title":"Learning Multi-Stage Pick-and-Place with a Legged Mobile Manipulator","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T5SRXTLROFNWXQUAUMMIT2MXTA","json":"https://pith.science/pith/T5SRXTLROFNWXQUAUMMIT2MXTA.json","graph_json":"https://pith.science/api/pith-number/T5SRXTLROFNWXQUAUMMIT2MXTA/graph.json","events_json":"https://pith.science/api/pith-number/T5SRXTLROFNWXQUAUMMIT2MXTA/events.json","paper":"https://pith.science/paper/T5SRXTLR"},"agent_actions":{"view_html":"https://pith.science/pith/T5SRXTLROFNWXQUAUMMIT2MXTA","download_json":"https://pith.science/pith/T5SRXTLROFNWXQUAUMMIT2MXTA.json","view_paper":"https://pith.science/paper/T5SRXTLR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.12434&json=true","fetch_graph":"https://pith.science/api/pith-number/T5SRXTLROFNWXQUAUMMIT2MXTA/graph.json","fetch_events":"https://pith.science/api/pith-number/T5SRXTLROFNWXQUAUMMIT2MXTA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T5SRXTLROFNWXQUAUMMIT2MXTA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T5SRXTLROFNWXQUAUMMIT2MXTA/action/storage_attestation","attest_author":"https://pith.science/pith/T5SRXTLROFNWXQUAUMMIT2MXTA/action/author_attestation","sign_citation":"https://pith.science/pith/T5SRXTLROFNWXQUAUMMIT2MXTA/action/citation_signature","submit_replication":"https://pith.science/pith/T5SRXTLROFNWXQUAUMMIT2MXTA/action/replication_record"}},"created_at":"2026-07-05T03:52:36.460295+00:00","updated_at":"2026-07-05T03:52:36.460295+00:00"}