{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:ON5XCBDNPSHYSCUYJ5KIKDCNJR","short_pith_number":"pith:ON5XCBDN","schema_version":"1.0","canonical_sha256":"737b71046d7c8f890a984f54850c4d4c76e402ede5e3ca6ce8809046831fd4e9","source":{"kind":"arxiv","id":"2110.11280","version":2},"attestation_state":"computed","paper":{"title":"Actor-critic is implicitly biased towards high entropy optimal policies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Matus Telgarsky, Yuzheng Hu, Ziwei Ji","submitted_at":"2021-10-21T17:06:59Z","abstract_excerpt":"We show that the simplest actor-critic method -- a linear softmax policy updated with TD through interaction with a linear MDP, but featuring no explicit regularization or exploration -- does not merely find an optimal policy, but moreover prefers high entropy optimal policies. To demonstrate the strength of this bias, the algorithm not only has no regularization, no projections, and no exploration like $\\epsilon$-greedy, but is moreover trained on a single trajectory with no resets. The key consequence of the high entropy bias is that uniform mixing assumptions on the MDP, which exist in some"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.11280","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-10-21T17:06:59Z","cross_cats_sorted":[],"title_canon_sha256":"cf1ed7a46d8eba0ac6cbcd24f34a88dd47952fdad1f103c595d92e7ba6b50eff","abstract_canon_sha256":"b1c0b3696c85b61e06441b2a89a2ad7789b369d2281a340837ca9e73de2ee1e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:04:42.896526Z","signature_b64":"G+vkzrqMW/x4rbZk8/Bar+FtI+SjzGAijfnG8JIw4qsYBxQFH/sLy1NkJ9T1dQQ5RnI2FvAcysCnitzVVUmmAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"737b71046d7c8f890a984f54850c4d4c76e402ede5e3ca6ce8809046831fd4e9","last_reissued_at":"2026-07-05T04:04:42.896084Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:04:42.896084Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Actor-critic is implicitly biased towards high entropy optimal policies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Matus Telgarsky, Yuzheng Hu, Ziwei Ji","submitted_at":"2021-10-21T17:06:59Z","abstract_excerpt":"We show that the simplest actor-critic method -- a linear softmax policy updated with TD through interaction with a linear MDP, but featuring no explicit regularization or exploration -- does not merely find an optimal policy, but moreover prefers high entropy optimal policies. To demonstrate the strength of this bias, the algorithm not only has no regularization, no projections, and no exploration like $\\epsilon$-greedy, but is moreover trained on a single trajectory with no resets. The key consequence of the high entropy bias is that uniform mixing assumptions on the MDP, which exist in some"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.11280","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.11280/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.11280","created_at":"2026-07-05T04:04:42.896141+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.11280v2","created_at":"2026-07-05T04:04:42.896141+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.11280","created_at":"2026-07-05T04:04:42.896141+00:00"},{"alias_kind":"pith_short_12","alias_value":"ON5XCBDNPSHY","created_at":"2026-07-05T04:04:42.896141+00:00"},{"alias_kind":"pith_short_16","alias_value":"ON5XCBDNPSHYSCUY","created_at":"2026-07-05T04:04:42.896141+00:00"},{"alias_kind":"pith_short_8","alias_value":"ON5XCBDN","created_at":"2026-07-05T04:04:42.896141+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ON5XCBDNPSHYSCUYJ5KIKDCNJR","json":"https://pith.science/pith/ON5XCBDNPSHYSCUYJ5KIKDCNJR.json","graph_json":"https://pith.science/api/pith-number/ON5XCBDNPSHYSCUYJ5KIKDCNJR/graph.json","events_json":"https://pith.science/api/pith-number/ON5XCBDNPSHYSCUYJ5KIKDCNJR/events.json","paper":"https://pith.science/paper/ON5XCBDN"},"agent_actions":{"view_html":"https://pith.science/pith/ON5XCBDNPSHYSCUYJ5KIKDCNJR","download_json":"https://pith.science/pith/ON5XCBDNPSHYSCUYJ5KIKDCNJR.json","view_paper":"https://pith.science/paper/ON5XCBDN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.11280&json=true","fetch_graph":"https://pith.science/api/pith-number/ON5XCBDNPSHYSCUYJ5KIKDCNJR/graph.json","fetch_events":"https://pith.science/api/pith-number/ON5XCBDNPSHYSCUYJ5KIKDCNJR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ON5XCBDNPSHYSCUYJ5KIKDCNJR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ON5XCBDNPSHYSCUYJ5KIKDCNJR/action/storage_attestation","attest_author":"https://pith.science/pith/ON5XCBDNPSHYSCUYJ5KIKDCNJR/action/author_attestation","sign_citation":"https://pith.science/pith/ON5XCBDNPSHYSCUYJ5KIKDCNJR/action/citation_signature","submit_replication":"https://pith.science/pith/ON5XCBDNPSHYSCUYJ5KIKDCNJR/action/replication_record"}},"created_at":"2026-07-05T04:04:42.896141+00:00","updated_at":"2026-07-05T04:04:42.896141+00:00"}