{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:B3S7UVX4YJQA2Z6HZCDZWRU7WW","short_pith_number":"pith:B3S7UVX4","schema_version":"1.0","canonical_sha256":"0ee5fa56fcc2600d67c7c8879b469fb5ab7248e7fae7441cbad69e3085b8327c","source":{"kind":"arxiv","id":"2102.10330","version":2},"attestation_state":"computed","paper":{"title":"Decoupling Value and Policy for Generalization in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Roberta Raileanu, Rob Fergus","submitted_at":"2021-02-20T12:40:11Z","abstract_excerpt":"Standard deep reinforcement learning algorithms use a shared representation for the policy and value function, especially when training directly from images. However, we argue that more information is needed to accurately estimate the value function than to learn the optimal policy. Consequently, the use of a shared representation for the policy and value function can lead to overfitting. To alleviate this problem, we propose two approaches which are combined to create IDAAC: Invariant Decoupled Advantage Actor-Critic. First, IDAAC decouples the optimization of the policy and value function, u"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.10330","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-02-20T12:40:11Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0d2b1881ce66b8b3308cbdb5113993eb82eb3672358e34a35b598e7cf91779bc","abstract_canon_sha256":"599fe56fac5cd401f5e3a226002c18117955582c891a6a41303f584aae0d1ba8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:49:25.428272Z","signature_b64":"aYcKNUinMamVaalFFhayB6Rkeqy4px2D0iXUngqy8SBzadIdWSYxWHFI3Otn5RoBTKlZvO6vci7gCcBeYNuTAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ee5fa56fcc2600d67c7c8879b469fb5ab7248e7fae7441cbad69e3085b8327c","last_reissued_at":"2026-07-05T02:49:25.427812Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:49:25.427812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Decoupling Value and Policy for Generalization in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Roberta Raileanu, Rob Fergus","submitted_at":"2021-02-20T12:40:11Z","abstract_excerpt":"Standard deep reinforcement learning algorithms use a shared representation for the policy and value function, especially when training directly from images. However, we argue that more information is needed to accurately estimate the value function than to learn the optimal policy. Consequently, the use of a shared representation for the policy and value function can lead to overfitting. To alleviate this problem, we propose two approaches which are combined to create IDAAC: Invariant Decoupled Advantage Actor-Critic. First, IDAAC decouples the optimization of the policy and value function, u"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.10330","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.10330/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.10330","created_at":"2026-07-05T02:49:25.427869+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.10330v2","created_at":"2026-07-05T02:49:25.427869+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.10330","created_at":"2026-07-05T02:49:25.427869+00:00"},{"alias_kind":"pith_short_12","alias_value":"B3S7UVX4YJQA","created_at":"2026-07-05T02:49:25.427869+00:00"},{"alias_kind":"pith_short_16","alias_value":"B3S7UVX4YJQA2Z6H","created_at":"2026-07-05T02:49:25.427869+00:00"},{"alias_kind":"pith_short_8","alias_value":"B3S7UVX4","created_at":"2026-07-05T02:49:25.427869+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B3S7UVX4YJQA2Z6HZCDZWRU7WW","json":"https://pith.science/pith/B3S7UVX4YJQA2Z6HZCDZWRU7WW.json","graph_json":"https://pith.science/api/pith-number/B3S7UVX4YJQA2Z6HZCDZWRU7WW/graph.json","events_json":"https://pith.science/api/pith-number/B3S7UVX4YJQA2Z6HZCDZWRU7WW/events.json","paper":"https://pith.science/paper/B3S7UVX4"},"agent_actions":{"view_html":"https://pith.science/pith/B3S7UVX4YJQA2Z6HZCDZWRU7WW","download_json":"https://pith.science/pith/B3S7UVX4YJQA2Z6HZCDZWRU7WW.json","view_paper":"https://pith.science/paper/B3S7UVX4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.10330&json=true","fetch_graph":"https://pith.science/api/pith-number/B3S7UVX4YJQA2Z6HZCDZWRU7WW/graph.json","fetch_events":"https://pith.science/api/pith-number/B3S7UVX4YJQA2Z6HZCDZWRU7WW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B3S7UVX4YJQA2Z6HZCDZWRU7WW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B3S7UVX4YJQA2Z6HZCDZWRU7WW/action/storage_attestation","attest_author":"https://pith.science/pith/B3S7UVX4YJQA2Z6HZCDZWRU7WW/action/author_attestation","sign_citation":"https://pith.science/pith/B3S7UVX4YJQA2Z6HZCDZWRU7WW/action/citation_signature","submit_replication":"https://pith.science/pith/B3S7UVX4YJQA2Z6HZCDZWRU7WW/action/replication_record"}},"created_at":"2026-07-05T02:49:25.427869+00:00","updated_at":"2026-07-05T02:49:25.427869+00:00"}