{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MDE23LFHYS2ZQSVRUZBW72LLIF","short_pith_number":"pith:MDE23LFH","schema_version":"1.0","canonical_sha256":"60c9adaca7c4b5984ab1a6436fe96b416630c2cc9b6c6c43b75259dc1f7a571e","source":{"kind":"arxiv","id":"2503.06343","version":2},"attestation_state":"computed","paper":{"title":"Studying the Interplay Between the Actor and Critic Representations in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Christopher G. Lucas, David Abel, Pablo Samuel Castro, Prakash Panangaden, Samuel Garcin, Stefano V. Albrecht, Trevor McInroe","submitted_at":"2025-03-08T21:29:20Z","abstract_excerpt":"Extracting relevant information from a stream of high-dimensional observations is a central challenge for deep reinforcement learning agents. Actor-critic algorithms add further complexity to this challenge, as it is often unclear whether the same information will be relevant to both the actor and the critic. To this end, we here explore the principles that underlie effective representations for the actor and for the critic in on-policy algorithms. We focus our study on understanding whether the actor and critic will benefit from separate, rather than shared, representations. Our primary findi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.06343","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-08T21:29:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5f2482d4d15d9766c379de632f86c4c617f88e801983fede5786f376760b50f6","abstract_canon_sha256":"3a2b054085a87eafb09b5608b4eecbeae8ede5a99754172696445443b87a74d9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:53.228746Z","signature_b64":"a4F1/mWbQi+ZU2EfpEVwwOirommu4H0naMnxMmP2X3IB7VX04jZZoquwtmigcPPMCJEdFvslb+cwCe120g9cAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"60c9adaca7c4b5984ab1a6436fe96b416630c2cc9b6c6c43b75259dc1f7a571e","last_reissued_at":"2026-07-05T10:41:53.228184Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:53.228184Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Studying the Interplay Between the Actor and Critic Representations in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Christopher G. Lucas, David Abel, Pablo Samuel Castro, Prakash Panangaden, Samuel Garcin, Stefano V. Albrecht, Trevor McInroe","submitted_at":"2025-03-08T21:29:20Z","abstract_excerpt":"Extracting relevant information from a stream of high-dimensional observations is a central challenge for deep reinforcement learning agents. Actor-critic algorithms add further complexity to this challenge, as it is often unclear whether the same information will be relevant to both the actor and the critic. To this end, we here explore the principles that underlie effective representations for the actor and for the critic in on-policy algorithms. We focus our study on understanding whether the actor and critic will benefit from separate, rather than shared, representations. Our primary findi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.06343","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.06343/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.06343","created_at":"2026-07-05T10:41:53.228243+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.06343v2","created_at":"2026-07-05T10:41:53.228243+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.06343","created_at":"2026-07-05T10:41:53.228243+00:00"},{"alias_kind":"pith_short_12","alias_value":"MDE23LFHYS2Z","created_at":"2026-07-05T10:41:53.228243+00:00"},{"alias_kind":"pith_short_16","alias_value":"MDE23LFHYS2ZQSVR","created_at":"2026-07-05T10:41:53.228243+00:00"},{"alias_kind":"pith_short_8","alias_value":"MDE23LFH","created_at":"2026-07-05T10:41:53.228243+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.14427","citing_title":"Self-Supervised Multisensory Pretraining for Contact-Rich Robot Reinforcement Learning","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MDE23LFHYS2ZQSVRUZBW72LLIF","json":"https://pith.science/pith/MDE23LFHYS2ZQSVRUZBW72LLIF.json","graph_json":"https://pith.science/api/pith-number/MDE23LFHYS2ZQSVRUZBW72LLIF/graph.json","events_json":"https://pith.science/api/pith-number/MDE23LFHYS2ZQSVRUZBW72LLIF/events.json","paper":"https://pith.science/paper/MDE23LFH"},"agent_actions":{"view_html":"https://pith.science/pith/MDE23LFHYS2ZQSVRUZBW72LLIF","download_json":"https://pith.science/pith/MDE23LFHYS2ZQSVRUZBW72LLIF.json","view_paper":"https://pith.science/paper/MDE23LFH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.06343&json=true","fetch_graph":"https://pith.science/api/pith-number/MDE23LFHYS2ZQSVRUZBW72LLIF/graph.json","fetch_events":"https://pith.science/api/pith-number/MDE23LFHYS2ZQSVRUZBW72LLIF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MDE23LFHYS2ZQSVRUZBW72LLIF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MDE23LFHYS2ZQSVRUZBW72LLIF/action/storage_attestation","attest_author":"https://pith.science/pith/MDE23LFHYS2ZQSVRUZBW72LLIF/action/author_attestation","sign_citation":"https://pith.science/pith/MDE23LFHYS2ZQSVRUZBW72LLIF/action/citation_signature","submit_replication":"https://pith.science/pith/MDE23LFHYS2ZQSVRUZBW72LLIF/action/replication_record"}},"created_at":"2026-07-05T10:41:53.228243+00:00","updated_at":"2026-07-05T10:41:53.228243+00:00"}