{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VYNVNGKJA6NICG6NTSFRH5SSRY","short_pith_number":"pith:VYNVNGKJ","schema_version":"1.0","canonical_sha256":"ae1b569949079a811bcd9c8b13f6528e286a089ba33cf178bc8f55572501d08a","source":{"kind":"arxiv","id":"2401.10240","version":2},"attestation_state":"computed","paper":{"title":"Policy Evaluation in Distributional LQR (Extended Version)","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Alessandro Abate, Karl H. Johansson, Michael M. Zavlanos, Siyi Wang, Yulong Gao, Zifan Wang","submitted_at":"2023-11-28T17:15:07Z","abstract_excerpt":"Distributional reinforcement learning (DRL) enhances the understanding of the effects of the randomness in the environment by letting agents learn the distribution of a random return, rather than its expected value as in standard reinforcement learning. Meanwhile, a challenge in DRL is that the policy evaluation typically relies on the representation of the return distribution, which needs to be carefully designed. In this paper, we address this challenge for the special class of DRL problems that rely on a discounted linear quadratic regulator (LQR), which we call \\emph{distributional LQR}. S"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.10240","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2023-11-28T17:15:07Z","cross_cats_sorted":[],"title_canon_sha256":"de089ebba6b5f38f18a5b3daa49193e676a1ccf3affafe5fdbfa6b41017d1d81","abstract_canon_sha256":"e4086b58c27fd561a4c179e26b2c28bd71fb5e28105531c43f5d76f2f59866fa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:00:00.327593Z","signature_b64":"hEoVTPFQfPu/roJeDi2rZ2WVYTWUbmA6VRhk96ZE0I42mBpGqJ0DsGjZ7441Mot1Nnk6mjLjdu5W6sD5glbnDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae1b569949079a811bcd9c8b13f6528e286a089ba33cf178bc8f55572501d08a","last_reissued_at":"2026-07-05T08:00:00.327074Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:00:00.327074Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Policy Evaluation in Distributional LQR (Extended Version)","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Alessandro Abate, Karl H. Johansson, Michael M. Zavlanos, Siyi Wang, Yulong Gao, Zifan Wang","submitted_at":"2023-11-28T17:15:07Z","abstract_excerpt":"Distributional reinforcement learning (DRL) enhances the understanding of the effects of the randomness in the environment by letting agents learn the distribution of a random return, rather than its expected value as in standard reinforcement learning. Meanwhile, a challenge in DRL is that the policy evaluation typically relies on the representation of the return distribution, which needs to be carefully designed. In this paper, we address this challenge for the special class of DRL problems that rely on a discounted linear quadratic regulator (LQR), which we call \\emph{distributional LQR}. S"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.10240","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.10240/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.10240","created_at":"2026-07-05T08:00:00.327132+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.10240v2","created_at":"2026-07-05T08:00:00.327132+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.10240","created_at":"2026-07-05T08:00:00.327132+00:00"},{"alias_kind":"pith_short_12","alias_value":"VYNVNGKJA6NI","created_at":"2026-07-05T08:00:00.327132+00:00"},{"alias_kind":"pith_short_16","alias_value":"VYNVNGKJA6NICG6N","created_at":"2026-07-05T08:00:00.327132+00:00"},{"alias_kind":"pith_short_8","alias_value":"VYNVNGKJ","created_at":"2026-07-05T08:00:00.327132+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VYNVNGKJA6NICG6NTSFRH5SSRY","json":"https://pith.science/pith/VYNVNGKJA6NICG6NTSFRH5SSRY.json","graph_json":"https://pith.science/api/pith-number/VYNVNGKJA6NICG6NTSFRH5SSRY/graph.json","events_json":"https://pith.science/api/pith-number/VYNVNGKJA6NICG6NTSFRH5SSRY/events.json","paper":"https://pith.science/paper/VYNVNGKJ"},"agent_actions":{"view_html":"https://pith.science/pith/VYNVNGKJA6NICG6NTSFRH5SSRY","download_json":"https://pith.science/pith/VYNVNGKJA6NICG6NTSFRH5SSRY.json","view_paper":"https://pith.science/paper/VYNVNGKJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.10240&json=true","fetch_graph":"https://pith.science/api/pith-number/VYNVNGKJA6NICG6NTSFRH5SSRY/graph.json","fetch_events":"https://pith.science/api/pith-number/VYNVNGKJA6NICG6NTSFRH5SSRY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VYNVNGKJA6NICG6NTSFRH5SSRY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VYNVNGKJA6NICG6NTSFRH5SSRY/action/storage_attestation","attest_author":"https://pith.science/pith/VYNVNGKJA6NICG6NTSFRH5SSRY/action/author_attestation","sign_citation":"https://pith.science/pith/VYNVNGKJA6NICG6NTSFRH5SSRY/action/citation_signature","submit_replication":"https://pith.science/pith/VYNVNGKJA6NICG6NTSFRH5SSRY/action/replication_record"}},"created_at":"2026-07-05T08:00:00.327132+00:00","updated_at":"2026-07-05T08:00:00.327132+00:00"}