{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:D2VN3U64HKMZRCI5FLZNUBFI2K","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e0c155fdb559479b8be12ad463938a9bec07a22b9995ddbb5774660f9aa66c0d","cross_cats_sorted":["math.OC"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-07-13T06:03:15Z","title_canon_sha256":"92f300ea9682487535c54d00025fa3c55129e1fbcd0fc8cdf6d2f9946a1b02bb"},"schema_version":"1.0","source":{"id":"2007.06202","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2007.06202","created_at":"2026-07-05T01:18:11Z"},{"alias_kind":"arxiv_version","alias_value":"2007.06202v1","created_at":"2026-07-05T01:18:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.06202","created_at":"2026-07-05T01:18:11Z"},{"alias_kind":"pith_short_12","alias_value":"D2VN3U64HKMZ","created_at":"2026-07-05T01:18:11Z"},{"alias_kind":"pith_short_16","alias_value":"D2VN3U64HKMZRCI5","created_at":"2026-07-05T01:18:11Z"},{"alias_kind":"pith_short_8","alias_value":"D2VN3U64","created_at":"2026-07-05T01:18:11Z"}],"graph_snapshots":[{"event_id":"sha256:8dad3d6cfe10c01f225c1afcc82a3ee5d0080a7e34ff925daad8140a8a64c647","target":"graph","created_at":"2026-07-05T01:18:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2007.06202/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Linear quadratic regulator (LQR) is one of the most popular frameworks to tackle continuous Markov decision process tasks. With its fundamental theory and tractable optimal policy, LQR has been revisited and analyzed in recent years, in terms of reinforcement learning scenarios such as the model-free or model-based setting. In this paper, we introduce the \\textit{Structured Policy Iteration} (S-PI) for LQR, a method capable of deriving a structured linear policy. Such a structured policy with (block) sparsity or low-rank can have significant advantages over the standard LQR policy: more interp","authors_text":"Gang Wu, Handong Zhao, Ryan A. Rossi, Youngsuk Park, Zheng Wen","cross_cats":["math.OC"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-07-13T06:03:15Z","title":"Structured Policy Iteration for Linear Quadratic Regulator"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.06202","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:16ce86f78c8a0042d39e3fb69885259992101d805f4d82897bda0b82d82388cb","target":"record","created_at":"2026-07-05T01:18:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e0c155fdb559479b8be12ad463938a9bec07a22b9995ddbb5774660f9aa66c0d","cross_cats_sorted":["math.OC"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-07-13T06:03:15Z","title_canon_sha256":"92f300ea9682487535c54d00025fa3c55129e1fbcd0fc8cdf6d2f9946a1b02bb"},"schema_version":"1.0","source":{"id":"2007.06202","kind":"arxiv","version":1}},"canonical_sha256":"1eaaddd3dc3a9998891d2af2da04a8d2a65c4d5b68b7ad3a053e29d1719c3605","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1eaaddd3dc3a9998891d2af2da04a8d2a65c4d5b68b7ad3a053e29d1719c3605","first_computed_at":"2026-07-05T01:18:11.004985Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T01:18:11.004985Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"AhW1iZ8hIyOSEV1DX8RfieC3XvpM2Yh76IdHfjj7cmvC3LHXx/z4cTCOzmE6qZg1AAdnFiSJGkW1xu6h1N6XCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T01:18:11.005343Z","signed_message":"canonical_sha256_bytes"},"source_id":"2007.06202","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:16ce86f78c8a0042d39e3fb69885259992101d805f4d82897bda0b82d82388cb","sha256:8dad3d6cfe10c01f225c1afcc82a3ee5d0080a7e34ff925daad8140a8a64c647"],"state_sha256":"643006911828ea96c0856c15278c2fff997d7bb567ec6e0fbbd2203516d96993"}