{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VZXMOOWEERYJIOD27ZS2BXRQ3G","short_pith_number":"pith:VZXMOOWE","schema_version":"1.0","canonical_sha256":"ae6ec73ac4247094387afe65a0de30d9b506a3106b6323cfe4af21d372183981","source":{"kind":"arxiv","id":"2505.03706","version":2},"attestation_state":"computed","paper":{"title":"Policy Gradient Adaptive Control for the LQR: Indirect and Direct Approaches","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"math.OC","authors_text":"Alessandro Chiuso, Feiran Zhao, Florian D\\\"orfler","submitted_at":"2025-05-06T17:26:04Z","abstract_excerpt":"Motivated by recent advances of reinforcement learning and direct data-driven control, we propose policy gradient adaptive control (PGAC) for the linear quadratic regulator (LQR), which uses online closed-loop data to improve the control policy while maintaining stability. Our method adaptively updates the policy in feedback by descending the gradient of the LQR cost and is categorized as indirect, when gradients are computed via an estimated model, versus direct, when gradients are derived from data using sample covariance parameterization. Beyond the vanilla gradient, we also showcase the me"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.03706","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"math.OC","submitted_at":"2025-05-06T17:26:04Z","cross_cats_sorted":["cs.SY","eess.SY"],"title_canon_sha256":"0cfe33aed724b074a739d623b5f0dd973ef368fed901e798905a00fbab325a95","abstract_canon_sha256":"bd13513e1595e77946e74db4d55e0e301b80acd3ef45f0a1800d5e1d4d58dfa9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:53.902239Z","signature_b64":"VQHQo6GodsVKxlF/FlgRNwg5VJ0unFijyPGR3mhPfk4xE1gbAHFR2iU5rSg1tjB0DVnKbljjuG8FTx0daJJgDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae6ec73ac4247094387afe65a0de30d9b506a3106b6323cfe4af21d372183981","last_reissued_at":"2026-07-05T11:20:53.901777Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:53.901777Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Policy Gradient Adaptive Control for the LQR: Indirect and Direct Approaches","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"math.OC","authors_text":"Alessandro Chiuso, Feiran Zhao, Florian D\\\"orfler","submitted_at":"2025-05-06T17:26:04Z","abstract_excerpt":"Motivated by recent advances of reinforcement learning and direct data-driven control, we propose policy gradient adaptive control (PGAC) for the linear quadratic regulator (LQR), which uses online closed-loop data to improve the control policy while maintaining stability. Our method adaptively updates the policy in feedback by descending the gradient of the LQR cost and is categorized as indirect, when gradients are computed via an estimated model, versus direct, when gradients are derived from data using sample covariance parameterization. Beyond the vanilla gradient, we also showcase the me"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.03706","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.03706/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.03706","created_at":"2026-07-05T11:20:53.901831+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.03706v2","created_at":"2026-07-05T11:20:53.901831+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.03706","created_at":"2026-07-05T11:20:53.901831+00:00"},{"alias_kind":"pith_short_12","alias_value":"VZXMOOWEERYJ","created_at":"2026-07-05T11:20:53.901831+00:00"},{"alias_kind":"pith_short_16","alias_value":"VZXMOOWEERYJIOD2","created_at":"2026-07-05T11:20:53.901831+00:00"},{"alias_kind":"pith_short_8","alias_value":"VZXMOOWE","created_at":"2026-07-05T11:20:53.901831+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00644","citing_title":"A Data-Enabled Primal-Dual Approach for Policy Learning with SDP Formulations","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15563","citing_title":"Direct Data-Driven Linear Quadratic Tracking via Policy Optimization","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2511.08236","citing_title":"Stability of Certainty-Equivalent Adaptive LQR for Linear Systems with Unknown Time-Varying Parameters","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03764","citing_title":"Sample-Efficient Model-Free Policy Gradient Methods for Stochastic LQR via Robust Linear Regression","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22138","citing_title":"Global Convergence of Policy Gradient Methods for ReLU Controllers in Linear Quadratic Regulation","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VZXMOOWEERYJIOD27ZS2BXRQ3G","json":"https://pith.science/pith/VZXMOOWEERYJIOD27ZS2BXRQ3G.json","graph_json":"https://pith.science/api/pith-number/VZXMOOWEERYJIOD27ZS2BXRQ3G/graph.json","events_json":"https://pith.science/api/pith-number/VZXMOOWEERYJIOD27ZS2BXRQ3G/events.json","paper":"https://pith.science/paper/VZXMOOWE"},"agent_actions":{"view_html":"https://pith.science/pith/VZXMOOWEERYJIOD27ZS2BXRQ3G","download_json":"https://pith.science/pith/VZXMOOWEERYJIOD27ZS2BXRQ3G.json","view_paper":"https://pith.science/paper/VZXMOOWE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.03706&json=true","fetch_graph":"https://pith.science/api/pith-number/VZXMOOWEERYJIOD27ZS2BXRQ3G/graph.json","fetch_events":"https://pith.science/api/pith-number/VZXMOOWEERYJIOD27ZS2BXRQ3G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VZXMOOWEERYJIOD27ZS2BXRQ3G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VZXMOOWEERYJIOD27ZS2BXRQ3G/action/storage_attestation","attest_author":"https://pith.science/pith/VZXMOOWEERYJIOD27ZS2BXRQ3G/action/author_attestation","sign_citation":"https://pith.science/pith/VZXMOOWEERYJIOD27ZS2BXRQ3G/action/citation_signature","submit_replication":"https://pith.science/pith/VZXMOOWEERYJIOD27ZS2BXRQ3G/action/replication_record"}},"created_at":"2026-07-05T11:20:53.901831+00:00","updated_at":"2026-07-05T11:20:53.901831+00:00"}