{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GHHMP2MXG6GPDAN4TXBOFCJ222","short_pith_number":"pith:GHHMP2MX","schema_version":"1.0","canonical_sha256":"31cec7e997378cf181bc9dc2e2893ad693ba25c0862907c24020066652841633","source":{"kind":"arxiv","id":"2402.07875","version":2},"attestation_state":"computed","paper":{"title":"Implicit Bias of Policy Gradient in Linear Quadratic Control: Extrapolation to Unseen Initial States","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY","stat.ML"],"primary_cat":"cs.LG","authors_text":"Amir Globerson, Edo Cohen-Karlik, Nadav Cohen, Noam Razin, Raja Giryes, Yotam Alexander","submitted_at":"2024-02-12T18:41:31Z","abstract_excerpt":"In modern machine learning, models can often fit training data in numerous ways, some of which perform well on unseen (test) data, while others do not. Remarkably, in such cases gradient descent frequently exhibits an implicit bias that leads to excellent performance on unseen data. This implicit bias was extensively studied in supervised learning, but is far less understood in optimal control (reinforcement learning). There, learning a controller applied to a system via gradient descent is known as policy gradient, and a question of prime importance is the extent to which a learned controller"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.07875","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-12T18:41:31Z","cross_cats_sorted":["cs.AI","cs.SY","eess.SY","stat.ML"],"title_canon_sha256":"9608743fabe30f5ba33b9341e683c97f959cbcd2eb916a840526641c0d1c61cc","abstract_canon_sha256":"26c911bd41a91a1ecefacf735cb4964ea20372acde39e788e3b81767d685ca9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:52.382778Z","signature_b64":"CPj0nF+hYJmwtl/8hnlEkzY1cZTgTEGMZDsYHI3tk/kz3qhEYDOKfzb6GIlpmcdsDYR0M1RQuW9TZhZx0J19AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"31cec7e997378cf181bc9dc2e2893ad693ba25c0862907c24020066652841633","last_reissued_at":"2026-07-05T08:25:52.382215Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:52.382215Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Implicit Bias of Policy Gradient in Linear Quadratic Control: Extrapolation to Unseen Initial States","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY","stat.ML"],"primary_cat":"cs.LG","authors_text":"Amir Globerson, Edo Cohen-Karlik, Nadav Cohen, Noam Razin, Raja Giryes, Yotam Alexander","submitted_at":"2024-02-12T18:41:31Z","abstract_excerpt":"In modern machine learning, models can often fit training data in numerous ways, some of which perform well on unseen (test) data, while others do not. Remarkably, in such cases gradient descent frequently exhibits an implicit bias that leads to excellent performance on unseen data. This implicit bias was extensively studied in supervised learning, but is far less understood in optimal control (reinforcement learning). There, learning a controller applied to a system via gradient descent is known as policy gradient, and a question of prime importance is the extent to which a learned controller"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.07875","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.07875/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.07875","created_at":"2026-07-05T08:25:52.382278+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.07875v2","created_at":"2026-07-05T08:25:52.382278+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.07875","created_at":"2026-07-05T08:25:52.382278+00:00"},{"alias_kind":"pith_short_12","alias_value":"GHHMP2MXG6GP","created_at":"2026-07-05T08:25:52.382278+00:00"},{"alias_kind":"pith_short_16","alias_value":"GHHMP2MXG6GPDAN4","created_at":"2026-07-05T08:25:52.382278+00:00"},{"alias_kind":"pith_short_8","alias_value":"GHHMP2MX","created_at":"2026-07-05T08:25:52.382278+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.08436","citing_title":"Toward Optimal Statistical Inference in Noisy Linear Quadratic Reinforcement Learning over a Finite Horizon","ref_index":38,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GHHMP2MXG6GPDAN4TXBOFCJ222","json":"https://pith.science/pith/GHHMP2MXG6GPDAN4TXBOFCJ222.json","graph_json":"https://pith.science/api/pith-number/GHHMP2MXG6GPDAN4TXBOFCJ222/graph.json","events_json":"https://pith.science/api/pith-number/GHHMP2MXG6GPDAN4TXBOFCJ222/events.json","paper":"https://pith.science/paper/GHHMP2MX"},"agent_actions":{"view_html":"https://pith.science/pith/GHHMP2MXG6GPDAN4TXBOFCJ222","download_json":"https://pith.science/pith/GHHMP2MXG6GPDAN4TXBOFCJ222.json","view_paper":"https://pith.science/paper/GHHMP2MX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.07875&json=true","fetch_graph":"https://pith.science/api/pith-number/GHHMP2MXG6GPDAN4TXBOFCJ222/graph.json","fetch_events":"https://pith.science/api/pith-number/GHHMP2MXG6GPDAN4TXBOFCJ222/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GHHMP2MXG6GPDAN4TXBOFCJ222/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GHHMP2MXG6GPDAN4TXBOFCJ222/action/storage_attestation","attest_author":"https://pith.science/pith/GHHMP2MXG6GPDAN4TXBOFCJ222/action/author_attestation","sign_citation":"https://pith.science/pith/GHHMP2MXG6GPDAN4TXBOFCJ222/action/citation_signature","submit_replication":"https://pith.science/pith/GHHMP2MXG6GPDAN4TXBOFCJ222/action/replication_record"}},"created_at":"2026-07-05T08:25:52.382278+00:00","updated_at":"2026-07-05T08:25:52.382278+00:00"}