{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:RXYOBC6DR7HVWWK5TH5ZW6CTZJ","short_pith_number":"pith:RXYOBC6D","schema_version":"1.0","canonical_sha256":"8df0e08bc38fcf5b595d99fb9b7853ca70d4627d5db9cb7e2e010ad6cb97881e","source":{"kind":"arxiv","id":"1910.08412","version":3},"attestation_state":"computed","paper":{"title":"On the Sample Complexity of Actor-Critic Method for Reinforcement Learning with Function Approximation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alec Koppel, Alejandro Ribeiro, Harshat Kumar","submitted_at":"2019-10-18T13:33:17Z","abstract_excerpt":"Reinforcement learning, mathematically described by Markov Decision Problems, may be approached either through dynamic programming or policy search. Actor-critic algorithms combine the merits of both approaches by alternating between steps to estimate the value function and policy gradient updates. Due to the fact that the updates exhibit correlated noise and biased gradient updates, only the asymptotic behavior of actor-critic is known by connecting its behavior to dynamical systems. This work puts forth a new variant of actor-critic that employs Monte Carlo rollouts during the policy search "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.08412","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-10-18T13:33:17Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"36f5b1af32264b7c45fabc29c95461c4b0ca3a8a57c6005bd36ab4fec02661cb","abstract_canon_sha256":"8fc9d800c557577babc84602d12c390222793e55faf29ff43f047cd01b345f02"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:36:28.210010Z","signature_b64":"pOJzElZim0X1olcY27j3MMp+DOKJnyZY7LXw5enmuL6MskKzcPuzHmb/UTQDokuGj63Dyl3Ohbi0Z6h97iIEAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8df0e08bc38fcf5b595d99fb9b7853ca70d4627d5db9cb7e2e010ad6cb97881e","last_reissued_at":"2026-07-05T05:36:28.209529Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:36:28.209529Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Sample Complexity of Actor-Critic Method for Reinforcement Learning with Function Approximation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alec Koppel, Alejandro Ribeiro, Harshat Kumar","submitted_at":"2019-10-18T13:33:17Z","abstract_excerpt":"Reinforcement learning, mathematically described by Markov Decision Problems, may be approached either through dynamic programming or policy search. Actor-critic algorithms combine the merits of both approaches by alternating between steps to estimate the value function and policy gradient updates. Due to the fact that the updates exhibit correlated noise and biased gradient updates, only the asymptotic behavior of actor-critic is known by connecting its behavior to dynamical systems. This work puts forth a new variant of actor-critic that employs Monte Carlo rollouts during the policy search "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.08412","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.08412/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.08412","created_at":"2026-07-05T05:36:28.209587+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.08412v3","created_at":"2026-07-05T05:36:28.209587+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.08412","created_at":"2026-07-05T05:36:28.209587+00:00"},{"alias_kind":"pith_short_12","alias_value":"RXYOBC6DR7HV","created_at":"2026-07-05T05:36:28.209587+00:00"},{"alias_kind":"pith_short_16","alias_value":"RXYOBC6DR7HVWWK5","created_at":"2026-07-05T05:36:28.209587+00:00"},{"alias_kind":"pith_short_8","alias_value":"RXYOBC6D","created_at":"2026-07-05T05:36:28.209587+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05863","citing_title":"Strategic Bargaining in Multi-Buyer Markets: Reinforcement Learning from Verifiable Rewards for LLM Negotiations","ref_index":172,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22579","citing_title":"Stationary Robust Mean-Field Games under Model Mismatches","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01505","citing_title":"Optimal Sample Complexity for Single Time-Scale Actor-Critic with Momentum","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RXYOBC6DR7HVWWK5TH5ZW6CTZJ","json":"https://pith.science/pith/RXYOBC6DR7HVWWK5TH5ZW6CTZJ.json","graph_json":"https://pith.science/api/pith-number/RXYOBC6DR7HVWWK5TH5ZW6CTZJ/graph.json","events_json":"https://pith.science/api/pith-number/RXYOBC6DR7HVWWK5TH5ZW6CTZJ/events.json","paper":"https://pith.science/paper/RXYOBC6D"},"agent_actions":{"view_html":"https://pith.science/pith/RXYOBC6DR7HVWWK5TH5ZW6CTZJ","download_json":"https://pith.science/pith/RXYOBC6DR7HVWWK5TH5ZW6CTZJ.json","view_paper":"https://pith.science/paper/RXYOBC6D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.08412&json=true","fetch_graph":"https://pith.science/api/pith-number/RXYOBC6DR7HVWWK5TH5ZW6CTZJ/graph.json","fetch_events":"https://pith.science/api/pith-number/RXYOBC6DR7HVWWK5TH5ZW6CTZJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RXYOBC6DR7HVWWK5TH5ZW6CTZJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RXYOBC6DR7HVWWK5TH5ZW6CTZJ/action/storage_attestation","attest_author":"https://pith.science/pith/RXYOBC6DR7HVWWK5TH5ZW6CTZJ/action/author_attestation","sign_citation":"https://pith.science/pith/RXYOBC6DR7HVWWK5TH5ZW6CTZJ/action/citation_signature","submit_replication":"https://pith.science/pith/RXYOBC6DR7HVWWK5TH5ZW6CTZJ/action/replication_record"}},"created_at":"2026-07-05T05:36:28.209587+00:00","updated_at":"2026-07-05T05:36:28.209587+00:00"}