{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:UHRRQKBEDKP643C52EVGXCDKEZ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d852ea9818e024fe4833092d5c8e0d6f042e2c40c7d719441edfb120caf8a80e","cross_cats_sorted":["math.OC","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-08-02T14:01:49Z","title_canon_sha256":"ab02f9223ba8c4df81d093b5b647b1e67b9218b0049f2685bdca67bf8bde2100"},"schema_version":"1.0","source":{"id":"2008.00483","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2008.00483","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"arxiv_version","alias_value":"2008.00483v2","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2008.00483","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"pith_short_12","alias_value":"UHRRQKBEDKP6","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"pith_short_16","alias_value":"UHRRQKBEDKP643C5","created_at":"2026-07-05T02:48:33Z"},{"alias_kind":"pith_short_8","alias_value":"UHRRQKBE","created_at":"2026-07-05T02:48:33Z"}],"graph_snapshots":[{"event_id":"sha256:1426385e007a8640279748c68af924080545568f057b59648fc76db355ba17dd","target":"graph","created_at":"2026-07-05T02:48:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2008.00483/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We study the global convergence and global optimality of actor-critic, one of the most popular families of reinforcement learning algorithms. While most existing works on actor-critic employ bi-level or two-timescale updates, we focus on the more practical single-timescale setting, where the actor and critic are updated simultaneously. Specifically, in each iteration, the critic update is obtained by applying the Bellman evaluation operator only once while the actor is updated in the policy gradient direction computed using the critic. Moreover, we consider two function approximation settings ","authors_text":"Zhaoran Wang, Zhuoran Yang, Zuyue Fu","cross_cats":["math.OC","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-08-02T14:01:49Z","title":"Single-Timescale Actor-Critic Provably Finds Globally Optimal Policy"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2008.00483","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:552fa4c78a8341058b317f7738d7105659a58561c2dda3ce2f1e7fd56bc0863f","target":"record","created_at":"2026-07-05T02:48:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d852ea9818e024fe4833092d5c8e0d6f042e2c40c7d719441edfb120caf8a80e","cross_cats_sorted":["math.OC","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-08-02T14:01:49Z","title_canon_sha256":"ab02f9223ba8c4df81d093b5b647b1e67b9218b0049f2685bdca67bf8bde2100"},"schema_version":"1.0","source":{"id":"2008.00483","kind":"arxiv","version":2}},"canonical_sha256":"a1e31828241a9fee6c5dd12a6b886a2651ad30c20118c85d9904d3347a912ce3","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"a1e31828241a9fee6c5dd12a6b886a2651ad30c20118c85d9904d3347a912ce3","first_computed_at":"2026-07-05T02:48:33.709411Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:48:33.709411Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"WzFsKnz34JufBqHi6YMa+Ozt2ZtH5uEZ9fXgTtDXvBvFcmDSAx7UPkYqLUqkb3Rg92rNUcC5CKPkFlo18/ajAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T02:48:33.709847Z","signed_message":"canonical_sha256_bytes"},"source_id":"2008.00483","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:552fa4c78a8341058b317f7738d7105659a58561c2dda3ce2f1e7fd56bc0863f","sha256:1426385e007a8640279748c68af924080545568f057b59648fc76db355ba17dd"],"state_sha256":"7aee616a7115203943ebc1ac3fa9f64942fe1c0aee16bbb01befdb48c156e524"}