{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:LSMGYQCDT3YLZNJ7YBO7OWQQGO","short_pith_number":"pith:LSMGYQCD","schema_version":"1.0","canonical_sha256":"5c986c40439ef0bcb53fc05df75a10338c833e6b2a9d516cfa89102331203663","source":{"kind":"arxiv","id":"2202.07960","version":3},"attestation_state":"computed","paper":{"title":"Temporal Difference Learning with Continuous Time and State in the Stochastic Setting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.AP","math.OC"],"primary_cat":"cs.LG","authors_text":"DI-ENS, Francis Bach (SIERRA, PSL), Ziad Kobeissi (SIERRA)","submitted_at":"2022-02-16T10:10:53Z","abstract_excerpt":"We consider the problem of continuous-time policy evaluation. This consists in learning through observations the value function associated with an uncontrolled continuous-time stochastic dynamic and a reward function. We propose two original variants of the well-known TD(0) method using vanishing time steps. One is model-free and the other is model-based. For both methods, we prove theoretical convergence rates that we subsequently verify through numerical simulations. Alternatively, those methods can be interpreted as novel reinforcement learning approaches for approximating solutions of line"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.07960","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-16T10:10:53Z","cross_cats_sorted":["cs.AI","math.AP","math.OC"],"title_canon_sha256":"24180fb7590fca8061270e8dcccfff81043565e719f71be5784317b40157487b","abstract_canon_sha256":"4e527343bd18a8736b10bea1c1a0cd8e27524e8f6304147b80aca3eb4685a9ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:18:15.677958Z","signature_b64":"sqJpk6vNgHkBzfG7K11qvLfBOgfA52MqHOsRrLsJUSdCv3LbLITkvg/UnMc848i4N46z/+YCHI/+vOgXaQrjBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5c986c40439ef0bcb53fc05df75a10338c833e6b2a9d516cfa89102331203663","last_reissued_at":"2026-07-05T06:18:15.677412Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:18:15.677412Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Temporal Difference Learning with Continuous Time and State in the Stochastic Setting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.AP","math.OC"],"primary_cat":"cs.LG","authors_text":"DI-ENS, Francis Bach (SIERRA, PSL), Ziad Kobeissi (SIERRA)","submitted_at":"2022-02-16T10:10:53Z","abstract_excerpt":"We consider the problem of continuous-time policy evaluation. This consists in learning through observations the value function associated with an uncontrolled continuous-time stochastic dynamic and a reward function. We propose two original variants of the well-known TD(0) method using vanishing time steps. One is model-free and the other is model-based. For both methods, we prove theoretical convergence rates that we subsequently verify through numerical simulations. Alternatively, those methods can be interpreted as novel reinforcement learning approaches for approximating solutions of line"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.07960","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.07960/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.07960","created_at":"2026-07-05T06:18:15.677475+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.07960v3","created_at":"2026-07-05T06:18:15.677475+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.07960","created_at":"2026-07-05T06:18:15.677475+00:00"},{"alias_kind":"pith_short_12","alias_value":"LSMGYQCDT3YL","created_at":"2026-07-05T06:18:15.677475+00:00"},{"alias_kind":"pith_short_16","alias_value":"LSMGYQCDT3YLZNJ7","created_at":"2026-07-05T06:18:15.677475+00:00"},{"alias_kind":"pith_short_8","alias_value":"LSMGYQCD","created_at":"2026-07-05T06:18:15.677475+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05967","citing_title":"Fast and Robust Convergence Rate for TD(0) with Linear Function Approximation, Universal Learning Steps and I.I.D. Samples","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LSMGYQCDT3YLZNJ7YBO7OWQQGO","json":"https://pith.science/pith/LSMGYQCDT3YLZNJ7YBO7OWQQGO.json","graph_json":"https://pith.science/api/pith-number/LSMGYQCDT3YLZNJ7YBO7OWQQGO/graph.json","events_json":"https://pith.science/api/pith-number/LSMGYQCDT3YLZNJ7YBO7OWQQGO/events.json","paper":"https://pith.science/paper/LSMGYQCD"},"agent_actions":{"view_html":"https://pith.science/pith/LSMGYQCDT3YLZNJ7YBO7OWQQGO","download_json":"https://pith.science/pith/LSMGYQCDT3YLZNJ7YBO7OWQQGO.json","view_paper":"https://pith.science/paper/LSMGYQCD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.07960&json=true","fetch_graph":"https://pith.science/api/pith-number/LSMGYQCDT3YLZNJ7YBO7OWQQGO/graph.json","fetch_events":"https://pith.science/api/pith-number/LSMGYQCDT3YLZNJ7YBO7OWQQGO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LSMGYQCDT3YLZNJ7YBO7OWQQGO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LSMGYQCDT3YLZNJ7YBO7OWQQGO/action/storage_attestation","attest_author":"https://pith.science/pith/LSMGYQCDT3YLZNJ7YBO7OWQQGO/action/author_attestation","sign_citation":"https://pith.science/pith/LSMGYQCDT3YLZNJ7YBO7OWQQGO/action/citation_signature","submit_replication":"https://pith.science/pith/LSMGYQCDT3YLZNJ7YBO7OWQQGO/action/replication_record"}},"created_at":"2026-07-05T06:18:15.677475+00:00","updated_at":"2026-07-05T06:18:15.677475+00:00"}