{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:M3RYMVRCBAD7GIZSOYIPDL7UDP","short_pith_number":"pith:M3RYMVRC","schema_version":"1.0","canonical_sha256":"66e38656220807f323327610f1aff41bd7f9f66e604a6f719cf230f91646f491","source":{"kind":"arxiv","id":"2106.06239","version":1},"attestation_state":"computed","paper":{"title":"Safe Reinforcement Learning with Linear Function Approximation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Christos Thrampoulidis, Lin F. Yang, Sanae Amani","submitted_at":"2021-06-11T08:46:57Z","abstract_excerpt":"Safety in reinforcement learning has become increasingly important in recent years. Yet, existing solutions either fail to strictly avoid choosing unsafe actions, which may lead to catastrophic results in safety-critical systems, or fail to provide regret guarantees for settings where safety constraints need to be learned. In this paper, we address both problems by first modeling safety as an unknown linear cost function of states and actions, which must always fall below a certain threshold. We then present algorithms, termed SLUCB-QVI and RSLUCB-QVI, for episodic Markov decision processes (M"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.06239","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-11T08:46:57Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"82fc0356f982f09a6f927bed866c12d60ebe5c3ac8cd029f7c5dcc728b57c335","abstract_canon_sha256":"b8d3d4e0e240283ae5767e382b551617ecfdcfad523cc4b18b28fbb7e48e4b88"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:27.612428Z","signature_b64":"xXqOimoU1urrIrK1Ir+mcb4ZUJ43MTzmhdDfEodF0CVFx8XF6/leXrbMDoZwhyBt4j16CYiEqaXQNfMrNVspCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"66e38656220807f323327610f1aff41bd7f9f66e604a6f719cf230f91646f491","last_reissued_at":"2026-07-05T02:48:27.611938Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:27.611938Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safe Reinforcement Learning with Linear Function Approximation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Christos Thrampoulidis, Lin F. Yang, Sanae Amani","submitted_at":"2021-06-11T08:46:57Z","abstract_excerpt":"Safety in reinforcement learning has become increasingly important in recent years. Yet, existing solutions either fail to strictly avoid choosing unsafe actions, which may lead to catastrophic results in safety-critical systems, or fail to provide regret guarantees for settings where safety constraints need to be learned. In this paper, we address both problems by first modeling safety as an unknown linear cost function of states and actions, which must always fall below a certain threshold. We then present algorithms, termed SLUCB-QVI and RSLUCB-QVI, for episodic Markov decision processes (M"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.06239","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.06239/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.06239","created_at":"2026-07-05T02:48:27.611998+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.06239v1","created_at":"2026-07-05T02:48:27.611998+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.06239","created_at":"2026-07-05T02:48:27.611998+00:00"},{"alias_kind":"pith_short_12","alias_value":"M3RYMVRCBAD7","created_at":"2026-07-05T02:48:27.611998+00:00"},{"alias_kind":"pith_short_16","alias_value":"M3RYMVRCBAD7GIZS","created_at":"2026-07-05T02:48:27.611998+00:00"},{"alias_kind":"pith_short_8","alias_value":"M3RYMVRC","created_at":"2026-07-05T02:48:27.611998+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.07101","citing_title":"Constrained Online Decision-Making: A Unified Framework","ref_index":5,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M3RYMVRCBAD7GIZSOYIPDL7UDP","json":"https://pith.science/pith/M3RYMVRCBAD7GIZSOYIPDL7UDP.json","graph_json":"https://pith.science/api/pith-number/M3RYMVRCBAD7GIZSOYIPDL7UDP/graph.json","events_json":"https://pith.science/api/pith-number/M3RYMVRCBAD7GIZSOYIPDL7UDP/events.json","paper":"https://pith.science/paper/M3RYMVRC"},"agent_actions":{"view_html":"https://pith.science/pith/M3RYMVRCBAD7GIZSOYIPDL7UDP","download_json":"https://pith.science/pith/M3RYMVRCBAD7GIZSOYIPDL7UDP.json","view_paper":"https://pith.science/paper/M3RYMVRC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.06239&json=true","fetch_graph":"https://pith.science/api/pith-number/M3RYMVRCBAD7GIZSOYIPDL7UDP/graph.json","fetch_events":"https://pith.science/api/pith-number/M3RYMVRCBAD7GIZSOYIPDL7UDP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M3RYMVRCBAD7GIZSOYIPDL7UDP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M3RYMVRCBAD7GIZSOYIPDL7UDP/action/storage_attestation","attest_author":"https://pith.science/pith/M3RYMVRCBAD7GIZSOYIPDL7UDP/action/author_attestation","sign_citation":"https://pith.science/pith/M3RYMVRCBAD7GIZSOYIPDL7UDP/action/citation_signature","submit_replication":"https://pith.science/pith/M3RYMVRCBAD7GIZSOYIPDL7UDP/action/replication_record"}},"created_at":"2026-07-05T02:48:27.611998+00:00","updated_at":"2026-07-05T02:48:27.611998+00:00"}