{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CUT6FJJ3I34BNLFNMOUQHGGITT","short_pith_number":"pith:CUT6FJJ3","schema_version":"1.0","canonical_sha256":"1527e2a53b46f816acad63a90398c89cc1e7a9e9599b52c7de6edb049a6ae74e","source":{"kind":"arxiv","id":"2311.03651","version":1},"attestation_state":"computed","paper":{"title":"SeRO: Self-Supervised Reinforcement Learning for Recovery from Out-of-Distribution Situations","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Chan Kim, Christophe Bobda, Jaekyung Cho, Seong-Woo Kim, Seung-Woo Seo","submitted_at":"2023-11-07T01:42:13Z","abstract_excerpt":"Robotic agents trained using reinforcement learning have the problem of taking unreliable actions in an out-of-distribution (OOD) state. Agents can easily become OOD in real-world environments because it is almost impossible for them to visit and learn the entire state space during training. Unfortunately, unreliable actions do not ensure that agents perform their original tasks successfully. Therefore, agents should be able to recognize whether they are in OOD states and learn how to return to the learned state distribution rather than continue to take unreliable actions. In this study, we pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.03651","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-07T01:42:13Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"ba7ee8245e00f3a82bb3bbf3d60af7fb4fcc17beda60467d5f358cbfebd0b9c7","abstract_canon_sha256":"5ba49532787e3fecb0bd44b9f29f90d807c10d9b439156913c9172a941967341"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:10:02.107906Z","signature_b64":"mBIqNGKOV7GdYXqUQbbBYBhGEXxc0ozzJw2FY97mnbuY3E2BNvsAAXozSiZRTyaZKzdh+0aBuTWHTWJrUsEDDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1527e2a53b46f816acad63a90398c89cc1e7a9e9599b52c7de6edb049a6ae74e","last_reissued_at":"2026-07-05T07:10:02.107455Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:10:02.107455Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SeRO: Self-Supervised Reinforcement Learning for Recovery from Out-of-Distribution Situations","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Chan Kim, Christophe Bobda, Jaekyung Cho, Seong-Woo Kim, Seung-Woo Seo","submitted_at":"2023-11-07T01:42:13Z","abstract_excerpt":"Robotic agents trained using reinforcement learning have the problem of taking unreliable actions in an out-of-distribution (OOD) state. Agents can easily become OOD in real-world environments because it is almost impossible for them to visit and learn the entire state space during training. Unfortunately, unreliable actions do not ensure that agents perform their original tasks successfully. Therefore, agents should be able to recognize whether they are in OOD states and learn how to return to the learned state distribution rather than continue to take unreliable actions. In this study, we pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.03651","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.03651/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.03651","created_at":"2026-07-05T07:10:02.107515+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.03651v1","created_at":"2026-07-05T07:10:02.107515+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.03651","created_at":"2026-07-05T07:10:02.107515+00:00"},{"alias_kind":"pith_short_12","alias_value":"CUT6FJJ3I34B","created_at":"2026-07-05T07:10:02.107515+00:00"},{"alias_kind":"pith_short_16","alias_value":"CUT6FJJ3I34BNLFN","created_at":"2026-07-05T07:10:02.107515+00:00"},{"alias_kind":"pith_short_8","alias_value":"CUT6FJJ3","created_at":"2026-07-05T07:10:02.107515+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CUT6FJJ3I34BNLFNMOUQHGGITT","json":"https://pith.science/pith/CUT6FJJ3I34BNLFNMOUQHGGITT.json","graph_json":"https://pith.science/api/pith-number/CUT6FJJ3I34BNLFNMOUQHGGITT/graph.json","events_json":"https://pith.science/api/pith-number/CUT6FJJ3I34BNLFNMOUQHGGITT/events.json","paper":"https://pith.science/paper/CUT6FJJ3"},"agent_actions":{"view_html":"https://pith.science/pith/CUT6FJJ3I34BNLFNMOUQHGGITT","download_json":"https://pith.science/pith/CUT6FJJ3I34BNLFNMOUQHGGITT.json","view_paper":"https://pith.science/paper/CUT6FJJ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.03651&json=true","fetch_graph":"https://pith.science/api/pith-number/CUT6FJJ3I34BNLFNMOUQHGGITT/graph.json","fetch_events":"https://pith.science/api/pith-number/CUT6FJJ3I34BNLFNMOUQHGGITT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CUT6FJJ3I34BNLFNMOUQHGGITT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CUT6FJJ3I34BNLFNMOUQHGGITT/action/storage_attestation","attest_author":"https://pith.science/pith/CUT6FJJ3I34BNLFNMOUQHGGITT/action/author_attestation","sign_citation":"https://pith.science/pith/CUT6FJJ3I34BNLFNMOUQHGGITT/action/citation_signature","submit_replication":"https://pith.science/pith/CUT6FJJ3I34BNLFNMOUQHGGITT/action/replication_record"}},"created_at":"2026-07-05T07:10:02.107515+00:00","updated_at":"2026-07-05T07:10:02.107515+00:00"}