{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QXW2QZOECMEBTDEDZ3BESPM3R6","short_pith_number":"pith:QXW2QZOE","schema_version":"1.0","canonical_sha256":"85eda865c41308198c83cec2493d9b8f90c2516fc8504bc25ab9d1a66903d8c9","source":{"kind":"arxiv","id":"2304.12090","version":2},"attestation_state":"computed","paper":{"title":"Reinforcement Learning with Knowledge Representation and Reasoning: A Brief Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chao Yu, Hankz Hankui Zhuo, Shicheng Ye","submitted_at":"2023-04-24T13:35:11Z","abstract_excerpt":"Reinforcement Learning (RL) has achieved tremendous development in recent years, but still faces significant obstacles in addressing complex real-life problems due to the issues of poor system generalization, low sample efficiency as well as safety and interpretability concerns. The core reason underlying such dilemmas can be attributed to the fact that most of the work has focused on the computational aspect of value functions or policies using a representational model to describe atomic components of rewards, states and actions etc, thus neglecting the rich high-level declarative domain know"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.12090","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-04-24T13:35:11Z","cross_cats_sorted":[],"title_canon_sha256":"f7a222217622a173df2cf413889eac47fd553e17c348510b171a239df032c7f3","abstract_canon_sha256":"3418a54b0e837a6425964c7948ba4e167be52af619063e9cffa7fc0da6f47ff4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:31.641579Z","signature_b64":"iBpS4WxKeG7+82bvzDzbrwS4dnMKAF6+/lg2UfOPNZC/FhejibULH/YgXHOip2JZPYt/BF2lgIU+Ww7fAN6zCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85eda865c41308198c83cec2493d9b8f90c2516fc8504bc25ab9d1a66903d8c9","last_reissued_at":"2026-07-05T10:18:31.641157Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:31.641157Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning with Knowledge Representation and Reasoning: A Brief Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chao Yu, Hankz Hankui Zhuo, Shicheng Ye","submitted_at":"2023-04-24T13:35:11Z","abstract_excerpt":"Reinforcement Learning (RL) has achieved tremendous development in recent years, but still faces significant obstacles in addressing complex real-life problems due to the issues of poor system generalization, low sample efficiency as well as safety and interpretability concerns. The core reason underlying such dilemmas can be attributed to the fact that most of the work has focused on the computational aspect of value functions or policies using a representational model to describe atomic components of rewards, states and actions etc, thus neglecting the rich high-level declarative domain know"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.12090","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.12090/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.12090","created_at":"2026-07-05T10:18:31.641211+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.12090v2","created_at":"2026-07-05T10:18:31.641211+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.12090","created_at":"2026-07-05T10:18:31.641211+00:00"},{"alias_kind":"pith_short_12","alias_value":"QXW2QZOECMEB","created_at":"2026-07-05T10:18:31.641211+00:00"},{"alias_kind":"pith_short_16","alias_value":"QXW2QZOECMEBTDED","created_at":"2026-07-05T10:18:31.641211+00:00"},{"alias_kind":"pith_short_8","alias_value":"QXW2QZOE","created_at":"2026-07-05T10:18:31.641211+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.15512","citing_title":"SALSA-RL: Stability Analysis in the Latent Space of Actions for Reinforcement Learning","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QXW2QZOECMEBTDEDZ3BESPM3R6","json":"https://pith.science/pith/QXW2QZOECMEBTDEDZ3BESPM3R6.json","graph_json":"https://pith.science/api/pith-number/QXW2QZOECMEBTDEDZ3BESPM3R6/graph.json","events_json":"https://pith.science/api/pith-number/QXW2QZOECMEBTDEDZ3BESPM3R6/events.json","paper":"https://pith.science/paper/QXW2QZOE"},"agent_actions":{"view_html":"https://pith.science/pith/QXW2QZOECMEBTDEDZ3BESPM3R6","download_json":"https://pith.science/pith/QXW2QZOECMEBTDEDZ3BESPM3R6.json","view_paper":"https://pith.science/paper/QXW2QZOE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.12090&json=true","fetch_graph":"https://pith.science/api/pith-number/QXW2QZOECMEBTDEDZ3BESPM3R6/graph.json","fetch_events":"https://pith.science/api/pith-number/QXW2QZOECMEBTDEDZ3BESPM3R6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QXW2QZOECMEBTDEDZ3BESPM3R6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QXW2QZOECMEBTDEDZ3BESPM3R6/action/storage_attestation","attest_author":"https://pith.science/pith/QXW2QZOECMEBTDEDZ3BESPM3R6/action/author_attestation","sign_citation":"https://pith.science/pith/QXW2QZOECMEBTDEDZ3BESPM3R6/action/citation_signature","submit_replication":"https://pith.science/pith/QXW2QZOECMEBTDEDZ3BESPM3R6/action/replication_record"}},"created_at":"2026-07-05T10:18:31.641211+00:00","updated_at":"2026-07-05T10:18:31.641211+00:00"}