{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:52SAU6IU6PV46LY5K63GFKXWXT","short_pith_number":"pith:52SAU6IU","schema_version":"1.0","canonical_sha256":"eea40a7914f3ebcf2f1d57b662aaf6bcc6531466db1446f47b09eb673dfbc1e3","source":{"kind":"arxiv","id":"2311.02013","version":2},"attestation_state":"computed","paper":{"title":"SMORE: Score Models for Offline Goal-Conditioned Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Ahmed Touati, Alborz Geramifard, Amy Zhang, Harshit Sikchi, Rohan Chitnis, Scott Niekum","submitted_at":"2023-11-03T16:19:33Z","abstract_excerpt":"Offline Goal-Conditioned Reinforcement Learning (GCRL) is tasked with learning to achieve multiple goals in an environment purely from offline datasets using sparse reward functions. Offline GCRL is pivotal for developing generalist agents capable of leveraging pre-existing datasets to learn diverse and reusable skills without hand-engineering reward functions. However, contemporary approaches to GCRL based on supervised learning and contrastive learning are often suboptimal in the offline setting. An alternative perspective on GCRL optimizes for occupancy matching, but necessitates learning a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.02013","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-11-03T16:19:33Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"e85289c4326510dc0ae4593cef7e97bee2226d11b0a10f026aabcb409e327e7b","abstract_canon_sha256":"62260a810e1f02fd17e35e5d3201bbb1dfc4b2e3f785d98007482d946349782c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:50:21.369447Z","signature_b64":"v3xsFZr5dUjFiG8G10cbJeMQONmkPCU12/yU9NBcJO1i5bqUpexlOIFj0gPCxyDaSbTHScmfqn1eORaX75LFDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eea40a7914f3ebcf2f1d57b662aaf6bcc6531466db1446f47b09eb673dfbc1e3","last_reissued_at":"2026-07-05T07:50:21.368951Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:50:21.368951Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SMORE: Score Models for Offline Goal-Conditioned Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Ahmed Touati, Alborz Geramifard, Amy Zhang, Harshit Sikchi, Rohan Chitnis, Scott Niekum","submitted_at":"2023-11-03T16:19:33Z","abstract_excerpt":"Offline Goal-Conditioned Reinforcement Learning (GCRL) is tasked with learning to achieve multiple goals in an environment purely from offline datasets using sparse reward functions. Offline GCRL is pivotal for developing generalist agents capable of leveraging pre-existing datasets to learn diverse and reusable skills without hand-engineering reward functions. However, contemporary approaches to GCRL based on supervised learning and contrastive learning are often suboptimal in the offline setting. An alternative perspective on GCRL optimizes for occupancy matching, but necessitates learning a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.02013","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.02013/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.02013","created_at":"2026-07-05T07:50:21.369012+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.02013v2","created_at":"2026-07-05T07:50:21.369012+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.02013","created_at":"2026-07-05T07:50:21.369012+00:00"},{"alias_kind":"pith_short_12","alias_value":"52SAU6IU6PV4","created_at":"2026-07-05T07:50:21.369012+00:00"},{"alias_kind":"pith_short_16","alias_value":"52SAU6IU6PV46LY5","created_at":"2026-07-05T07:50:21.369012+00:00"},{"alias_kind":"pith_short_8","alias_value":"52SAU6IU","created_at":"2026-07-05T07:50:21.369012+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30503","citing_title":"Physics-informed Goal-Conditioned Reinforcement Learning under Hybrid Contact Dynamics","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":163,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17551","citing_title":"SVL: Goal-Conditioned Reinforcement Learning as Survival Learning","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/52SAU6IU6PV46LY5K63GFKXWXT","json":"https://pith.science/pith/52SAU6IU6PV46LY5K63GFKXWXT.json","graph_json":"https://pith.science/api/pith-number/52SAU6IU6PV46LY5K63GFKXWXT/graph.json","events_json":"https://pith.science/api/pith-number/52SAU6IU6PV46LY5K63GFKXWXT/events.json","paper":"https://pith.science/paper/52SAU6IU"},"agent_actions":{"view_html":"https://pith.science/pith/52SAU6IU6PV46LY5K63GFKXWXT","download_json":"https://pith.science/pith/52SAU6IU6PV46LY5K63GFKXWXT.json","view_paper":"https://pith.science/paper/52SAU6IU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.02013&json=true","fetch_graph":"https://pith.science/api/pith-number/52SAU6IU6PV46LY5K63GFKXWXT/graph.json","fetch_events":"https://pith.science/api/pith-number/52SAU6IU6PV46LY5K63GFKXWXT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/52SAU6IU6PV46LY5K63GFKXWXT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/52SAU6IU6PV46LY5K63GFKXWXT/action/storage_attestation","attest_author":"https://pith.science/pith/52SAU6IU6PV46LY5K63GFKXWXT/action/author_attestation","sign_citation":"https://pith.science/pith/52SAU6IU6PV46LY5K63GFKXWXT/action/citation_signature","submit_replication":"https://pith.science/pith/52SAU6IU6PV46LY5K63GFKXWXT/action/replication_record"}},"created_at":"2026-07-05T07:50:21.369012+00:00","updated_at":"2026-07-05T07:50:21.369012+00:00"}