{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IOO22SVXJMLSU4GYXDCO3CAQAI","short_pith_number":"pith:IOO22SVX","schema_version":"1.0","canonical_sha256":"439dad4ab74b172a70d8b8c4ed8810023c68cc0c8a2be69242543ef5a3efbc9d","source":{"kind":"arxiv","id":"2304.13774","version":1},"attestation_state":"computed","paper":{"title":"Distance Weighted Supervised Learning for Offline Interaction Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dorsa Sadigh, Jensen Gao, Joey Hejna","submitted_at":"2023-04-26T18:35:49Z","abstract_excerpt":"Sequential decision making algorithms often struggle to leverage different sources of unstructured offline interaction data. Imitation learning (IL) methods based on supervised learning are robust, but require optimal demonstrations, which are hard to collect. Offline goal-conditioned reinforcement learning (RL) algorithms promise to learn from sub-optimal data, but face optimization challenges especially with high-dimensional data. To bridge the gap between IL and RL, we introduce Distance Weighted Supervised Learning or DWSL, a supervised method for learning goal-conditioned policies from of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.13774","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-26T18:35:49Z","cross_cats_sorted":[],"title_canon_sha256":"11c65f4751c1de75c0825636a326558d3fe45167371e1f6bfc95bacd8369e154","abstract_canon_sha256":"e087eccfabac576d7c16fac47689d0318585c7885a9dcd4efff97bca40a76484"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:04:57.092353Z","signature_b64":"d4CtQMON708F6InayIaJ+uPrSqZCGKBMwKS+G8JR6NYb7IH1hoTM3b/L25emESiz1JhmRWTspFxpkj+AK3F4AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"439dad4ab74b172a70d8b8c4ed8810023c68cc0c8a2be69242543ef5a3efbc9d","last_reissued_at":"2026-07-05T06:04:57.091907Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:04:57.091907Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distance Weighted Supervised Learning for Offline Interaction Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dorsa Sadigh, Jensen Gao, Joey Hejna","submitted_at":"2023-04-26T18:35:49Z","abstract_excerpt":"Sequential decision making algorithms often struggle to leverage different sources of unstructured offline interaction data. Imitation learning (IL) methods based on supervised learning are robust, but require optimal demonstrations, which are hard to collect. Offline goal-conditioned reinforcement learning (RL) algorithms promise to learn from sub-optimal data, but face optimization challenges especially with high-dimensional data. To bridge the gap between IL and RL, we introduce Distance Weighted Supervised Learning or DWSL, a supervised method for learning goal-conditioned policies from of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.13774","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.13774/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.13774","created_at":"2026-07-05T06:04:57.091963+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.13774v1","created_at":"2026-07-05T06:04:57.091963+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.13774","created_at":"2026-07-05T06:04:57.091963+00:00"},{"alias_kind":"pith_short_12","alias_value":"IOO22SVXJMLS","created_at":"2026-07-05T06:04:57.091963+00:00"},{"alias_kind":"pith_short_16","alias_value":"IOO22SVXJMLSU4GY","created_at":"2026-07-05T06:04:57.091963+00:00"},{"alias_kind":"pith_short_8","alias_value":"IOO22SVX","created_at":"2026-07-05T06:04:57.091963+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":69,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IOO22SVXJMLSU4GYXDCO3CAQAI","json":"https://pith.science/pith/IOO22SVXJMLSU4GYXDCO3CAQAI.json","graph_json":"https://pith.science/api/pith-number/IOO22SVXJMLSU4GYXDCO3CAQAI/graph.json","events_json":"https://pith.science/api/pith-number/IOO22SVXJMLSU4GYXDCO3CAQAI/events.json","paper":"https://pith.science/paper/IOO22SVX"},"agent_actions":{"view_html":"https://pith.science/pith/IOO22SVXJMLSU4GYXDCO3CAQAI","download_json":"https://pith.science/pith/IOO22SVXJMLSU4GYXDCO3CAQAI.json","view_paper":"https://pith.science/paper/IOO22SVX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.13774&json=true","fetch_graph":"https://pith.science/api/pith-number/IOO22SVXJMLSU4GYXDCO3CAQAI/graph.json","fetch_events":"https://pith.science/api/pith-number/IOO22SVXJMLSU4GYXDCO3CAQAI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IOO22SVXJMLSU4GYXDCO3CAQAI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IOO22SVXJMLSU4GYXDCO3CAQAI/action/storage_attestation","attest_author":"https://pith.science/pith/IOO22SVXJMLSU4GYXDCO3CAQAI/action/author_attestation","sign_citation":"https://pith.science/pith/IOO22SVXJMLSU4GYXDCO3CAQAI/action/citation_signature","submit_replication":"https://pith.science/pith/IOO22SVXJMLSU4GYXDCO3CAQAI/action/replication_record"}},"created_at":"2026-07-05T06:04:57.091963+00:00","updated_at":"2026-07-05T06:04:57.091963+00:00"}