{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:A6ZLAZOWCNVTTMOTNLQD35HSDX","short_pith_number":"pith:A6ZLAZOW","schema_version":"1.0","canonical_sha256":"07b2b065d6136b39b1d36ae03df4f21dcee256c234059bce612b7ded0d5886a1","source":{"kind":"arxiv","id":"1906.05838","version":3},"attestation_state":"computed","paper":{"title":"Goal-conditioned Imitation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.NE","stat.ML"],"primary_cat":"cs.LG","authors_text":"Carlos Florensa, Mariano Phielipp, Pieter Abbeel, Yiming Ding","submitted_at":"2019-06-13T17:39:52Z","abstract_excerpt":"Designing rewards for Reinforcement Learning (RL) is challenging because it needs to convey the desired task, be efficient to optimize, and be easy to compute. The latter is particularly problematic when applying RL to robotics, where detecting whether the desired configuration is reached might require considerable supervision and instrumentation. Furthermore, we are often interested in being able to reach a wide range of configurations, hence setting up a different reward every time might be unpractical. Methods like Hindsight Experience Replay (HER) have recently shown promise to learn polic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.05838","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-06-13T17:39:52Z","cross_cats_sorted":["cs.AI","cs.NE","stat.ML"],"title_canon_sha256":"f9fa1c180f090a5802d4b23b5a8a41a454d005d2906ab65ee8dba99f4b881bcc","abstract_canon_sha256":"21bccae31acc5577be087e9abac376e4a28e6417991d14d83321e6ee38829348"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:06:07.936663Z","signature_b64":"XjuMOWtkSNG2BPpwJmW8BUShaV06gvXGPw6iuRHr1Jabtdm4LpeD/sN0en5Vo47aCD6x+ZwTXquCV0HfEYuFBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07b2b065d6136b39b1d36ae03df4f21dcee256c234059bce612b7ded0d5886a1","last_reissued_at":"2026-07-05T01:06:07.936171Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:06:07.936171Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Goal-conditioned Imitation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.NE","stat.ML"],"primary_cat":"cs.LG","authors_text":"Carlos Florensa, Mariano Phielipp, Pieter Abbeel, Yiming Ding","submitted_at":"2019-06-13T17:39:52Z","abstract_excerpt":"Designing rewards for Reinforcement Learning (RL) is challenging because it needs to convey the desired task, be efficient to optimize, and be easy to compute. The latter is particularly problematic when applying RL to robotics, where detecting whether the desired configuration is reached might require considerable supervision and instrumentation. Furthermore, we are often interested in being able to reach a wide range of configurations, hence setting up a different reward every time might be unpractical. Methods like Hindsight Experience Replay (HER) have recently shown promise to learn polic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.05838","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1906.05838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.05838","created_at":"2026-07-05T01:06:07.936228+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.05838v3","created_at":"2026-07-05T01:06:07.936228+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.05838","created_at":"2026-07-05T01:06:07.936228+00:00"},{"alias_kind":"pith_short_12","alias_value":"A6ZLAZOWCNVT","created_at":"2026-07-05T01:06:07.936228+00:00"},{"alias_kind":"pith_short_16","alias_value":"A6ZLAZOWCNVTTMOT","created_at":"2026-07-05T01:06:07.936228+00:00"},{"alias_kind":"pith_short_8","alias_value":"A6ZLAZOW","created_at":"2026-07-05T01:06:07.936228+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.16139","citing_title":"Equivariant Goal Conditioned Contrastive Reinforcement Learning","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A6ZLAZOWCNVTTMOTNLQD35HSDX","json":"https://pith.science/pith/A6ZLAZOWCNVTTMOTNLQD35HSDX.json","graph_json":"https://pith.science/api/pith-number/A6ZLAZOWCNVTTMOTNLQD35HSDX/graph.json","events_json":"https://pith.science/api/pith-number/A6ZLAZOWCNVTTMOTNLQD35HSDX/events.json","paper":"https://pith.science/paper/A6ZLAZOW"},"agent_actions":{"view_html":"https://pith.science/pith/A6ZLAZOWCNVTTMOTNLQD35HSDX","download_json":"https://pith.science/pith/A6ZLAZOWCNVTTMOTNLQD35HSDX.json","view_paper":"https://pith.science/paper/A6ZLAZOW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.05838&json=true","fetch_graph":"https://pith.science/api/pith-number/A6ZLAZOWCNVTTMOTNLQD35HSDX/graph.json","fetch_events":"https://pith.science/api/pith-number/A6ZLAZOWCNVTTMOTNLQD35HSDX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A6ZLAZOWCNVTTMOTNLQD35HSDX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A6ZLAZOWCNVTTMOTNLQD35HSDX/action/storage_attestation","attest_author":"https://pith.science/pith/A6ZLAZOWCNVTTMOTNLQD35HSDX/action/author_attestation","sign_citation":"https://pith.science/pith/A6ZLAZOWCNVTTMOTNLQD35HSDX/action/citation_signature","submit_replication":"https://pith.science/pith/A6ZLAZOWCNVTTMOTNLQD35HSDX/action/replication_record"}},"created_at":"2026-07-05T01:06:07.936228+00:00","updated_at":"2026-07-05T01:06:07.936228+00:00"}