{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:KT77R2PNXJSBSC3PJGI6V5JYGC","short_pith_number":"pith:KT77R2PN","schema_version":"1.0","canonical_sha256":"54fff8e9edba64190b6f4991eaf53830863b0fc49fa8bb1747c2d70499b6bf08","source":{"kind":"arxiv","id":"1908.03731","version":1},"attestation_state":"computed","paper":{"title":"Learning to Explore in Motion and Interaction Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Ludovic Righetti, Miroslav Bogdanovic","submitted_at":"2019-08-10T11:04:42Z","abstract_excerpt":"Model free reinforcement learning suffers from the high sampling complexity inherent to robotic manipulation or locomotion tasks. Most successful approaches typically use random sampling strategies which leads to slow policy convergence. In this paper we present a novel approach for efficient exploration that leverages previously learned tasks. We exploit the fact that the same system is used across many tasks and build a generative model for exploration based on data from previously solved tasks to improve learning new tasks. The approach also enables continuous learning of improved explorati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.03731","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2019-08-10T11:04:42Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9b9b6fc1d7b6e5de61d8818dbd7fed2a234c821c2cad4273bb0315ce11606700","abstract_canon_sha256":"e198ef84d024a6071381472122b6783996d86e14124308f86d60d33ee99add3a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T23:53:17.804488Z","signature_b64":"s10SHWtTZ+ybgfsdEGNaBf7q26Ty1y7SWbl2NC1DVxe789EXQB7baXVjH9rINzmeXjlqlytfH84kjCBXagWnAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54fff8e9edba64190b6f4991eaf53830863b0fc49fa8bb1747c2d70499b6bf08","last_reissued_at":"2026-07-04T23:53:17.804092Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T23:53:17.804092Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Explore in Motion and Interaction Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Ludovic Righetti, Miroslav Bogdanovic","submitted_at":"2019-08-10T11:04:42Z","abstract_excerpt":"Model free reinforcement learning suffers from the high sampling complexity inherent to robotic manipulation or locomotion tasks. Most successful approaches typically use random sampling strategies which leads to slow policy convergence. In this paper we present a novel approach for efficient exploration that leverages previously learned tasks. We exploit the fact that the same system is used across many tasks and build a generative model for exploration based on data from previously solved tasks to improve learning new tasks. The approach also enables continuous learning of improved explorati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.03731","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.03731/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.03731","created_at":"2026-07-04T23:53:17.804158+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.03731v1","created_at":"2026-07-04T23:53:17.804158+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.03731","created_at":"2026-07-04T23:53:17.804158+00:00"},{"alias_kind":"pith_short_12","alias_value":"KT77R2PNXJSB","created_at":"2026-07-04T23:53:17.804158+00:00"},{"alias_kind":"pith_short_16","alias_value":"KT77R2PNXJSBSC3P","created_at":"2026-07-04T23:53:17.804158+00:00"},{"alias_kind":"pith_short_8","alias_value":"KT77R2PN","created_at":"2026-07-04T23:53:17.804158+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KT77R2PNXJSBSC3PJGI6V5JYGC","json":"https://pith.science/pith/KT77R2PNXJSBSC3PJGI6V5JYGC.json","graph_json":"https://pith.science/api/pith-number/KT77R2PNXJSBSC3PJGI6V5JYGC/graph.json","events_json":"https://pith.science/api/pith-number/KT77R2PNXJSBSC3PJGI6V5JYGC/events.json","paper":"https://pith.science/paper/KT77R2PN"},"agent_actions":{"view_html":"https://pith.science/pith/KT77R2PNXJSBSC3PJGI6V5JYGC","download_json":"https://pith.science/pith/KT77R2PNXJSBSC3PJGI6V5JYGC.json","view_paper":"https://pith.science/paper/KT77R2PN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.03731&json=true","fetch_graph":"https://pith.science/api/pith-number/KT77R2PNXJSBSC3PJGI6V5JYGC/graph.json","fetch_events":"https://pith.science/api/pith-number/KT77R2PNXJSBSC3PJGI6V5JYGC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KT77R2PNXJSBSC3PJGI6V5JYGC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KT77R2PNXJSBSC3PJGI6V5JYGC/action/storage_attestation","attest_author":"https://pith.science/pith/KT77R2PNXJSBSC3PJGI6V5JYGC/action/author_attestation","sign_citation":"https://pith.science/pith/KT77R2PNXJSBSC3PJGI6V5JYGC/action/citation_signature","submit_replication":"https://pith.science/pith/KT77R2PNXJSBSC3PJGI6V5JYGC/action/replication_record"}},"created_at":"2026-07-04T23:53:17.804158+00:00","updated_at":"2026-07-04T23:53:17.804158+00:00"}