{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:I2IFIJL4OC3G637RE3NOPJ2HED","short_pith_number":"pith:I2IFIJL4","schema_version":"1.0","canonical_sha256":"469054257c70b66f6ff126dae7a74720cb5c318f0464ee5c9c0934485fe31f7d","source":{"kind":"arxiv","id":"2209.04924","version":2},"attestation_state":"computed","paper":{"title":"Meta-Reinforcement Learning via Language Instructions","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Alexander Koch, Alois Knoll, Kai Huang, Xiangtong Yao, Zhenshan Bing","submitted_at":"2022-09-11T19:42:48Z","abstract_excerpt":"Although deep reinforcement learning has recently been very successful at learning complex behaviors, it requires a tremendous amount of data to learn a task. One of the fundamental reasons causing this limitation lies in the nature of the trial-and-error learning paradigm of reinforcement learning, where the agent communicates with the environment and progresses in the learning only relying on the reward signal. This is implicit and rather insufficient to learn a task well. On the contrary, humans are usually taught new skills via natural language instructions. Utilizing language instructions"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.04924","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2022-09-11T19:42:48Z","cross_cats_sorted":[],"title_canon_sha256":"f7a443488f9de6fe775ee4b7b318140d87965d05fe60b06f6530fed1237576bd","abstract_canon_sha256":"d7938e69fb06e1250163160cdf3159f52a534fc4e5a86b4f7c4d249b3b076563"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:58:06.154005Z","signature_b64":"HK9Yje4wkvSdqOrlZkbUM55k1mPTeoVwzSb6uXFdpt+kbJtaRG9Lk5P6x98gLRTR94RWgW22o0ApW31uOGkcCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"469054257c70b66f6ff126dae7a74720cb5c318f0464ee5c9c0934485fe31f7d","last_reissued_at":"2026-07-05T04:58:06.153524Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:58:06.153524Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Meta-Reinforcement Learning via Language Instructions","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Alexander Koch, Alois Knoll, Kai Huang, Xiangtong Yao, Zhenshan Bing","submitted_at":"2022-09-11T19:42:48Z","abstract_excerpt":"Although deep reinforcement learning has recently been very successful at learning complex behaviors, it requires a tremendous amount of data to learn a task. One of the fundamental reasons causing this limitation lies in the nature of the trial-and-error learning paradigm of reinforcement learning, where the agent communicates with the environment and progresses in the learning only relying on the reward signal. This is implicit and rather insufficient to learn a task well. On the contrary, humans are usually taught new skills via natural language instructions. Utilizing language instructions"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.04924","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.04924/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.04924","created_at":"2026-07-05T04:58:06.153583+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.04924v2","created_at":"2026-07-05T04:58:06.153583+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.04924","created_at":"2026-07-05T04:58:06.153583+00:00"},{"alias_kind":"pith_short_12","alias_value":"I2IFIJL4OC3G","created_at":"2026-07-05T04:58:06.153583+00:00"},{"alias_kind":"pith_short_16","alias_value":"I2IFIJL4OC3G637R","created_at":"2026-07-05T04:58:06.153583+00:00"},{"alias_kind":"pith_short_8","alias_value":"I2IFIJL4","created_at":"2026-07-05T04:58:06.153583+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED","json":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED.json","graph_json":"https://pith.science/api/pith-number/I2IFIJL4OC3G637RE3NOPJ2HED/graph.json","events_json":"https://pith.science/api/pith-number/I2IFIJL4OC3G637RE3NOPJ2HED/events.json","paper":"https://pith.science/paper/I2IFIJL4"},"agent_actions":{"view_html":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED","download_json":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED.json","view_paper":"https://pith.science/paper/I2IFIJL4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.04924&json=true","fetch_graph":"https://pith.science/api/pith-number/I2IFIJL4OC3G637RE3NOPJ2HED/graph.json","fetch_events":"https://pith.science/api/pith-number/I2IFIJL4OC3G637RE3NOPJ2HED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED/action/storage_attestation","attest_author":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED/action/author_attestation","sign_citation":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED/action/citation_signature","submit_replication":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED/action/replication_record"}},"created_at":"2026-07-05T04:58:06.153583+00:00","updated_at":"2026-07-05T04:58:06.153583+00:00"}