{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:EDGGAVPF62HR7475JNXVCPQ6ET","short_pith_number":"pith:EDGGAVPF","schema_version":"1.0","canonical_sha256":"20cc6055e5f68f1ff3fd4b6f513e1e24dc7ebcfa82980348026ce8a8b0ce3f9a","source":{"kind":"arxiv","id":"2302.09368","version":1},"attestation_state":"computed","paper":{"title":"Natural Language-conditioned Reinforcement Learning with Inside-out Task Language Development and Translation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jing-cheng Pang, Si-Hang Yang, Xin-Yu Yang, Yang Yu","submitted_at":"2023-02-18T15:49:09Z","abstract_excerpt":"Natural Language-conditioned reinforcement learning (RL) enables the agents to follow human instructions. Previous approaches generally implemented language-conditioned RL by providing human instructions in natural language (NL) and training a following policy. In this outside-in approach, the policy needs to comprehend the NL and manage the task simultaneously. However, the unbounded NL examples often bring much extra complexity for solving concrete RL tasks, which can distract policy learning from completing the task. To ease the learning burden of the policy, we investigate an inside-out sc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.09368","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-02-18T15:49:09Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f4b502deaa87b3c921289e07000f9b08af2b84fe335a6f37fbca7401aadf9f68","abstract_canon_sha256":"df7004719c4a0151be668d5ca2bf1080c655afc2f581d272fd2c0df451f1474e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:43:24.940580Z","signature_b64":"Hjpm4MMgraJ1lKE7km7NngXzIp5m5MWpDomDlfc2OYI/CbAnQbDaRBZua1cuhLmWd3WUhfn5CzbQSMQ2dPIzAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"20cc6055e5f68f1ff3fd4b6f513e1e24dc7ebcfa82980348026ce8a8b0ce3f9a","last_reissued_at":"2026-07-05T05:43:24.940225Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:43:24.940225Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Natural Language-conditioned Reinforcement Learning with Inside-out Task Language Development and Translation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jing-cheng Pang, Si-Hang Yang, Xin-Yu Yang, Yang Yu","submitted_at":"2023-02-18T15:49:09Z","abstract_excerpt":"Natural Language-conditioned reinforcement learning (RL) enables the agents to follow human instructions. Previous approaches generally implemented language-conditioned RL by providing human instructions in natural language (NL) and training a following policy. In this outside-in approach, the policy needs to comprehend the NL and manage the task simultaneously. However, the unbounded NL examples often bring much extra complexity for solving concrete RL tasks, which can distract policy learning from completing the task. To ease the learning burden of the policy, we investigate an inside-out sc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.09368","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.09368/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.09368","created_at":"2026-07-05T05:43:24.940286+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.09368v1","created_at":"2026-07-05T05:43:24.940286+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.09368","created_at":"2026-07-05T05:43:24.940286+00:00"},{"alias_kind":"pith_short_12","alias_value":"EDGGAVPF62HR","created_at":"2026-07-05T05:43:24.940286+00:00"},{"alias_kind":"pith_short_16","alias_value":"EDGGAVPF62HR7475","created_at":"2026-07-05T05:43:24.940286+00:00"},{"alias_kind":"pith_short_8","alias_value":"EDGGAVPF","created_at":"2026-07-05T05:43:24.940286+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EDGGAVPF62HR7475JNXVCPQ6ET","json":"https://pith.science/pith/EDGGAVPF62HR7475JNXVCPQ6ET.json","graph_json":"https://pith.science/api/pith-number/EDGGAVPF62HR7475JNXVCPQ6ET/graph.json","events_json":"https://pith.science/api/pith-number/EDGGAVPF62HR7475JNXVCPQ6ET/events.json","paper":"https://pith.science/paper/EDGGAVPF"},"agent_actions":{"view_html":"https://pith.science/pith/EDGGAVPF62HR7475JNXVCPQ6ET","download_json":"https://pith.science/pith/EDGGAVPF62HR7475JNXVCPQ6ET.json","view_paper":"https://pith.science/paper/EDGGAVPF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.09368&json=true","fetch_graph":"https://pith.science/api/pith-number/EDGGAVPF62HR7475JNXVCPQ6ET/graph.json","fetch_events":"https://pith.science/api/pith-number/EDGGAVPF62HR7475JNXVCPQ6ET/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EDGGAVPF62HR7475JNXVCPQ6ET/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EDGGAVPF62HR7475JNXVCPQ6ET/action/storage_attestation","attest_author":"https://pith.science/pith/EDGGAVPF62HR7475JNXVCPQ6ET/action/author_attestation","sign_citation":"https://pith.science/pith/EDGGAVPF62HR7475JNXVCPQ6ET/action/citation_signature","submit_replication":"https://pith.science/pith/EDGGAVPF62HR7475JNXVCPQ6ET/action/replication_record"}},"created_at":"2026-07-05T05:43:24.940286+00:00","updated_at":"2026-07-05T05:43:24.940286+00:00"}