{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KYASULWEYSJHUIKOYE4U4AVHXN","short_pith_number":"pith:KYASULWE","schema_version":"1.0","canonical_sha256":"56012a2ec4c4927a214ec1394e02a7bb6635f7f55bda721e97ddcdf15855c637","source":{"kind":"arxiv","id":"2410.08632","version":1},"attestation_state":"computed","paper":{"title":"Words as Beacons: Guiding RL Agents with High-Level Language Prompts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Alain Andres, Javier Del Ser, Pedro G.Bascoy, Unai Ruiz-Gonzalez","submitted_at":"2024-10-11T08:54:45Z","abstract_excerpt":"Sparse reward environments in reinforcement learning (RL) pose significant challenges for exploration, often leading to inefficient or incomplete learning processes. To tackle this issue, this work proposes a teacher-student RL framework that leverages Large Language Models (LLMs) as \"teachers\" to guide the agent's learning process by decomposing complex tasks into subgoals. Due to their inherent capability to understand RL environments based on a textual description of structure and purpose, LLMs can provide subgoals to accomplish the task defined for the environment in a similar fashion to h"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08632","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-10-11T08:54:45Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"3b615d0e231102502f984593a364daf9c42cebc49e14287d05e9db9b0030b80e","abstract_canon_sha256":"637f6d7e5f8aa3f1362ae585edf789ebca6d283177b0db79f02c485ba5677f84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:15.422717Z","signature_b64":"gktDWQGZru98hM1B73E9HxtweXsI1uIi9/E3iYkOBG7ew/RyaFUho5o+EefFWMeWrN7UYP2S+ed2JqaCc+CpCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"56012a2ec4c4927a214ec1394e02a7bb6635f7f55bda721e97ddcdf15855c637","last_reissued_at":"2026-07-05T09:19:15.422240Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:15.422240Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Words as Beacons: Guiding RL Agents with High-Level Language Prompts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Alain Andres, Javier Del Ser, Pedro G.Bascoy, Unai Ruiz-Gonzalez","submitted_at":"2024-10-11T08:54:45Z","abstract_excerpt":"Sparse reward environments in reinforcement learning (RL) pose significant challenges for exploration, often leading to inefficient or incomplete learning processes. To tackle this issue, this work proposes a teacher-student RL framework that leverages Large Language Models (LLMs) as \"teachers\" to guide the agent's learning process by decomposing complex tasks into subgoals. Due to their inherent capability to understand RL environments based on a textual description of structure and purpose, LLMs can provide subgoals to accomplish the task defined for the environment in a similar fashion to h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08632","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08632/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08632","created_at":"2026-07-05T09:19:15.422296+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08632v1","created_at":"2026-07-05T09:19:15.422296+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08632","created_at":"2026-07-05T09:19:15.422296+00:00"},{"alias_kind":"pith_short_12","alias_value":"KYASULWEYSJH","created_at":"2026-07-05T09:19:15.422296+00:00"},{"alias_kind":"pith_short_16","alias_value":"KYASULWEYSJHUIKO","created_at":"2026-07-05T09:19:15.422296+00:00"},{"alias_kind":"pith_short_8","alias_value":"KYASULWE","created_at":"2026-07-05T09:19:15.422296+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KYASULWEYSJHUIKOYE4U4AVHXN","json":"https://pith.science/pith/KYASULWEYSJHUIKOYE4U4AVHXN.json","graph_json":"https://pith.science/api/pith-number/KYASULWEYSJHUIKOYE4U4AVHXN/graph.json","events_json":"https://pith.science/api/pith-number/KYASULWEYSJHUIKOYE4U4AVHXN/events.json","paper":"https://pith.science/paper/KYASULWE"},"agent_actions":{"view_html":"https://pith.science/pith/KYASULWEYSJHUIKOYE4U4AVHXN","download_json":"https://pith.science/pith/KYASULWEYSJHUIKOYE4U4AVHXN.json","view_paper":"https://pith.science/paper/KYASULWE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08632&json=true","fetch_graph":"https://pith.science/api/pith-number/KYASULWEYSJHUIKOYE4U4AVHXN/graph.json","fetch_events":"https://pith.science/api/pith-number/KYASULWEYSJHUIKOYE4U4AVHXN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KYASULWEYSJHUIKOYE4U4AVHXN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KYASULWEYSJHUIKOYE4U4AVHXN/action/storage_attestation","attest_author":"https://pith.science/pith/KYASULWEYSJHUIKOYE4U4AVHXN/action/author_attestation","sign_citation":"https://pith.science/pith/KYASULWEYSJHUIKOYE4U4AVHXN/action/citation_signature","submit_replication":"https://pith.science/pith/KYASULWEYSJHUIKOYE4U4AVHXN/action/replication_record"}},"created_at":"2026-07-05T09:19:15.422296+00:00","updated_at":"2026-07-05T09:19:15.422296+00:00"}