{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DRP6ZV6NUDXKV23DHXXS3G5K4N","short_pith_number":"pith:DRP6ZV6N","schema_version":"1.0","canonical_sha256":"1c5fecd7cda0eeaaeb633def2d9baae3699fee573701c32fdd2637d29c19e7c3","source":{"kind":"arxiv","id":"2509.08329","version":1},"attestation_state":"computed","paper":{"title":"Accelerating Reinforcement Learning Algorithms Convergence using Pre-trained Large Language Models as Tutors With Advice Reusing","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Lukas Toral, Teddy Lazebnik","submitted_at":"2025-09-10T07:08:04Z","abstract_excerpt":"Reinforcement Learning (RL) algorithms often require long training to become useful, especially in complex environments with sparse rewards. While techniques like reward shaping and curriculum learning exist to accelerate training, these are often extremely specific and require the developer's professionalism and dedicated expertise in the problem's domain. Tackling this challenge, in this study, we explore the effectiveness of pre-trained Large Language Models (LLMs) as tutors in a student-teacher architecture with RL algorithms, hypothesizing that LLM-generated guidance allows for faster con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.08329","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-10T07:08:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"81c7ab783c6bc3b77cf39280c583ce26f73d6f551d87f793e00ff62a739058b9","abstract_canon_sha256":"7d5295d66db3987ae5b42f99eef7d442e8614a339ed0d5805f7e96aa3d393f18"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:08:27.902935Z","signature_b64":"eAFMwPh75JA+V9apsjDShtkWeihlfKBOp2ZqFKkE5jd9Q3pvJwCVirh2nWEXoQsij+XvuOk9RtFAy6OzkwepAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c5fecd7cda0eeaaeb633def2d9baae3699fee573701c32fdd2637d29c19e7c3","last_reissued_at":"2026-07-05T12:08:27.902455Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:08:27.902455Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Accelerating Reinforcement Learning Algorithms Convergence using Pre-trained Large Language Models as Tutors With Advice Reusing","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Lukas Toral, Teddy Lazebnik","submitted_at":"2025-09-10T07:08:04Z","abstract_excerpt":"Reinforcement Learning (RL) algorithms often require long training to become useful, especially in complex environments with sparse rewards. While techniques like reward shaping and curriculum learning exist to accelerate training, these are often extremely specific and require the developer's professionalism and dedicated expertise in the problem's domain. Tackling this challenge, in this study, we explore the effectiveness of pre-trained Large Language Models (LLMs) as tutors in a student-teacher architecture with RL algorithms, hypothesizing that LLM-generated guidance allows for faster con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.08329","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.08329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.08329","created_at":"2026-07-05T12:08:27.902533+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.08329v1","created_at":"2026-07-05T12:08:27.902533+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.08329","created_at":"2026-07-05T12:08:27.902533+00:00"},{"alias_kind":"pith_short_12","alias_value":"DRP6ZV6NUDXK","created_at":"2026-07-05T12:08:27.902533+00:00"},{"alias_kind":"pith_short_16","alias_value":"DRP6ZV6NUDXKV23D","created_at":"2026-07-05T12:08:27.902533+00:00"},{"alias_kind":"pith_short_8","alias_value":"DRP6ZV6N","created_at":"2026-07-05T12:08:27.902533+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DRP6ZV6NUDXKV23DHXXS3G5K4N","json":"https://pith.science/pith/DRP6ZV6NUDXKV23DHXXS3G5K4N.json","graph_json":"https://pith.science/api/pith-number/DRP6ZV6NUDXKV23DHXXS3G5K4N/graph.json","events_json":"https://pith.science/api/pith-number/DRP6ZV6NUDXKV23DHXXS3G5K4N/events.json","paper":"https://pith.science/paper/DRP6ZV6N"},"agent_actions":{"view_html":"https://pith.science/pith/DRP6ZV6NUDXKV23DHXXS3G5K4N","download_json":"https://pith.science/pith/DRP6ZV6NUDXKV23DHXXS3G5K4N.json","view_paper":"https://pith.science/paper/DRP6ZV6N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.08329&json=true","fetch_graph":"https://pith.science/api/pith-number/DRP6ZV6NUDXKV23DHXXS3G5K4N/graph.json","fetch_events":"https://pith.science/api/pith-number/DRP6ZV6NUDXKV23DHXXS3G5K4N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DRP6ZV6NUDXKV23DHXXS3G5K4N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DRP6ZV6NUDXKV23DHXXS3G5K4N/action/storage_attestation","attest_author":"https://pith.science/pith/DRP6ZV6NUDXKV23DHXXS3G5K4N/action/author_attestation","sign_citation":"https://pith.science/pith/DRP6ZV6NUDXKV23DHXXS3G5K4N/action/citation_signature","submit_replication":"https://pith.science/pith/DRP6ZV6NUDXKV23DHXXS3G5K4N/action/replication_record"}},"created_at":"2026-07-05T12:08:27.902533+00:00","updated_at":"2026-07-05T12:08:27.902533+00:00"}