{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CEKVJOO75DELLCO2OPMN33WZ2K","short_pith_number":"pith:CEKVJOO7","schema_version":"1.0","canonical_sha256":"111554b9dfe8c8b589da73d8ddeed9d2afcb2907c8b3de6db3cc17dc75c51ac5","source":{"kind":"arxiv","id":"2404.18978","version":1},"attestation_state":"computed","paper":{"title":"Towards Generalizable Agents in Text-Based Educational Environments: A Study of Integrating RL with LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.LG","authors_text":"Adish Singla, Bahar Radmehr, Tanja K\\\"aser","submitted_at":"2024-04-29T14:53:48Z","abstract_excerpt":"There has been a growing interest in developing learner models to enhance learning and teaching experiences in educational environments. However, existing works have primarily focused on structured environments relying on meticulously crafted representations of tasks, thereby limiting the agent's ability to generalize skills across tasks. In this paper, we aim to enhance the generalization capabilities of agents in open-ended text-based learning environments by integrating Reinforcement Learning (RL) with Large Language Models (LLMs). We investigate three types of agents: (i) RL-based agents t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.18978","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-04-29T14:53:48Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"a8a031567a1c35fe89fa8a4e72f67cfccd97ef3ce5ce12437c705783c4582f37","abstract_canon_sha256":"75418e6996560181d84e4a4f67ef70fb686feeefa3ba3e3836ae376d48dbde05"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:13:35.664117Z","signature_b64":"SNOsCsklyPb3NGGmSN9LqSDxtmgqlmYkfNKbNGEiMt2AcSE9zVktON4XUZ4d3pzLjwlu2tOv3CEnC04COVp0CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"111554b9dfe8c8b589da73d8ddeed9d2afcb2907c8b3de6db3cc17dc75c51ac5","last_reissued_at":"2026-07-05T08:13:35.663726Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:13:35.663726Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Generalizable Agents in Text-Based Educational Environments: A Study of Integrating RL with LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.LG","authors_text":"Adish Singla, Bahar Radmehr, Tanja K\\\"aser","submitted_at":"2024-04-29T14:53:48Z","abstract_excerpt":"There has been a growing interest in developing learner models to enhance learning and teaching experiences in educational environments. However, existing works have primarily focused on structured environments relying on meticulously crafted representations of tasks, thereby limiting the agent's ability to generalize skills across tasks. In this paper, we aim to enhance the generalization capabilities of agents in open-ended text-based learning environments by integrating Reinforcement Learning (RL) with Large Language Models (LLMs). We investigate three types of agents: (i) RL-based agents t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.18978","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.18978/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.18978","created_at":"2026-07-05T08:13:35.663779+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.18978v1","created_at":"2026-07-05T08:13:35.663779+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.18978","created_at":"2026-07-05T08:13:35.663779+00:00"},{"alias_kind":"pith_short_12","alias_value":"CEKVJOO75DEL","created_at":"2026-07-05T08:13:35.663779+00:00"},{"alias_kind":"pith_short_16","alias_value":"CEKVJOO75DELLCO2","created_at":"2026-07-05T08:13:35.663779+00:00"},{"alias_kind":"pith_short_8","alias_value":"CEKVJOO7","created_at":"2026-07-05T08:13:35.663779+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.08329","citing_title":"Accelerating Reinforcement Learning Algorithms Convergence using Pre-trained Large Language Models as Tutors With Advice Reusing","ref_index":30,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CEKVJOO75DELLCO2OPMN33WZ2K","json":"https://pith.science/pith/CEKVJOO75DELLCO2OPMN33WZ2K.json","graph_json":"https://pith.science/api/pith-number/CEKVJOO75DELLCO2OPMN33WZ2K/graph.json","events_json":"https://pith.science/api/pith-number/CEKVJOO75DELLCO2OPMN33WZ2K/events.json","paper":"https://pith.science/paper/CEKVJOO7"},"agent_actions":{"view_html":"https://pith.science/pith/CEKVJOO75DELLCO2OPMN33WZ2K","download_json":"https://pith.science/pith/CEKVJOO75DELLCO2OPMN33WZ2K.json","view_paper":"https://pith.science/paper/CEKVJOO7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.18978&json=true","fetch_graph":"https://pith.science/api/pith-number/CEKVJOO75DELLCO2OPMN33WZ2K/graph.json","fetch_events":"https://pith.science/api/pith-number/CEKVJOO75DELLCO2OPMN33WZ2K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CEKVJOO75DELLCO2OPMN33WZ2K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CEKVJOO75DELLCO2OPMN33WZ2K/action/storage_attestation","attest_author":"https://pith.science/pith/CEKVJOO75DELLCO2OPMN33WZ2K/action/author_attestation","sign_citation":"https://pith.science/pith/CEKVJOO75DELLCO2OPMN33WZ2K/action/citation_signature","submit_replication":"https://pith.science/pith/CEKVJOO75DELLCO2OPMN33WZ2K/action/replication_record"}},"created_at":"2026-07-05T08:13:35.663779+00:00","updated_at":"2026-07-05T08:13:35.663779+00:00"}