{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QO7SHORTWYYHW37DPA2UGZ725V","short_pith_number":"pith:QO7SHORT","schema_version":"1.0","canonical_sha256":"83bf23ba33b6307b6fe378354367faed7440f28bb0ad6ad04a1603fb4d3dbf5d","source":{"kind":"arxiv","id":"2505.03209","version":1},"attestation_state":"computed","paper":{"title":"DYSTIL: Dynamic Strategy Induction with Large Language Models for Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Borui Wang, Kathleen McKeown, Rex Ying","submitted_at":"2025-05-06T05:53:09Z","abstract_excerpt":"Reinforcement learning from expert demonstrations has long remained a challenging research problem, and existing state-of-the-art methods using behavioral cloning plus further RL training often suffer from poor generalization, low sample efficiency, and poor model interpretability. Inspired by the strong reasoning abilities of large language models (LLMs), we propose a novel strategy-based reinforcement learning framework integrated with LLMs called DYnamic STrategy Induction with Llms for reinforcement learning (DYSTIL) to overcome these limitations. DYSTIL dynamically queries a strategy-gene"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.03209","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-06T05:53:09Z","cross_cats_sorted":[],"title_canon_sha256":"3283edd01acd56799f9854b6b46937136a436ef5d9896d89ea6be650f071a43e","abstract_canon_sha256":"14b4efe2dffb8897b15f584f54a0bfed416da8e06ad60eb4df394c153bfbf759"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:12.195214Z","signature_b64":"8dycAsDZ65pWd9J5lmDi2jCcDcOq7TF9s4iM6xoisVlG9bb5GnvlRw/x9pnzYBownEOevlOS8EqQQaQaYvDwBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"83bf23ba33b6307b6fe378354367faed7440f28bb0ad6ad04a1603fb4d3dbf5d","last_reissued_at":"2026-07-05T10:59:12.194740Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:12.194740Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DYSTIL: Dynamic Strategy Induction with Large Language Models for Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Borui Wang, Kathleen McKeown, Rex Ying","submitted_at":"2025-05-06T05:53:09Z","abstract_excerpt":"Reinforcement learning from expert demonstrations has long remained a challenging research problem, and existing state-of-the-art methods using behavioral cloning plus further RL training often suffer from poor generalization, low sample efficiency, and poor model interpretability. Inspired by the strong reasoning abilities of large language models (LLMs), we propose a novel strategy-based reinforcement learning framework integrated with LLMs called DYnamic STrategy Induction with Llms for reinforcement learning (DYSTIL) to overcome these limitations. DYSTIL dynamically queries a strategy-gene"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.03209","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.03209/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.03209","created_at":"2026-07-05T10:59:12.194794+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.03209v1","created_at":"2026-07-05T10:59:12.194794+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.03209","created_at":"2026-07-05T10:59:12.194794+00:00"},{"alias_kind":"pith_short_12","alias_value":"QO7SHORTWYYH","created_at":"2026-07-05T10:59:12.194794+00:00"},{"alias_kind":"pith_short_16","alias_value":"QO7SHORTWYYHW37D","created_at":"2026-07-05T10:59:12.194794+00:00"},{"alias_kind":"pith_short_8","alias_value":"QO7SHORT","created_at":"2026-07-05T10:59:12.194794+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31509","citing_title":"Skill Reuse as Compression in Agentic RL","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21046","citing_title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","ref_index":265,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11427","citing_title":"METRO: Towards Strategy Induction from Expert Dialogue Transcripts for Non-collaborative Dialogues","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QO7SHORTWYYHW37DPA2UGZ725V","json":"https://pith.science/pith/QO7SHORTWYYHW37DPA2UGZ725V.json","graph_json":"https://pith.science/api/pith-number/QO7SHORTWYYHW37DPA2UGZ725V/graph.json","events_json":"https://pith.science/api/pith-number/QO7SHORTWYYHW37DPA2UGZ725V/events.json","paper":"https://pith.science/paper/QO7SHORT"},"agent_actions":{"view_html":"https://pith.science/pith/QO7SHORTWYYHW37DPA2UGZ725V","download_json":"https://pith.science/pith/QO7SHORTWYYHW37DPA2UGZ725V.json","view_paper":"https://pith.science/paper/QO7SHORT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.03209&json=true","fetch_graph":"https://pith.science/api/pith-number/QO7SHORTWYYHW37DPA2UGZ725V/graph.json","fetch_events":"https://pith.science/api/pith-number/QO7SHORTWYYHW37DPA2UGZ725V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QO7SHORTWYYHW37DPA2UGZ725V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QO7SHORTWYYHW37DPA2UGZ725V/action/storage_attestation","attest_author":"https://pith.science/pith/QO7SHORTWYYHW37DPA2UGZ725V/action/author_attestation","sign_citation":"https://pith.science/pith/QO7SHORTWYYHW37DPA2UGZ725V/action/citation_signature","submit_replication":"https://pith.science/pith/QO7SHORTWYYHW37DPA2UGZ725V/action/replication_record"}},"created_at":"2026-07-05T10:59:12.194794+00:00","updated_at":"2026-07-05T10:59:12.194794+00:00"}