{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CLEBFYI5EFASIMLEU62XSZ57VK","short_pith_number":"pith:CLEBFYI5","schema_version":"1.0","canonical_sha256":"12c812e11d2141243164a7b57967bfaaaaf82aee0776ff1c0f2f82e512e1b6a6","source":{"kind":"arxiv","id":"2410.07927","version":1},"attestation_state":"computed","paper":{"title":"Efficient Reinforcement Learning with Large Language Model Priors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Haifeng Zhang, Haitham Bou Ammar, Jun Wang, Mengyue Yang, Xidong Feng, Xue Yan, Yan Song","submitted_at":"2024-10-10T13:54:11Z","abstract_excerpt":"In sequential decision-making (SDM) tasks, methods like reinforcement learning (RL) and heuristic search have made notable advances in specific cases. However, they often require extensive exploration and face challenges in generalizing across diverse environments due to their limited grasp of the underlying decision dynamics. In contrast, large language models (LLMs) have recently emerged as powerful general-purpose tools, due to their capacity to maintain vast amounts of domain-specific knowledge. To harness this rich prior knowledge for efficiently solving complex SDM tasks, we propose trea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.07927","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-10T13:54:11Z","cross_cats_sorted":[],"title_canon_sha256":"589af001907bfd437bee4626efb66f0392fa807298acb6f51f79f033053726bc","abstract_canon_sha256":"ccfe0f460b305e145e1ce20ea65e03b38e51fdd3aaf5eb08f791c89ace758efb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:18:46.301897Z","signature_b64":"Iq0y6h7Jq78OJIFMXPhJYAkf9zyk4gMTAUd2ce7KL+l/B0MX0Xj1yLc4QwAJcFUwDCZKbiUPfrOtgGY1DoMgBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"12c812e11d2141243164a7b57967bfaaaaf82aee0776ff1c0f2f82e512e1b6a6","last_reissued_at":"2026-07-05T09:18:46.301410Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:18:46.301410Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Reinforcement Learning with Large Language Model Priors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Haifeng Zhang, Haitham Bou Ammar, Jun Wang, Mengyue Yang, Xidong Feng, Xue Yan, Yan Song","submitted_at":"2024-10-10T13:54:11Z","abstract_excerpt":"In sequential decision-making (SDM) tasks, methods like reinforcement learning (RL) and heuristic search have made notable advances in specific cases. However, they often require extensive exploration and face challenges in generalizing across diverse environments due to their limited grasp of the underlying decision dynamics. In contrast, large language models (LLMs) have recently emerged as powerful general-purpose tools, due to their capacity to maintain vast amounts of domain-specific knowledge. To harness this rich prior knowledge for efficiently solving complex SDM tasks, we propose trea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.07927","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.07927/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.07927","created_at":"2026-07-05T09:18:46.301467+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.07927v1","created_at":"2026-07-05T09:18:46.301467+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.07927","created_at":"2026-07-05T09:18:46.301467+00:00"},{"alias_kind":"pith_short_12","alias_value":"CLEBFYI5EFAS","created_at":"2026-07-05T09:18:46.301467+00:00"},{"alias_kind":"pith_short_16","alias_value":"CLEBFYI5EFASIMLE","created_at":"2026-07-05T09:18:46.301467+00:00"},{"alias_kind":"pith_short_8","alias_value":"CLEBFYI5","created_at":"2026-07-05T09:18:46.301467+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24669","citing_title":"LaGO: Latent Action Guidance for Online Reinforcement Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18812","citing_title":"Reinforcement Learning Foundation Models Should Already Be A Thing","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11446","citing_title":"Low-rank Optimization Trajectories Modeling for LLM RLVR Acceleration","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CLEBFYI5EFASIMLEU62XSZ57VK","json":"https://pith.science/pith/CLEBFYI5EFASIMLEU62XSZ57VK.json","graph_json":"https://pith.science/api/pith-number/CLEBFYI5EFASIMLEU62XSZ57VK/graph.json","events_json":"https://pith.science/api/pith-number/CLEBFYI5EFASIMLEU62XSZ57VK/events.json","paper":"https://pith.science/paper/CLEBFYI5"},"agent_actions":{"view_html":"https://pith.science/pith/CLEBFYI5EFASIMLEU62XSZ57VK","download_json":"https://pith.science/pith/CLEBFYI5EFASIMLEU62XSZ57VK.json","view_paper":"https://pith.science/paper/CLEBFYI5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.07927&json=true","fetch_graph":"https://pith.science/api/pith-number/CLEBFYI5EFASIMLEU62XSZ57VK/graph.json","fetch_events":"https://pith.science/api/pith-number/CLEBFYI5EFASIMLEU62XSZ57VK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CLEBFYI5EFASIMLEU62XSZ57VK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CLEBFYI5EFASIMLEU62XSZ57VK/action/storage_attestation","attest_author":"https://pith.science/pith/CLEBFYI5EFASIMLEU62XSZ57VK/action/author_attestation","sign_citation":"https://pith.science/pith/CLEBFYI5EFASIMLEU62XSZ57VK/action/citation_signature","submit_replication":"https://pith.science/pith/CLEBFYI5EFASIMLEU62XSZ57VK/action/replication_record"}},"created_at":"2026-07-05T09:18:46.301467+00:00","updated_at":"2026-07-05T09:18:46.301467+00:00"}