{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CJZXT6KREC4OLYMAAJSIVTFKSN","short_pith_number":"pith:CJZXT6KR","schema_version":"1.0","canonical_sha256":"127379f95120b8e5e18002648accaa93594a2c2f9c71b63f59675cd8047635ee","source":{"kind":"arxiv","id":"2510.11686","version":2},"attestation_state":"computed","paper":{"title":"Representation-Based Exploration for Language Models: From Test-Time to Post-Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Akshay Krishnamurthy, Dylan J. Foster, Jens Tuyls, Jordan T. Ash","submitted_at":"2025-10-13T17:49:05Z","abstract_excerpt":"Reinforcement learning (RL) promises to expand the capabilities of language models, but it is unclear if current RL techniques promote the discovery of novel behaviors, or simply sharpen those already present in the base model. In this paper, we investigate the value of deliberate exploration -- explicitly incentivizing the model to discover novel and diverse behaviors -- and aim to understand how the knowledge in pre-trained models can guide this search. Our main finding is that exploration with a simple, principled, representation-based bonus derived from the pre-trained language model's hid"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2510.11686","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-10-13T17:49:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0cd4f5a32ac88cce740d8f87723e392506447276c421e83227f5068326149277","abstract_canon_sha256":"fa6a33030fe47672f6e0027c23953f2af9f4c5b4f420effd9143a319a55c0f07"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-16T01:22:28.690884Z","signature_b64":"dzhUXbUJxKOds9qmMgr7CfcluZr5rq8xNtKbTX+HGMpiK+Xntpp5TAUE09G2nVlLp10xp9dHTeKc2F7IQ81LCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"127379f95120b8e5e18002648accaa93594a2c2f9c71b63f59675cd8047635ee","last_reissued_at":"2026-07-16T01:22:28.689992Z","signature_status":"signed_v1","first_computed_at":"2026-07-16T01:22:28.689992Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Representation-Based Exploration for Language Models: From Test-Time to Post-Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Akshay Krishnamurthy, Dylan J. Foster, Jens Tuyls, Jordan T. Ash","submitted_at":"2025-10-13T17:49:05Z","abstract_excerpt":"Reinforcement learning (RL) promises to expand the capabilities of language models, but it is unclear if current RL techniques promote the discovery of novel behaviors, or simply sharpen those already present in the base model. In this paper, we investigate the value of deliberate exploration -- explicitly incentivizing the model to discover novel and diverse behaviors -- and aim to understand how the knowledge in pre-trained models can guide this search. Our main finding is that exploration with a simple, principled, representation-based bonus derived from the pre-trained language model's hid"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.11686","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.11686/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2510.11686","created_at":"2026-07-16T01:22:28.690420+00:00"},{"alias_kind":"arxiv_version","alias_value":"2510.11686v2","created_at":"2026-07-16T01:22:28.690420+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.11686","created_at":"2026-07-16T01:22:28.690420+00:00"},{"alias_kind":"pith_short_12","alias_value":"CJZXT6KREC4O","created_at":"2026-07-16T01:22:28.690420+00:00"},{"alias_kind":"pith_short_16","alias_value":"CJZXT6KREC4OLYMA","created_at":"2026-07-16T01:22:28.690420+00:00"},{"alias_kind":"pith_short_8","alias_value":"CJZXT6KR","created_at":"2026-07-16T01:22:28.690420+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":8,"sample":[{"citing_arxiv_id":"2606.06096","citing_title":"OrderGrad: Optimizing Beyond the Mean with Order-Statistic Policy Gradient Estimation","ref_index":96,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06080","citing_title":"On Advantage Estimates for Max@K Policy Gradients","ref_index":53,"is_internal_anchor":true},{"citing_arxiv_id":"2605.04477","citing_title":"Data-dependent Exploration for Online Reinforcement Learning from Human Feedback","ref_index":31,"is_internal_anchor":true},{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":29,"is_internal_anchor":true},{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":29,"is_internal_anchor":true},{"citing_arxiv_id":"2605.11361","citing_title":"The tractability landscape of diffusion alignment: regularization, rewards, and computational primitives","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2605.04477","citing_title":"Data-dependent Exploration for Online Reinforcement Learning from Human Feedback","ref_index":85,"is_internal_anchor":true},{"citing_arxiv_id":"2604.04855","citing_title":"The Role of Generator Access in Autoregressive Post-Training","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CJZXT6KREC4OLYMAAJSIVTFKSN","json":"https://pith.science/pith/CJZXT6KREC4OLYMAAJSIVTFKSN.json","graph_json":"https://pith.science/api/pith-number/CJZXT6KREC4OLYMAAJSIVTFKSN/graph.json","events_json":"https://pith.science/api/pith-number/CJZXT6KREC4OLYMAAJSIVTFKSN/events.json","paper":"https://pith.science/paper/CJZXT6KR"},"agent_actions":{"view_html":"https://pith.science/pith/CJZXT6KREC4OLYMAAJSIVTFKSN","download_json":"https://pith.science/pith/CJZXT6KREC4OLYMAAJSIVTFKSN.json","view_paper":"https://pith.science/paper/CJZXT6KR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2510.11686&json=true","fetch_graph":"https://pith.science/api/pith-number/CJZXT6KREC4OLYMAAJSIVTFKSN/graph.json","fetch_events":"https://pith.science/api/pith-number/CJZXT6KREC4OLYMAAJSIVTFKSN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CJZXT6KREC4OLYMAAJSIVTFKSN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CJZXT6KREC4OLYMAAJSIVTFKSN/action/storage_attestation","attest_author":"https://pith.science/pith/CJZXT6KREC4OLYMAAJSIVTFKSN/action/author_attestation","sign_citation":"https://pith.science/pith/CJZXT6KREC4OLYMAAJSIVTFKSN/action/citation_signature","submit_replication":"https://pith.science/pith/CJZXT6KREC4OLYMAAJSIVTFKSN/action/replication_record"}},"created_at":"2026-07-16T01:22:28.690420+00:00","updated_at":"2026-07-16T01:22:28.690420+00:00"}