{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:U4EMEGOOLHUZT5PQRIZOKXE7QE","short_pith_number":"pith:U4EMEGOO","schema_version":"1.0","canonical_sha256":"a708c219ce59e999f5f08a32e55c9f81026b18a356cfea65baddc5b6b6e811d1","source":{"kind":"arxiv","id":"2501.08925","version":3},"attestation_state":"computed","paper":{"title":"Disentangling Exploration of Large Language Models by Optimal Exploitation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Christian Bartelt, Patrick Betz, Sascha Marton, Stefan L\\\"udtke, Tim Grams","submitted_at":"2025-01-15T16:30:29Z","abstract_excerpt":"Exploration is a crucial skill for in-context reinforcement learning in unknown environments. However, it remains unclear if large language models can effectively explore a partially hidden state space. This work isolates exploration as the sole objective, tasking an agent with gathering information that enhances future returns. Within this framework, we argue that measuring agent returns is not sufficient for a fair evaluation. Hence, we decompose missing rewards into their exploration and exploitation components based on the optimal achievable return. Experiments with various models reveal t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.08925","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-15T16:30:29Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"4f98415388c511afc734b60014ba9f3cc80e13278d0ea51ae0a104dc67ee477e","abstract_canon_sha256":"dec9712dfc934dac297270bd1dec8b1dbe8ad292259e37a9bb28dfc7121e6694"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:58:23.854095Z","signature_b64":"GPZCrDGkQlmjDWHUrHNQo5xIJA7NQ9yEdel+U1hFOD6wz4h/ObZUZGotfjT/es7IAGtowHowZ0C0QCj+qrMpBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a708c219ce59e999f5f08a32e55c9f81026b18a356cfea65baddc5b6b6e811d1","last_reissued_at":"2026-07-05T11:58:23.853559Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:58:23.853559Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Disentangling Exploration of Large Language Models by Optimal Exploitation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Christian Bartelt, Patrick Betz, Sascha Marton, Stefan L\\\"udtke, Tim Grams","submitted_at":"2025-01-15T16:30:29Z","abstract_excerpt":"Exploration is a crucial skill for in-context reinforcement learning in unknown environments. However, it remains unclear if large language models can effectively explore a partially hidden state space. This work isolates exploration as the sole objective, tasking an agent with gathering information that enhances future returns. Within this framework, we argue that measuring agent returns is not sufficient for a fair evaluation. Hence, we decompose missing rewards into their exploration and exploitation components based on the optimal achievable return. Experiments with various models reveal t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.08925","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.08925/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.08925","created_at":"2026-07-05T11:58:23.853627+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.08925v3","created_at":"2026-07-05T11:58:23.853627+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.08925","created_at":"2026-07-05T11:58:23.853627+00:00"},{"alias_kind":"pith_short_12","alias_value":"U4EMEGOOLHUZ","created_at":"2026-07-05T11:58:23.853627+00:00"},{"alias_kind":"pith_short_16","alias_value":"U4EMEGOOLHUZT5PQ","created_at":"2026-07-05T11:58:23.853627+00:00"},{"alias_kind":"pith_short_8","alias_value":"U4EMEGOO","created_at":"2026-07-05T11:58:23.853627+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.08925","citing_title":"Disentangling Exploration of Large Language Models by Optimal Exploitation","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U4EMEGOOLHUZT5PQRIZOKXE7QE","json":"https://pith.science/pith/U4EMEGOOLHUZT5PQRIZOKXE7QE.json","graph_json":"https://pith.science/api/pith-number/U4EMEGOOLHUZT5PQRIZOKXE7QE/graph.json","events_json":"https://pith.science/api/pith-number/U4EMEGOOLHUZT5PQRIZOKXE7QE/events.json","paper":"https://pith.science/paper/U4EMEGOO"},"agent_actions":{"view_html":"https://pith.science/pith/U4EMEGOOLHUZT5PQRIZOKXE7QE","download_json":"https://pith.science/pith/U4EMEGOOLHUZT5PQRIZOKXE7QE.json","view_paper":"https://pith.science/paper/U4EMEGOO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.08925&json=true","fetch_graph":"https://pith.science/api/pith-number/U4EMEGOOLHUZT5PQRIZOKXE7QE/graph.json","fetch_events":"https://pith.science/api/pith-number/U4EMEGOOLHUZT5PQRIZOKXE7QE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U4EMEGOOLHUZT5PQRIZOKXE7QE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U4EMEGOOLHUZT5PQRIZOKXE7QE/action/storage_attestation","attest_author":"https://pith.science/pith/U4EMEGOOLHUZT5PQRIZOKXE7QE/action/author_attestation","sign_citation":"https://pith.science/pith/U4EMEGOOLHUZT5PQRIZOKXE7QE/action/citation_signature","submit_replication":"https://pith.science/pith/U4EMEGOOLHUZT5PQRIZOKXE7QE/action/replication_record"}},"created_at":"2026-07-05T11:58:23.853627+00:00","updated_at":"2026-07-05T11:58:23.853627+00:00"}