{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:LHW2ZNT3CIYJWIHBNIYZCMSA55","short_pith_number":"pith:LHW2ZNT3","schema_version":"1.0","canonical_sha256":"59edacb67b12309b20e16a31913240ef7e86201900a8513fccaeb0f367bb2a85","source":{"kind":"arxiv","id":"2608.04934","version":1},"attestation_state":"computed","paper":{"title":"State2State: Environment-Derived Mid-Training for LLM Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Chenliang Li, Jieping Ye, Kaiming Liu, Ming Yan, Peng Li, Xuanyu Lei, Yang Liu, Ya-Qin Zhang, Yiqi Zhu","submitted_at":"2026-08-05T15:02:41Z","abstract_excerpt":"Training LLM agents commonly relies on supervised fine-tuning from expert trajectories or online reinforcement learning over human-specified tasks with handcrafted verifiers. Though effective, both remain bottlenecked by externally specified tasks and supervision signals, limiting the scalability and diversity of agent training. We study an environment learning paradigm in which agents acquire interaction and manipulation capabilities solely through environment interaction, without externally specified tasks. We propose State2State, an environment-derived mid-training method that converts expl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.04934","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-08-05T15:02:41Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"60b745b2a52ff084ee3b711766b8453e40488d61bfa1993584542bc28ca09842","abstract_canon_sha256":"5ffb04f5bb30d08229a8bae091aa7559159726b6606a599cf04cf31e03c7bf9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-06T01:47:43.843116Z","signature_b64":"IHgNRpoBu3t/C2b6n1JvNyQ4RTalvNmOURQsVaU7czZGWISX+TN6ixw9Nd2nf1Ii4f2C2DwahmXJI8HPBUV5AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"59edacb67b12309b20e16a31913240ef7e86201900a8513fccaeb0f367bb2a85","last_reissued_at":"2026-08-06T01:47:43.841733Z","signature_status":"signed_v1","first_computed_at":"2026-08-06T01:47:43.841733Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"State2State: Environment-Derived Mid-Training for LLM Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Chenliang Li, Jieping Ye, Kaiming Liu, Ming Yan, Peng Li, Xuanyu Lei, Yang Liu, Ya-Qin Zhang, Yiqi Zhu","submitted_at":"2026-08-05T15:02:41Z","abstract_excerpt":"Training LLM agents commonly relies on supervised fine-tuning from expert trajectories or online reinforcement learning over human-specified tasks with handcrafted verifiers. Though effective, both remain bottlenecked by externally specified tasks and supervision signals, limiting the scalability and diversity of agent training. We study an environment learning paradigm in which agents acquire interaction and manipulation capabilities solely through environment interaction, without externally specified tasks. We propose State2State, an environment-derived mid-training method that converts expl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.04934","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.04934/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.04934","created_at":"2026-08-06T01:47:43.843536+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.04934v1","created_at":"2026-08-06T01:47:43.843536+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.04934","created_at":"2026-08-06T01:47:43.843536+00:00"},{"alias_kind":"pith_short_12","alias_value":"LHW2ZNT3CIYJ","created_at":"2026-08-06T01:47:43.843536+00:00"},{"alias_kind":"pith_short_16","alias_value":"LHW2ZNT3CIYJWIHB","created_at":"2026-08-06T01:47:43.843536+00:00"},{"alias_kind":"pith_short_8","alias_value":"LHW2ZNT3","created_at":"2026-08-06T01:47:43.843536+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LHW2ZNT3CIYJWIHBNIYZCMSA55","json":"https://pith.science/pith/LHW2ZNT3CIYJWIHBNIYZCMSA55.json","graph_json":"https://pith.science/api/pith-number/LHW2ZNT3CIYJWIHBNIYZCMSA55/graph.json","events_json":"https://pith.science/api/pith-number/LHW2ZNT3CIYJWIHBNIYZCMSA55/events.json","paper":"https://pith.science/paper/LHW2ZNT3"},"agent_actions":{"view_html":"https://pith.science/pith/LHW2ZNT3CIYJWIHBNIYZCMSA55","download_json":"https://pith.science/pith/LHW2ZNT3CIYJWIHBNIYZCMSA55.json","view_paper":"https://pith.science/paper/LHW2ZNT3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.04934&json=true","fetch_graph":"https://pith.science/api/pith-number/LHW2ZNT3CIYJWIHBNIYZCMSA55/graph.json","fetch_events":"https://pith.science/api/pith-number/LHW2ZNT3CIYJWIHBNIYZCMSA55/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LHW2ZNT3CIYJWIHBNIYZCMSA55/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LHW2ZNT3CIYJWIHBNIYZCMSA55/action/storage_attestation","attest_author":"https://pith.science/pith/LHW2ZNT3CIYJWIHBNIYZCMSA55/action/author_attestation","sign_citation":"https://pith.science/pith/LHW2ZNT3CIYJWIHBNIYZCMSA55/action/citation_signature","submit_replication":"https://pith.science/pith/LHW2ZNT3CIYJWIHBNIYZCMSA55/action/replication_record"}},"created_at":"2026-08-06T01:47:43.843536+00:00","updated_at":"2026-08-06T01:47:43.843536+00:00"}