{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:FECWYRHZV4HGTBI23JNXPMCZZ5","short_pith_number":"pith:FECWYRHZ","canonical_record":{"source":{"id":"2604.20209","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-22T05:50:23Z","cross_cats_sorted":[],"title_canon_sha256":"044af8ce9dda481f70b29f85b5995f6ef1f8d7c79250e8ce9d5d389abecece86","abstract_canon_sha256":"df56fd36436c87a8539b0e963e7154b53daa97bbd620c1f9ed2762223e1be77f"},"schema_version":"1.0"},"canonical_sha256":"29056c44f9af0e69851ada5b77b059cf5fe35b3ea1cc36fa8af3a3d44591c159","source":{"kind":"arxiv","id":"2604.20209","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.20209","created_at":"2026-08-12T01:24:20Z"},{"alias_kind":"arxiv_version","alias_value":"2604.20209v2","created_at":"2026-08-12T01:24:20Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.20209","created_at":"2026-08-12T01:24:20Z"},{"alias_kind":"pith_short_12","alias_value":"FECWYRHZV4HG","created_at":"2026-08-12T01:24:20Z"},{"alias_kind":"pith_short_16","alias_value":"FECWYRHZV4HGTBI2","created_at":"2026-08-12T01:24:20Z"},{"alias_kind":"pith_short_8","alias_value":"FECWYRHZ","created_at":"2026-08-12T01:24:20Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:FECWYRHZV4HGTBI23JNXPMCZZ5","target":"record","payload":{"canonical_record":{"source":{"id":"2604.20209","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-22T05:50:23Z","cross_cats_sorted":[],"title_canon_sha256":"044af8ce9dda481f70b29f85b5995f6ef1f8d7c79250e8ce9d5d389abecece86","abstract_canon_sha256":"df56fd36436c87a8539b0e963e7154b53daa97bbd620c1f9ed2762223e1be77f"},"schema_version":"1.0"},"canonical_sha256":"29056c44f9af0e69851ada5b77b059cf5fe35b3ea1cc36fa8af3a3d44591c159","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-12T01:24:20.549370Z","signature_b64":"JRlWT/0mon3AOCfeCY9uZWk9A0uMV0BZLPmi0dXvtek5/Ys0cKZwzzKmRaBnW7KATK7JcRLa/7yDZj+NolKPDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"29056c44f9af0e69851ada5b77b059cf5fe35b3ea1cc36fa8af3a3d44591c159","last_reissued_at":"2026-08-12T01:24:20.547455Z","signature_status":"signed_v1","first_computed_at":"2026-08-12T01:24:20.547455Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.20209","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-12T01:24:20Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"DjgzHZfB43gG4Wi20qkFoHbvxHAOHNFn2MeApBbo/M5tUce+2GA6Fi8LqU562VFWkKvZC9vvxVA6ipKQCoGNBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-15T02:16:54.796061Z"},"content_sha256":"622ad9b5711b131802da69fa60b13165dbfdf8cecb26cbc89133887b508a06dd","schema_version":"1.0","event_id":"sha256:622ad9b5711b131802da69fa60b13165dbfdf8cecb26cbc89133887b508a06dd"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:FECWYRHZV4HGTBI23JNXPMCZZ5","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Scaling Self-Play with Self-Guidance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Self-guided self-play prevents language models from generating useless problems during training, allowing continued improvement on theorem proving.","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Kaiyue Wen, Kefan Dong, Luke Bailey, Tatsunori Hashimoto, Tengyu Ma","submitted_at":"2026-04-22T05:50:23Z","abstract_excerpt":"LLM self-play algorithms are notable in that, in principle, nothing bounds their learning: a Conjecturer model creates problems for a Solver, and both improve together. However, in practice, existing LLM self-play methods do not scale well with large amounts of compute, instead hitting learning plateaus. We argue this is because over long training runs, the Conjecturer learns to hack its reward, collapsing to artificially complex problems that do not help the Solver improve. To overcome this, we introduce Self-Guided Self-Play (SGS), a self-play algorithm in which the language model itself gui"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Applying SGS to formal theorem proving in Lean4, we find that it surpasses the asymptotic solve rate of our strongest RL baseline in fewer than 80 rounds of self-play and enables a 7B parameter model, after 200 rounds of self-play, to solve more problems than a 671B parameter model pass@4.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"Our core hypothesis is that language models can assess whether a subproblem is useful for achieving a goal.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"SGS adds self-guidance to LLM self-play for Lean4 theorem proving, surpassing RL baselines and enabling a 7B model to outperform a 671B model after 200 rounds.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Self-guided self-play prevents language models from generating useless problems during training, allowing continued improvement on theorem proving.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"e02a150961ed4c367d02afed1854795d2b2a75b74b7fde27e3bbb80fdca07eaf"},"source":{"id":"2604.20209","kind":"arxiv","version":2},"verdict":{"id":"b57b7d9d-62a7-4516-87a1-1ade742792ea","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T01:26:48.614239Z","strongest_claim":"Applying SGS to formal theorem proving in Lean4, we find that it surpasses the asymptotic solve rate of our strongest RL baseline in fewer than 80 rounds of self-play and enables a 7B parameter model, after 200 rounds of self-play, to solve more problems than a 671B parameter model pass@4.","one_line_summary":"SGS adds self-guidance to LLM self-play for Lean4 theorem proving, surpassing RL baselines and enabling a 7B model to outperform a 671B model after 200 rounds.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"Our core hypothesis is that language models can assess whether a subproblem is useful for achieving a goal.","pith_extraction_headline":"Self-guided self-play prevents language models from generating useless problems during training, allowing continued improvement on theorem proving."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.20209/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-21T15:34:35.124388Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-20T02:13:02.275246Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"3bc6801eff556cc68a7385a50dd20d0ed383b42582718eb62e4d2bbfbe77ab09"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"b57b7d9d-62a7-4516-87a1-1ade742792ea"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-12T01:24:20Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2Tt/UbWUNYH+tpRT/jVkQna5ayopeS+P+7kpN+F5+j+h48GwEGnmkusBjtNIp2Kpb6hH7id67Nbg3V3jXUNtCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-15T02:16:54.796734Z"},"content_sha256":"b6f4624ec6b4f26694674c9eef6187c03a1a9209e7d520d6a82b1bf72ec673cd","schema_version":"1.0","event_id":"sha256:b6f4624ec6b4f26694674c9eef6187c03a1a9209e7d520d6a82b1bf72ec673cd"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/FECWYRHZV4HGTBI23JNXPMCZZ5/bundle.json","state_url":"https://pith.science/pith/FECWYRHZV4HGTBI23JNXPMCZZ5/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/FECWYRHZV4HGTBI23JNXPMCZZ5/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-15T02:16:54Z","links":{"resolver":"https://pith.science/pith/FECWYRHZV4HGTBI23JNXPMCZZ5","bundle":"https://pith.science/pith/FECWYRHZV4HGTBI23JNXPMCZZ5/bundle.json","state":"https://pith.science/pith/FECWYRHZV4HGTBI23JNXPMCZZ5/state.json","well_known_bundle":"https://pith.science/.well-known/pith/FECWYRHZV4HGTBI23JNXPMCZZ5/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:FECWYRHZV4HGTBI23JNXPMCZZ5","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"df56fd36436c87a8539b0e963e7154b53daa97bbd620c1f9ed2762223e1be77f","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-22T05:50:23Z","title_canon_sha256":"044af8ce9dda481f70b29f85b5995f6ef1f8d7c79250e8ce9d5d389abecece86"},"schema_version":"1.0","source":{"id":"2604.20209","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.20209","created_at":"2026-08-12T01:24:20Z"},{"alias_kind":"arxiv_version","alias_value":"2604.20209v2","created_at":"2026-08-12T01:24:20Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.20209","created_at":"2026-08-12T01:24:20Z"},{"alias_kind":"pith_short_12","alias_value":"FECWYRHZV4HG","created_at":"2026-08-12T01:24:20Z"},{"alias_kind":"pith_short_16","alias_value":"FECWYRHZV4HGTBI2","created_at":"2026-08-12T01:24:20Z"},{"alias_kind":"pith_short_8","alias_value":"FECWYRHZ","created_at":"2026-08-12T01:24:20Z"}],"graph_snapshots":[{"event_id":"sha256:b6f4624ec6b4f26694674c9eef6187c03a1a9209e7d520d6a82b1bf72ec673cd","target":"graph","created_at":"2026-08-12T01:24:20Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Applying SGS to formal theorem proving in Lean4, we find that it surpasses the asymptotic solve rate of our strongest RL baseline in fewer than 80 rounds of self-play and enables a 7B parameter model, after 200 rounds of self-play, to solve more problems than a 671B parameter model pass@4."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"Our core hypothesis is that language models can assess whether a subproblem is useful for achieving a goal."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"SGS adds self-guidance to LLM self-play for Lean4 theorem proving, surpassing RL baselines and enabling a 7B model to outperform a 671B model after 200 rounds."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Self-guided self-play prevents language models from generating useless problems during training, allowing continued improvement on theorem proving."}],"snapshot_sha256":"e02a150961ed4c367d02afed1854795d2b2a75b74b7fde27e3bbb80fdca07eaf"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-21T15:34:35.124388Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-20T02:13:02.275246Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2604.20209/integrity.json","findings":[],"snapshot_sha256":"3bc6801eff556cc68a7385a50dd20d0ed383b42582718eb62e4d2bbfbe77ab09","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"LLM self-play algorithms are notable in that, in principle, nothing bounds their learning: a Conjecturer model creates problems for a Solver, and both improve together. However, in practice, existing LLM self-play methods do not scale well with large amounts of compute, instead hitting learning plateaus. We argue this is because over long training runs, the Conjecturer learns to hack its reward, collapsing to artificially complex problems that do not help the Solver improve. To overcome this, we introduce Self-Guided Self-Play (SGS), a self-play algorithm in which the language model itself gui","authors_text":"Kaiyue Wen, Kefan Dong, Luke Bailey, Tatsunori Hashimoto, Tengyu Ma","cross_cats":[],"headline":"Self-guided self-play prevents language models from generating useless problems during training, allowing continued improvement on theorem proving.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-22T05:50:23Z","title":"Scaling Self-Play with Self-Guidance"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.20209","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-10T01:26:48.614239Z","id":"b57b7d9d-62a7-4516-87a1-1ade742792ea","model_set":{"reader":"grok-4.3"},"one_line_summary":"SGS adds self-guidance to LLM self-play for Lean4 theorem proving, surpassing RL baselines and enabling a 7B model to outperform a 671B model after 200 rounds.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Self-guided self-play prevents language models from generating useless problems during training, allowing continued improvement on theorem proving.","strongest_claim":"Applying SGS to formal theorem proving in Lean4, we find that it surpasses the asymptotic solve rate of our strongest RL baseline in fewer than 80 rounds of self-play and enables a 7B parameter model, after 200 rounds of self-play, to solve more problems than a 671B parameter model pass@4.","weakest_assumption":"Our core hypothesis is that language models can assess whether a subproblem is useful for achieving a goal."}},"verdict_id":"b57b7d9d-62a7-4516-87a1-1ade742792ea"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:622ad9b5711b131802da69fa60b13165dbfdf8cecb26cbc89133887b508a06dd","target":"record","created_at":"2026-08-12T01:24:20Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"df56fd36436c87a8539b0e963e7154b53daa97bbd620c1f9ed2762223e1be77f","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-22T05:50:23Z","title_canon_sha256":"044af8ce9dda481f70b29f85b5995f6ef1f8d7c79250e8ce9d5d389abecece86"},"schema_version":"1.0","source":{"id":"2604.20209","kind":"arxiv","version":2}},"canonical_sha256":"29056c44f9af0e69851ada5b77b059cf5fe35b3ea1cc36fa8af3a3d44591c159","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"29056c44f9af0e69851ada5b77b059cf5fe35b3ea1cc36fa8af3a3d44591c159","first_computed_at":"2026-08-12T01:24:20.547455Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-08-12T01:24:20.547455Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"JRlWT/0mon3AOCfeCY9uZWk9A0uMV0BZLPmi0dXvtek5/Ys0cKZwzzKmRaBnW7KATK7JcRLa/7yDZj+NolKPDA==","signature_status":"signed_v1","signed_at":"2026-08-12T01:24:20.549370Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.20209","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:622ad9b5711b131802da69fa60b13165dbfdf8cecb26cbc89133887b508a06dd","sha256:b6f4624ec6b4f26694674c9eef6187c03a1a9209e7d520d6a82b1bf72ec673cd"],"state_sha256":"bfa25c653383df427889e5617473975957acd8efd5d5ff1c79aa1a32c341e725"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Li60PlzEvNrNdMCyScedlLzPlElxiHfFaVSZZg5GNAYc2qkadi/AnRsT2BX9f839X6z7mAEkUZ4qQ0IXsbixCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-15T02:16:54.801010Z","bundle_sha256":"89677b60aa53564bbd0f41519e3ef7271515b27c284b184e7844bdabbc9edc93"}}