{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:RZV247E7VMX5O47XNLHYPWCGSF","short_pith_number":"pith:RZV247E7","canonical_record":{"source":{"id":"2607.26358","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T00:19:09Z","cross_cats_sorted":["cs.AI","cs.GT"],"title_canon_sha256":"f90122553d6647370be0430f0b5428ddafbf172b595b314be91277c74bf213e2","abstract_canon_sha256":"237c38b4a9a7be39020e29a25e444699f40f18dcc7cc542e592373a0c074ab39"},"schema_version":"1.0"},"canonical_sha256":"8e6bae7c9fab2fd773f76acf87d846917364787496d10c935325d03a0d1cab61","source":{"kind":"arxiv","id":"2607.26358","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.26358","created_at":"2026-07-30T01:18:11Z"},{"alias_kind":"arxiv_version","alias_value":"2607.26358v1","created_at":"2026-07-30T01:18:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.26358","created_at":"2026-07-30T01:18:11Z"},{"alias_kind":"pith_short_12","alias_value":"RZV247E7VMX5","created_at":"2026-07-30T01:18:11Z"},{"alias_kind":"pith_short_16","alias_value":"RZV247E7VMX5O47X","created_at":"2026-07-30T01:18:11Z"},{"alias_kind":"pith_short_8","alias_value":"RZV247E7","created_at":"2026-07-30T01:18:11Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:RZV247E7VMX5O47XNLHYPWCGSF","target":"record","payload":{"canonical_record":{"source":{"id":"2607.26358","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T00:19:09Z","cross_cats_sorted":["cs.AI","cs.GT"],"title_canon_sha256":"f90122553d6647370be0430f0b5428ddafbf172b595b314be91277c74bf213e2","abstract_canon_sha256":"237c38b4a9a7be39020e29a25e444699f40f18dcc7cc542e592373a0c074ab39"},"schema_version":"1.0"},"canonical_sha256":"8e6bae7c9fab2fd773f76acf87d846917364787496d10c935325d03a0d1cab61","receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8e6bae7c9fab2fd773f76acf87d846917364787496d10c935325d03a0d1cab61","last_reissued_at":"2026-07-30T01:18:11.802139Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-30T01:18:11.802139Z"},"source_kind":"arxiv","source_id":"2607.26358","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-30T01:18:11Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"tcdWxrzGIE+j3rAOsnzxk4c1eiosU5eeK2bZE5sBGVSyxHHIKoHaLpzucJXtADyCLxM5TaxbkXAss0iZibMwBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T11:18:13.826498Z"},"content_sha256":"9e439731ad3bc63832dbd4deff121d9bc626ccdec32525200dfe177cec6283b9","schema_version":"1.0","event_id":"sha256:9e439731ad3bc63832dbd4deff121d9bc626ccdec32525200dfe177cec6283b9"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:RZV247E7VMX5O47XNLHYPWCGSF","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Post-Training at the Edge of Detectability: A Game-Theoretic Approach to Fine-Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.GT"],"primary_cat":"cs.LG","authors_text":"Brian W. Lee, Ian Waudby-Smith, Keegan Harris, Michael I. Jordan, Nika Haghtalab, Philip Amortila","submitted_at":"2026-07-29T00:19:09Z","abstract_excerpt":"Reinforcement learning (RL) fine-tuning is widely used in language model training to improve model performance on a target task while limiting drift from a reference policy. A standard way to balance this trade-off is via a KL-regularized RL objective, although this formulation does not by itself provide a principled way to set the regularization coefficient. In practice, the coefficient is typically chosen heuristically or via hyperparameter search, which can lead to unnecessary overhead in training cost or undesirable reward-retention trade-offs. We instead propose a game-theoretic framework"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.26358","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.26358/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-30T01:18:11Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"runr9RZxoGGG5aDDm+JxA+ubQQoKuqwVjp4VeK1go2hZVV3g/mv50TdvXnx4dh5tBwvYSAfAk3Z5ST8+YP7wCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T11:18:13.827041Z"},"content_sha256":"cd31f76a60bb06a568d6f257f8e33dbd1eb68638d3dd177a85a638291f62657e","schema_version":"1.0","event_id":"sha256:cd31f76a60bb06a568d6f257f8e33dbd1eb68638d3dd177a85a638291f62657e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RZV247E7VMX5O47XNLHYPWCGSF/bundle.json","state_url":"https://pith.science/pith/RZV247E7VMX5O47XNLHYPWCGSF/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RZV247E7VMX5O47XNLHYPWCGSF/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-03T11:18:13Z","links":{"resolver":"https://pith.science/pith/RZV247E7VMX5O47XNLHYPWCGSF","bundle":"https://pith.science/pith/RZV247E7VMX5O47XNLHYPWCGSF/bundle.json","state":"https://pith.science/pith/RZV247E7VMX5O47XNLHYPWCGSF/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RZV247E7VMX5O47XNLHYPWCGSF/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:RZV247E7VMX5O47XNLHYPWCGSF","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"237c38b4a9a7be39020e29a25e444699f40f18dcc7cc542e592373a0c074ab39","cross_cats_sorted":["cs.AI","cs.GT"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T00:19:09Z","title_canon_sha256":"f90122553d6647370be0430f0b5428ddafbf172b595b314be91277c74bf213e2"},"schema_version":"1.0","source":{"id":"2607.26358","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.26358","created_at":"2026-07-30T01:18:11Z"},{"alias_kind":"arxiv_version","alias_value":"2607.26358v1","created_at":"2026-07-30T01:18:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.26358","created_at":"2026-07-30T01:18:11Z"},{"alias_kind":"pith_short_12","alias_value":"RZV247E7VMX5","created_at":"2026-07-30T01:18:11Z"},{"alias_kind":"pith_short_16","alias_value":"RZV247E7VMX5O47X","created_at":"2026-07-30T01:18:11Z"},{"alias_kind":"pith_short_8","alias_value":"RZV247E7","created_at":"2026-07-30T01:18:11Z"}],"graph_snapshots":[{"event_id":"sha256:cd31f76a60bb06a568d6f257f8e33dbd1eb68638d3dd177a85a638291f62657e","target":"graph","created_at":"2026-07-30T01:18:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.26358/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) fine-tuning is widely used in language model training to improve model performance on a target task while limiting drift from a reference policy. A standard way to balance this trade-off is via a KL-regularized RL objective, although this formulation does not by itself provide a principled way to set the regularization coefficient. In practice, the coefficient is typically chosen heuristically or via hyperparameter search, which can lead to unnecessary overhead in training cost or undesirable reward-retention trade-offs. We instead propose a game-theoretic framework","authors_text":"Brian W. Lee, Ian Waudby-Smith, Keegan Harris, Michael I. Jordan, Nika Haghtalab, Philip Amortila","cross_cats":["cs.AI","cs.GT"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T00:19:09Z","title":"Post-Training at the Edge of Detectability: A Game-Theoretic Approach to Fine-Tuning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.26358","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:9e439731ad3bc63832dbd4deff121d9bc626ccdec32525200dfe177cec6283b9","target":"record","created_at":"2026-07-30T01:18:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"237c38b4a9a7be39020e29a25e444699f40f18dcc7cc542e592373a0c074ab39","cross_cats_sorted":["cs.AI","cs.GT"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T00:19:09Z","title_canon_sha256":"f90122553d6647370be0430f0b5428ddafbf172b595b314be91277c74bf213e2"},"schema_version":"1.0","source":{"id":"2607.26358","kind":"arxiv","version":1}},"canonical_sha256":"8e6bae7c9fab2fd773f76acf87d846917364787496d10c935325d03a0d1cab61","receipt":{"builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8e6bae7c9fab2fd773f76acf87d846917364787496d10c935325d03a0d1cab61","first_computed_at":"2026-07-30T01:18:11.802139Z","kind":"pith_receipt","last_reissued_at":"2026-07-30T01:18:11.802139Z","receipt_version":"0.3","signature_status":"unsigned_v0"},"source_id":"2607.26358","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:9e439731ad3bc63832dbd4deff121d9bc626ccdec32525200dfe177cec6283b9","sha256:cd31f76a60bb06a568d6f257f8e33dbd1eb68638d3dd177a85a638291f62657e"],"state_sha256":"3b433cc58fdc6c476536c13efdf0260bcad53e2b36c6562a7cfeca9b116246aa"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UG28DW5+o3QFf9Iuggm34XSyBTuRmYZFBtu1uCudzwklG0Q0ypCdl7SLRgY0o8NwhP/k4Fglt9k9vagbU9SfCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-03T11:18:13.832722Z","bundle_sha256":"e153e89b9ac673a1dcbb224f59a9b8a514ced67f18b82b8b3ca5ae7616b6cf35"}}