{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:GU5S4R6HARDPI3TCXNICD2HFP3","short_pith_number":"pith:GU5S4R6H","canonical_record":{"source":{"id":"2608.06735","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-07T02:59:53Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8a6f270189047beb8229474e67c271deda0b27a20d06ef00d72596b582c94c21","abstract_canon_sha256":"fb76376418dfadda6b79a5b51cfb7a5c9282a842364f694720ef854b97e7d7dc"},"schema_version":"1.0"},"canonical_sha256":"353b2e47c70446f46e62bb5021e8e57eef4edfc970f2c5aa0ff720892fce671f","source":{"kind":"arxiv","id":"2608.06735","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2608.06735","created_at":"2026-08-10T01:11:51Z"},{"alias_kind":"arxiv_version","alias_value":"2608.06735v1","created_at":"2026-08-10T01:11:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.06735","created_at":"2026-08-10T01:11:51Z"},{"alias_kind":"pith_short_12","alias_value":"GU5S4R6HARDP","created_at":"2026-08-10T01:11:51Z"},{"alias_kind":"pith_short_16","alias_value":"GU5S4R6HARDPI3TC","created_at":"2026-08-10T01:11:51Z"},{"alias_kind":"pith_short_8","alias_value":"GU5S4R6H","created_at":"2026-08-10T01:11:51Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:GU5S4R6HARDPI3TCXNICD2HFP3","target":"record","payload":{"canonical_record":{"source":{"id":"2608.06735","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-07T02:59:53Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8a6f270189047beb8229474e67c271deda0b27a20d06ef00d72596b582c94c21","abstract_canon_sha256":"fb76376418dfadda6b79a5b51cfb7a5c9282a842364f694720ef854b97e7d7dc"},"schema_version":"1.0"},"canonical_sha256":"353b2e47c70446f46e62bb5021e8e57eef4edfc970f2c5aa0ff720892fce671f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-10T01:11:51.884544Z","signature_b64":"ZhrQCfXfA6dY7zdrj2zBWcIjxnxi9O/71lRwdO4slpM7MvjnwrKZwU4+hVI1UzEhvWQEjEGKVsmYIw0zumpOBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"353b2e47c70446f46e62bb5021e8e57eef4edfc970f2c5aa0ff720892fce671f","last_reissued_at":"2026-08-10T01:11:51.882139Z","signature_status":"signed_v1","first_computed_at":"2026-08-10T01:11:51.882139Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2608.06735","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-10T01:11:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"aHyjapaFCa7a2GI7UTzuTZFf9KXq5p7e99utMa8Msl+gmvxsTDY+4O12eCNn1oCHBMeuGMAkxnYl3AN8aOjACA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T10:22:24.816147Z"},"content_sha256":"f53d7e975794855c2a9a3bdc934f5c0eda40e2e620aea130faa3c2fa5889ad17","schema_version":"1.0","event_id":"sha256:f53d7e975794855c2a9a3bdc934f5c0eda40e2e620aea130faa3c2fa5889ad17"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:GU5S4R6HARDPI3TCXNICD2HFP3","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"IB-RL: Isolated Bilateral Reinforcement Learning for Strategic Dialogue Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Chenghao Cai, Haitao Hu, Mingxing Huang, Senhao Wang, Wenhao Li, Xingguang Wang, Zecheng Lin","submitted_at":"2026-08-07T02:59:53Z","abstract_excerpt":"Reinforcement learning (RL) has achieved strong results in improving large language models (LLMs) on tasks with stationary, verifiable rewards, such as mathematical reasoning and code execution. In these settings, the environment follows fixed rules and does not adapt strategically to the agent. Strategic dialogue differs in this respect: the environment is another agent that adapts to the policy, and success depends on the interaction between the two sides. Despite this interactive nature, current RL approaches typically train a target agent against a fixed counterpart or simulator. We find t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.06735","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.06735/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-10T01:11:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0jdyIDX2D9Z647o63DVUvnEbP/0w3hdonZGx7dmud5L6HaiUJKQQx7L/yAEbP+Ynt/mWFiEAgRSW1DFugDWeDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T10:22:24.817117Z"},"content_sha256":"df1bd5b59d1d76602d37345dd154114416615a8789d26d8aa51c4eea942540a2","schema_version":"1.0","event_id":"sha256:df1bd5b59d1d76602d37345dd154114416615a8789d26d8aa51c4eea942540a2"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GU5S4R6HARDPI3TCXNICD2HFP3/bundle.json","state_url":"https://pith.science/pith/GU5S4R6HARDPI3TCXNICD2HFP3/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GU5S4R6HARDPI3TCXNICD2HFP3/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T10:22:24Z","links":{"resolver":"https://pith.science/pith/GU5S4R6HARDPI3TCXNICD2HFP3","bundle":"https://pith.science/pith/GU5S4R6HARDPI3TCXNICD2HFP3/bundle.json","state":"https://pith.science/pith/GU5S4R6HARDPI3TCXNICD2HFP3/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GU5S4R6HARDPI3TCXNICD2HFP3/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:GU5S4R6HARDPI3TCXNICD2HFP3","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"fb76376418dfadda6b79a5b51cfb7a5c9282a842364f694720ef854b97e7d7dc","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-07T02:59:53Z","title_canon_sha256":"8a6f270189047beb8229474e67c271deda0b27a20d06ef00d72596b582c94c21"},"schema_version":"1.0","source":{"id":"2608.06735","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2608.06735","created_at":"2026-08-10T01:11:51Z"},{"alias_kind":"arxiv_version","alias_value":"2608.06735v1","created_at":"2026-08-10T01:11:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.06735","created_at":"2026-08-10T01:11:51Z"},{"alias_kind":"pith_short_12","alias_value":"GU5S4R6HARDP","created_at":"2026-08-10T01:11:51Z"},{"alias_kind":"pith_short_16","alias_value":"GU5S4R6HARDPI3TC","created_at":"2026-08-10T01:11:51Z"},{"alias_kind":"pith_short_8","alias_value":"GU5S4R6H","created_at":"2026-08-10T01:11:51Z"}],"graph_snapshots":[{"event_id":"sha256:df1bd5b59d1d76602d37345dd154114416615a8789d26d8aa51c4eea942540a2","target":"graph","created_at":"2026-08-10T01:11:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2608.06735/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) has achieved strong results in improving large language models (LLMs) on tasks with stationary, verifiable rewards, such as mathematical reasoning and code execution. In these settings, the environment follows fixed rules and does not adapt strategically to the agent. Strategic dialogue differs in this respect: the environment is another agent that adapts to the policy, and success depends on the interaction between the two sides. Despite this interactive nature, current RL approaches typically train a target agent against a fixed counterpart or simulator. We find t","authors_text":"Chenghao Cai, Haitao Hu, Mingxing Huang, Senhao Wang, Wenhao Li, Xingguang Wang, Zecheng Lin","cross_cats":["cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-07T02:59:53Z","title":"IB-RL: Isolated Bilateral Reinforcement Learning for Strategic Dialogue Agents"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.06735","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f53d7e975794855c2a9a3bdc934f5c0eda40e2e620aea130faa3c2fa5889ad17","target":"record","created_at":"2026-08-10T01:11:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"fb76376418dfadda6b79a5b51cfb7a5c9282a842364f694720ef854b97e7d7dc","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-07T02:59:53Z","title_canon_sha256":"8a6f270189047beb8229474e67c271deda0b27a20d06ef00d72596b582c94c21"},"schema_version":"1.0","source":{"id":"2608.06735","kind":"arxiv","version":1}},"canonical_sha256":"353b2e47c70446f46e62bb5021e8e57eef4edfc970f2c5aa0ff720892fce671f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"353b2e47c70446f46e62bb5021e8e57eef4edfc970f2c5aa0ff720892fce671f","first_computed_at":"2026-08-10T01:11:51.882139Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-08-10T01:11:51.882139Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ZhrQCfXfA6dY7zdrj2zBWcIjxnxi9O/71lRwdO4slpM7MvjnwrKZwU4+hVI1UzEhvWQEjEGKVsmYIw0zumpOBw==","signature_status":"signed_v1","signed_at":"2026-08-10T01:11:51.884544Z","signed_message":"canonical_sha256_bytes"},"source_id":"2608.06735","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f53d7e975794855c2a9a3bdc934f5c0eda40e2e620aea130faa3c2fa5889ad17","sha256:df1bd5b59d1d76602d37345dd154114416615a8789d26d8aa51c4eea942540a2"],"state_sha256":"7ad0366ad84bb3493d60dc8b2e16f287aab1c98bcaf233a345cee78f1df129c8"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"v59t2CIACjr+C9A5ax2mMCeb7VLAVZUGHYTQP9ByM6tkTwNgKzo++NVEBQ0iTrloZOHG821zr8SxR6oYVcPdBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T10:22:24.824085Z","bundle_sha256":"e3a360b4d6855533108d27068d52340799434083e713654f6084bb67eb833512"}}