{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2020:OULGDDJI573TZB7JQ5ZFGS7G3Z","short_pith_number":"pith:OULGDDJI","canonical_record":{"source":{"id":"2004.03267","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-04-07T11:03:17Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"633db6cacf3b8f29f8e02e6a4d99ce7ac9e49b69b55f5bca39b0520fe4d62544","abstract_canon_sha256":"0e3de047a5029d617fc63945b6e976209100ea1e3e736209b3fd71666e2d323d"},"schema_version":"1.0"},"canonical_sha256":"7516618d28eff73c87e98772534be6de7fcb53d20533f977aa5ca7ca6e7957cb","source":{"kind":"arxiv","id":"2004.03267","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2004.03267","created_at":"2026-07-05T01:35:57Z"},{"alias_kind":"arxiv_version","alias_value":"2004.03267v2","created_at":"2026-07-05T01:35:57Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.03267","created_at":"2026-07-05T01:35:57Z"},{"alias_kind":"pith_short_12","alias_value":"OULGDDJI573T","created_at":"2026-07-05T01:35:57Z"},{"alias_kind":"pith_short_16","alias_value":"OULGDDJI573TZB7J","created_at":"2026-07-05T01:35:57Z"},{"alias_kind":"pith_short_8","alias_value":"OULGDDJI","created_at":"2026-07-05T01:35:57Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2020:OULGDDJI573TZB7JQ5ZFGS7G3Z","target":"record","payload":{"canonical_record":{"source":{"id":"2004.03267","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-04-07T11:03:17Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"633db6cacf3b8f29f8e02e6a4d99ce7ac9e49b69b55f5bca39b0520fe4d62544","abstract_canon_sha256":"0e3de047a5029d617fc63945b6e976209100ea1e3e736209b3fd71666e2d323d"},"schema_version":"1.0"},"canonical_sha256":"7516618d28eff73c87e98772534be6de7fcb53d20533f977aa5ca7ca6e7957cb","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:35:57.635458Z","signature_b64":"Sj/ITiCAIBZEkbe2ij9b0ysz5/W+N78fWI7/Q53UmVeY5w2+WS4u4mF9GxRNuwrd3k+h1wLcLWJvEzNR0MqYAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7516618d28eff73c87e98772534be6de7fcb53d20533f977aa5ca7ca6e7957cb","last_reissued_at":"2026-07-05T01:35:57.635015Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:35:57.635015Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2004.03267","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T01:35:57Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VX5d3vCprec0xE9kFmS2Et9MAphexcPITxWUl0lt5L3N1PjRxdfly9ARUHEhLashMKUX8cEE8e9fEtMu69HJDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T15:57:09.827723Z"},"content_sha256":"922515786d84e66cbeb4eaaf79358ab27c6b3ea08b809af65ad19fab97decf55","schema_version":"1.0","event_id":"sha256:922515786d84e66cbeb4eaaf79358ab27c6b3ea08b809af65ad19fab97decf55"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2020:OULGDDJI573TZB7JQ5ZFGS7G3Z","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Guided Dialog Policy Learning without Adversarial Learning in the Loop","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Baolin Peng, Jianfeng Gao, Jinchao Li, Julia Kiseleva, Maarten de Rijke, Shahin Shayandeh, Sungjin Lee, Ziming Li","submitted_at":"2020-04-07T11:03:17Z","abstract_excerpt":"Reinforcement Learning (RL) methods have emerged as a popular choice for training an efficient and effective dialogue policy. However, these methods suffer from sparse and unstable reward signals returned by a user simulator only when a dialogue finishes. Besides, the reward signal is manually designed by human experts, which requires domain knowledge. Recently, a number of adversarial learning methods have been proposed to learn the reward function together with the dialogue policy. However, to alternatively update the dialogue policy and the reward model on the fly, we are limited to policy-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.03267","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.03267/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T01:35:57Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4uqkSpfJI5Vr7cMprrog1f+jjV05n5kNzaJV5OPnWRxfIsc9RbYvjvv5EKA3G38sgkzmonmTEGmX8PE6N/NLDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T15:57:09.828642Z"},"content_sha256":"aca9a72bccfe26d5186fb7b8583240c6420978b9169e447fdbe2b5617fc2ec14","schema_version":"1.0","event_id":"sha256:aca9a72bccfe26d5186fb7b8583240c6420978b9169e447fdbe2b5617fc2ec14"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/OULGDDJI573TZB7JQ5ZFGS7G3Z/bundle.json","state_url":"https://pith.science/pith/OULGDDJI573TZB7JQ5ZFGS7G3Z/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/OULGDDJI573TZB7JQ5ZFGS7G3Z/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-13T15:57:09Z","links":{"resolver":"https://pith.science/pith/OULGDDJI573TZB7JQ5ZFGS7G3Z","bundle":"https://pith.science/pith/OULGDDJI573TZB7JQ5ZFGS7G3Z/bundle.json","state":"https://pith.science/pith/OULGDDJI573TZB7JQ5ZFGS7G3Z/state.json","well_known_bundle":"https://pith.science/.well-known/pith/OULGDDJI573TZB7JQ5ZFGS7G3Z/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:OULGDDJI573TZB7JQ5ZFGS7G3Z","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0e3de047a5029d617fc63945b6e976209100ea1e3e736209b3fd71666e2d323d","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-04-07T11:03:17Z","title_canon_sha256":"633db6cacf3b8f29f8e02e6a4d99ce7ac9e49b69b55f5bca39b0520fe4d62544"},"schema_version":"1.0","source":{"id":"2004.03267","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2004.03267","created_at":"2026-07-05T01:35:57Z"},{"alias_kind":"arxiv_version","alias_value":"2004.03267v2","created_at":"2026-07-05T01:35:57Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.03267","created_at":"2026-07-05T01:35:57Z"},{"alias_kind":"pith_short_12","alias_value":"OULGDDJI573T","created_at":"2026-07-05T01:35:57Z"},{"alias_kind":"pith_short_16","alias_value":"OULGDDJI573TZB7J","created_at":"2026-07-05T01:35:57Z"},{"alias_kind":"pith_short_8","alias_value":"OULGDDJI","created_at":"2026-07-05T01:35:57Z"}],"graph_snapshots":[{"event_id":"sha256:aca9a72bccfe26d5186fb7b8583240c6420978b9169e447fdbe2b5617fc2ec14","target":"graph","created_at":"2026-07-05T01:35:57Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2004.03267/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning (RL) methods have emerged as a popular choice for training an efficient and effective dialogue policy. However, these methods suffer from sparse and unstable reward signals returned by a user simulator only when a dialogue finishes. Besides, the reward signal is manually designed by human experts, which requires domain knowledge. Recently, a number of adversarial learning methods have been proposed to learn the reward function together with the dialogue policy. However, to alternatively update the dialogue policy and the reward model on the fly, we are limited to policy-","authors_text":"Baolin Peng, Jianfeng Gao, Jinchao Li, Julia Kiseleva, Maarten de Rijke, Shahin Shayandeh, Sungjin Lee, Ziming Li","cross_cats":["cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-04-07T11:03:17Z","title":"Guided Dialog Policy Learning without Adversarial Learning in the Loop"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.03267","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:922515786d84e66cbeb4eaaf79358ab27c6b3ea08b809af65ad19fab97decf55","target":"record","created_at":"2026-07-05T01:35:57Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0e3de047a5029d617fc63945b6e976209100ea1e3e736209b3fd71666e2d323d","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2020-04-07T11:03:17Z","title_canon_sha256":"633db6cacf3b8f29f8e02e6a4d99ce7ac9e49b69b55f5bca39b0520fe4d62544"},"schema_version":"1.0","source":{"id":"2004.03267","kind":"arxiv","version":2}},"canonical_sha256":"7516618d28eff73c87e98772534be6de7fcb53d20533f977aa5ca7ca6e7957cb","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"7516618d28eff73c87e98772534be6de7fcb53d20533f977aa5ca7ca6e7957cb","first_computed_at":"2026-07-05T01:35:57.635015Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T01:35:57.635015Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Sj/ITiCAIBZEkbe2ij9b0ysz5/W+N78fWI7/Q53UmVeY5w2+WS4u4mF9GxRNuwrd3k+h1wLcLWJvEzNR0MqYAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T01:35:57.635458Z","signed_message":"canonical_sha256_bytes"},"source_id":"2004.03267","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:922515786d84e66cbeb4eaaf79358ab27c6b3ea08b809af65ad19fab97decf55","sha256:aca9a72bccfe26d5186fb7b8583240c6420978b9169e447fdbe2b5617fc2ec14"],"state_sha256":"738fe4acac6ca6a52727e651b322594d9f8cc37233d98c3d5d54c737392a0b2d"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0WWmR1j/JyoBvgLuLwoKUDlewhBunQB3qKFpwm5EtKMIeLYihSEHs67eeOQIitnxJUdAhYRW4lXT/4BC9FumBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-13T15:57:09.835937Z","bundle_sha256":"61eb7bc4d2a5bd37c13cfbf4e1c3ba90a414bf54525fe83d6fb663a19da15678"}}