{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:GEXETFBFFIIN4SC7NUZMJLN47K","short_pith_number":"pith:GEXETFBF","canonical_record":{"source":{"id":"2103.06370","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-03-10T22:34:28Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"ddb6f6972cb0d4f9ca2971a7c07532734fd1e14a9306cf9b8e2674dbebe652ca","abstract_canon_sha256":"8b6222e28a7e99f06e460c73380ed7a4b79106c4e24cb773f11b1c27a622be68"},"schema_version":"1.0"},"canonical_sha256":"312e4994252a10de485f6d32c4adbcfab3361d61361eac7aa1be6daf2bce653e","source":{"kind":"arxiv","id":"2103.06370","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2103.06370","created_at":"2026-07-05T05:51:23Z"},{"alias_kind":"arxiv_version","alias_value":"2103.06370v1","created_at":"2026-07-05T05:51:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.06370","created_at":"2026-07-05T05:51:23Z"},{"alias_kind":"pith_short_12","alias_value":"GEXETFBFFIIN","created_at":"2026-07-05T05:51:23Z"},{"alias_kind":"pith_short_16","alias_value":"GEXETFBFFIIN4SC7","created_at":"2026-07-05T05:51:23Z"},{"alias_kind":"pith_short_8","alias_value":"GEXETFBF","created_at":"2026-07-05T05:51:23Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:GEXETFBFFIIN4SC7NUZMJLN47K","target":"record","payload":{"canonical_record":{"source":{"id":"2103.06370","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-03-10T22:34:28Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"ddb6f6972cb0d4f9ca2971a7c07532734fd1e14a9306cf9b8e2674dbebe652ca","abstract_canon_sha256":"8b6222e28a7e99f06e460c73380ed7a4b79106c4e24cb773f11b1c27a622be68"},"schema_version":"1.0"},"canonical_sha256":"312e4994252a10de485f6d32c4adbcfab3361d61361eac7aa1be6daf2bce653e","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:51:23.651763Z","signature_b64":"4yRqyOGIAuGVSiNb0IjqpjEsxAD73Nmrvu6TyrYhJC8IprJo9z+zlBYEvvo8wG+t0uSlRU/6pRc399f1cu7mCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"312e4994252a10de485f6d32c4adbcfab3361d61361eac7aa1be6daf2bce653e","last_reissued_at":"2026-07-05T05:51:23.651395Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:51:23.651395Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2103.06370","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:51:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"tOsOjB8yZT/CEMFjMQHnnF+2LyI5H7A2wH6p3UNw4nyguVWlY1ws9VZXoTBASCwrtLSU9WHhsV4oUbFnNMrdBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T15:28:39.941229Z"},"content_sha256":"d57aac7532f7465eaca3d3ca83932b47601f22c0d83350e172757980206bffc0","schema_version":"1.0","event_id":"sha256:d57aac7532f7465eaca3d3ca83932b47601f22c0d83350e172757980206bffc0"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:GEXETFBFFIIN4SC7NUZMJLN47K","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Causal-aware Safe Policy Improvement for Task-oriented dialogue","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Caiming Xiong, Govardana Sachithanandam Ramachandran, Kazuma Hashimoto","submitted_at":"2021-03-10T22:34:28Z","abstract_excerpt":"The recent success of reinforcement learning's (RL) in solving complex tasks is most often attributed to its capacity to explore and exploit an environment where it has been trained. Sample efficiency is usually not an issue since cheap simulators are available to sample data on-policy. On the other hand, task oriented dialogues are usually learnt from offline data collected using human demonstrations. Collecting diverse demonstrations and annotating them is expensive. Unfortunately, use of RL methods trained on off-policy data are prone to issues of bias and generalization, which are further "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.06370","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.06370/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:51:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HsDtY9CbUuhm+ZJBO0sXTrfzXDSZylxygM3vjVa+dPLkmbbNcjzdiS5T0SMpZ4GqD9cxc7gOK7D6Vkc1J6P7BA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T15:28:39.941885Z"},"content_sha256":"1bdaafa8fe224835c0840f817e5f668dea9c76387ff2aea1a1a354c43de7a111","schema_version":"1.0","event_id":"sha256:1bdaafa8fe224835c0840f817e5f668dea9c76387ff2aea1a1a354c43de7a111"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GEXETFBFFIIN4SC7NUZMJLN47K/bundle.json","state_url":"https://pith.science/pith/GEXETFBFFIIN4SC7NUZMJLN47K/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GEXETFBFFIIN4SC7NUZMJLN47K/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-20T15:28:39Z","links":{"resolver":"https://pith.science/pith/GEXETFBFFIIN4SC7NUZMJLN47K","bundle":"https://pith.science/pith/GEXETFBFFIIN4SC7NUZMJLN47K/bundle.json","state":"https://pith.science/pith/GEXETFBFFIIN4SC7NUZMJLN47K/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GEXETFBFFIIN4SC7NUZMJLN47K/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:GEXETFBFFIIN4SC7NUZMJLN47K","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8b6222e28a7e99f06e460c73380ed7a4b79106c4e24cb773f11b1c27a622be68","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-03-10T22:34:28Z","title_canon_sha256":"ddb6f6972cb0d4f9ca2971a7c07532734fd1e14a9306cf9b8e2674dbebe652ca"},"schema_version":"1.0","source":{"id":"2103.06370","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2103.06370","created_at":"2026-07-05T05:51:23Z"},{"alias_kind":"arxiv_version","alias_value":"2103.06370v1","created_at":"2026-07-05T05:51:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.06370","created_at":"2026-07-05T05:51:23Z"},{"alias_kind":"pith_short_12","alias_value":"GEXETFBFFIIN","created_at":"2026-07-05T05:51:23Z"},{"alias_kind":"pith_short_16","alias_value":"GEXETFBFFIIN4SC7","created_at":"2026-07-05T05:51:23Z"},{"alias_kind":"pith_short_8","alias_value":"GEXETFBF","created_at":"2026-07-05T05:51:23Z"}],"graph_snapshots":[{"event_id":"sha256:1bdaafa8fe224835c0840f817e5f668dea9c76387ff2aea1a1a354c43de7a111","target":"graph","created_at":"2026-07-05T05:51:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2103.06370/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The recent success of reinforcement learning's (RL) in solving complex tasks is most often attributed to its capacity to explore and exploit an environment where it has been trained. Sample efficiency is usually not an issue since cheap simulators are available to sample data on-policy. On the other hand, task oriented dialogues are usually learnt from offline data collected using human demonstrations. Collecting diverse demonstrations and annotating them is expensive. Unfortunately, use of RL methods trained on off-policy data are prone to issues of bias and generalization, which are further ","authors_text":"Caiming Xiong, Govardana Sachithanandam Ramachandran, Kazuma Hashimoto","cross_cats":["cs.AI","cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-03-10T22:34:28Z","title":"Causal-aware Safe Policy Improvement for Task-oriented dialogue"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.06370","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d57aac7532f7465eaca3d3ca83932b47601f22c0d83350e172757980206bffc0","target":"record","created_at":"2026-07-05T05:51:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8b6222e28a7e99f06e460c73380ed7a4b79106c4e24cb773f11b1c27a622be68","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-03-10T22:34:28Z","title_canon_sha256":"ddb6f6972cb0d4f9ca2971a7c07532734fd1e14a9306cf9b8e2674dbebe652ca"},"schema_version":"1.0","source":{"id":"2103.06370","kind":"arxiv","version":1}},"canonical_sha256":"312e4994252a10de485f6d32c4adbcfab3361d61361eac7aa1be6daf2bce653e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"312e4994252a10de485f6d32c4adbcfab3361d61361eac7aa1be6daf2bce653e","first_computed_at":"2026-07-05T05:51:23.651395Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:51:23.651395Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"4yRqyOGIAuGVSiNb0IjqpjEsxAD73Nmrvu6TyrYhJC8IprJo9z+zlBYEvvo8wG+t0uSlRU/6pRc399f1cu7mCg==","signature_status":"signed_v1","signed_at":"2026-07-05T05:51:23.651763Z","signed_message":"canonical_sha256_bytes"},"source_id":"2103.06370","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d57aac7532f7465eaca3d3ca83932b47601f22c0d83350e172757980206bffc0","sha256:1bdaafa8fe224835c0840f817e5f668dea9c76387ff2aea1a1a354c43de7a111"],"state_sha256":"a3207bd4a35ab858f960144d01394dd0e7b13ca4674c9f074eb8fd79d3cc85e7"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"j7V8nSm29EBU3KYqSxip445RT+ikMocqNYQCxgSkoWp8i5LgrFYlrYwJlYxbFUSRNN9xI11UxyBOiDNi+ExUAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-20T15:28:39.946696Z","bundle_sha256":"9f9a92ca1883fbfa07e847eaf422e3afa603899d0477916fa4115d8cfc70ddfc"}}