{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:Q5455EGTMAMD6YDX6NPVMNXLFH","short_pith_number":"pith:Q5455EGT","canonical_record":{"source":{"id":"2406.14457","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-06-20T16:15:40Z","cross_cats_sorted":[],"title_canon_sha256":"a9e4ee242f910dcbc7c2eedbd54e20edeb77df38f5b933463dc7006873cbec6d","abstract_canon_sha256":"5c197c2409d43d4392693176df7858ad99da30bf0203f472510e16c10e6fd173"},"schema_version":"1.0"},"canonical_sha256":"8779de90d360183f6077f35f5636eb29e68f47f62d486dae4ea42fe0ad3da35a","source":{"kind":"arxiv","id":"2406.14457","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.14457","created_at":"2026-07-05T08:34:51Z"},{"alias_kind":"arxiv_version","alias_value":"2406.14457v1","created_at":"2026-07-05T08:34:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.14457","created_at":"2026-07-05T08:34:51Z"},{"alias_kind":"pith_short_12","alias_value":"Q5455EGTMAMD","created_at":"2026-07-05T08:34:51Z"},{"alias_kind":"pith_short_16","alias_value":"Q5455EGTMAMD6YDX","created_at":"2026-07-05T08:34:51Z"},{"alias_kind":"pith_short_8","alias_value":"Q5455EGT","created_at":"2026-07-05T08:34:51Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:Q5455EGTMAMD6YDX6NPVMNXLFH","target":"record","payload":{"canonical_record":{"source":{"id":"2406.14457","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-06-20T16:15:40Z","cross_cats_sorted":[],"title_canon_sha256":"a9e4ee242f910dcbc7c2eedbd54e20edeb77df38f5b933463dc7006873cbec6d","abstract_canon_sha256":"5c197c2409d43d4392693176df7858ad99da30bf0203f472510e16c10e6fd173"},"schema_version":"1.0"},"canonical_sha256":"8779de90d360183f6077f35f5636eb29e68f47f62d486dae4ea42fe0ad3da35a","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:51.040513Z","signature_b64":"PJdFF8QzyWq6R49EeMmS5FqzbStJr+uUCCZMCsumimGmaiCfxnxf81+l5ns5sEIcgigD7jmJRL1ccqZGlEwlDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8779de90d360183f6077f35f5636eb29e68f47f62d486dae4ea42fe0ad3da35a","last_reissued_at":"2026-07-05T08:34:51.040035Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:51.040035Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2406.14457","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:34:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Rm/ShyYYmgNa1k9ekBICFTJaj0kAKCLU+jADqvV8lpcKkcOhNnUsrS95JV2jQPRO09PJsTEAV6eQtN9ahYVRCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T21:47:45.547281Z"},"content_sha256":"1b01baa54b6f5790c0743b6d6a3b2bc2aa22b2dde4d744060fc674b2e50b486c","schema_version":"1.0","event_id":"sha256:1b01baa54b6f5790c0743b6d6a3b2bc2aa22b2dde4d744060fc674b2e50b486c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:Q5455EGTMAMD6YDX6NPVMNXLFH","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Rewarding What Matters: Step-by-Step Reinforcement Learning for Task-Oriented Dialogue","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Haofen Wang, Huifang Du, Minghao Wu, Shuqin Li, Xuejing Feng, Yuan-Fang Li","submitted_at":"2024-06-20T16:15:40Z","abstract_excerpt":"Reinforcement learning (RL) is a powerful approach to enhance task-oriented dialogue (TOD) systems. However, existing RL methods tend to mainly focus on generation tasks, such as dialogue policy learning (DPL) or response generation (RG), while neglecting dialogue state tracking (DST) for understanding. This narrow focus limits the systems to achieve globally optimal performance by overlooking the interdependence between understanding and generation. Additionally, RL methods face challenges with sparse and delayed rewards, which complicates training and optimization. To address these issues, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.14457","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.14457/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:34:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"lTz0pi0HolxpDt3gj4FqsASl7NYZ/CqQh+GQbfNhMbuCJzzWIqIyALcf73LVocFtxujTUNCQ5z+qDXLHNQ/mCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T21:47:45.547934Z"},"content_sha256":"146330db81c0891a85c788287a3fb84a3042a950b1205fcb14a564d423856057","schema_version":"1.0","event_id":"sha256:146330db81c0891a85c788287a3fb84a3042a950b1205fcb14a564d423856057"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/Q5455EGTMAMD6YDX6NPVMNXLFH/bundle.json","state_url":"https://pith.science/pith/Q5455EGTMAMD6YDX6NPVMNXLFH/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/Q5455EGTMAMD6YDX6NPVMNXLFH/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T21:47:45Z","links":{"resolver":"https://pith.science/pith/Q5455EGTMAMD6YDX6NPVMNXLFH","bundle":"https://pith.science/pith/Q5455EGTMAMD6YDX6NPVMNXLFH/bundle.json","state":"https://pith.science/pith/Q5455EGTMAMD6YDX6NPVMNXLFH/state.json","well_known_bundle":"https://pith.science/.well-known/pith/Q5455EGTMAMD6YDX6NPVMNXLFH/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:Q5455EGTMAMD6YDX6NPVMNXLFH","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5c197c2409d43d4392693176df7858ad99da30bf0203f472510e16c10e6fd173","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-06-20T16:15:40Z","title_canon_sha256":"a9e4ee242f910dcbc7c2eedbd54e20edeb77df38f5b933463dc7006873cbec6d"},"schema_version":"1.0","source":{"id":"2406.14457","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.14457","created_at":"2026-07-05T08:34:51Z"},{"alias_kind":"arxiv_version","alias_value":"2406.14457v1","created_at":"2026-07-05T08:34:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.14457","created_at":"2026-07-05T08:34:51Z"},{"alias_kind":"pith_short_12","alias_value":"Q5455EGTMAMD","created_at":"2026-07-05T08:34:51Z"},{"alias_kind":"pith_short_16","alias_value":"Q5455EGTMAMD6YDX","created_at":"2026-07-05T08:34:51Z"},{"alias_kind":"pith_short_8","alias_value":"Q5455EGT","created_at":"2026-07-05T08:34:51Z"}],"graph_snapshots":[{"event_id":"sha256:146330db81c0891a85c788287a3fb84a3042a950b1205fcb14a564d423856057","target":"graph","created_at":"2026-07-05T08:34:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2406.14457/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) is a powerful approach to enhance task-oriented dialogue (TOD) systems. However, existing RL methods tend to mainly focus on generation tasks, such as dialogue policy learning (DPL) or response generation (RG), while neglecting dialogue state tracking (DST) for understanding. This narrow focus limits the systems to achieve globally optimal performance by overlooking the interdependence between understanding and generation. Additionally, RL methods face challenges with sparse and delayed rewards, which complicates training and optimization. To address these issues, w","authors_text":"Haofen Wang, Huifang Du, Minghao Wu, Shuqin Li, Xuejing Feng, Yuan-Fang Li","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-06-20T16:15:40Z","title":"Rewarding What Matters: Step-by-Step Reinforcement Learning for Task-Oriented Dialogue"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.14457","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1b01baa54b6f5790c0743b6d6a3b2bc2aa22b2dde4d744060fc674b2e50b486c","target":"record","created_at":"2026-07-05T08:34:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5c197c2409d43d4392693176df7858ad99da30bf0203f472510e16c10e6fd173","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-06-20T16:15:40Z","title_canon_sha256":"a9e4ee242f910dcbc7c2eedbd54e20edeb77df38f5b933463dc7006873cbec6d"},"schema_version":"1.0","source":{"id":"2406.14457","kind":"arxiv","version":1}},"canonical_sha256":"8779de90d360183f6077f35f5636eb29e68f47f62d486dae4ea42fe0ad3da35a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8779de90d360183f6077f35f5636eb29e68f47f62d486dae4ea42fe0ad3da35a","first_computed_at":"2026-07-05T08:34:51.040035Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:34:51.040035Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"PJdFF8QzyWq6R49EeMmS5FqzbStJr+uUCCZMCsumimGmaiCfxnxf81+l5ns5sEIcgigD7jmJRL1ccqZGlEwlDg==","signature_status":"signed_v1","signed_at":"2026-07-05T08:34:51.040513Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.14457","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1b01baa54b6f5790c0743b6d6a3b2bc2aa22b2dde4d744060fc674b2e50b486c","sha256:146330db81c0891a85c788287a3fb84a3042a950b1205fcb14a564d423856057"],"state_sha256":"77c245920abcd3e6c935db28cd04f926533ae3d8c158508d29757a8884453431"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"R1ZmM6KF0j2a6mtmYFz9fbLxDUGodEjXIt3iy1iqq6dmrJAOoLnykSdV4K0Qy5rGlvcAfi0jaBsvqWSAwUvhAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T21:47:45.553287Z","bundle_sha256":"6cc2faff1c51e27b9935347f6d281e885a6dfb777faaac0891fc07329e721dcb"}}