{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:K5NZJG3O6CF7DKIEMOXLY65CPH","short_pith_number":"pith:K5NZJG3O","schema_version":"1.0","canonical_sha256":"575b949b6ef08bf1a90463aebc7ba279e3aef63f1801604781ed0e875fecd977","source":{"kind":"arxiv","id":"2007.12720","version":1},"attestation_state":"computed","paper":{"title":"MultiWOZ 2.2 : A Dialogue Dataset with Additional Annotation Corrections and State Tracking Baselines","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Abhinav Rastogi, Jianguo Zhang, Jindong Chen, Raghav Gupta, Srinivas Sunkara, Xiaoxue Zang","submitted_at":"2020-07-10T22:52:14Z","abstract_excerpt":"MultiWOZ is a well-known task-oriented dialogue dataset containing over 10,000 annotated dialogues spanning 8 domains. It is extensively used as a benchmark for dialogue state tracking. However, recent works have reported presence of substantial noise in the dialogue state annotations. MultiWOZ 2.1 identified and fixed many of these erroneous annotations and user utterances, resulting in an improved version of this dataset. This work introduces MultiWOZ 2.2, which is a yet another improved version of this dataset. Firstly, we identify and fix dialogue state annotation errors across 17.3% of th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2007.12720","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-07-10T22:52:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"94c0347360f40eb2cc14abcb1598695156cd5ffc1dca8c218d6d966c032353dc","abstract_canon_sha256":"1b495d63583a0f8dfdce6a1dae19a03ce12d0f65a6c83655bd71e507d60c5550"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:22:03.340615Z","signature_b64":"6hVT/d5Wze4lo0LUEXZXfIHTeeCNKwhjRqA6JgfiZT+wI2gamIpYUP2bqr8/M6aG/QzSqKDLP2PaXEJ0CTroCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"575b949b6ef08bf1a90463aebc7ba279e3aef63f1801604781ed0e875fecd977","last_reissued_at":"2026-07-05T01:22:03.340129Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:22:03.340129Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MultiWOZ 2.2 : A Dialogue Dataset with Additional Annotation Corrections and State Tracking Baselines","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Abhinav Rastogi, Jianguo Zhang, Jindong Chen, Raghav Gupta, Srinivas Sunkara, Xiaoxue Zang","submitted_at":"2020-07-10T22:52:14Z","abstract_excerpt":"MultiWOZ is a well-known task-oriented dialogue dataset containing over 10,000 annotated dialogues spanning 8 domains. It is extensively used as a benchmark for dialogue state tracking. However, recent works have reported presence of substantial noise in the dialogue state annotations. MultiWOZ 2.1 identified and fixed many of these erroneous annotations and user utterances, resulting in an improved version of this dataset. This work introduces MultiWOZ 2.2, which is a yet another improved version of this dataset. Firstly, we identify and fix dialogue state annotation errors across 17.3% of th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.12720","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.12720/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2007.12720","created_at":"2026-07-05T01:22:03.340186+00:00"},{"alias_kind":"arxiv_version","alias_value":"2007.12720v1","created_at":"2026-07-05T01:22:03.340186+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.12720","created_at":"2026-07-05T01:22:03.340186+00:00"},{"alias_kind":"pith_short_12","alias_value":"K5NZJG3O6CF7","created_at":"2026-07-05T01:22:03.340186+00:00"},{"alias_kind":"pith_short_16","alias_value":"K5NZJG3O6CF7DKIE","created_at":"2026-07-05T01:22:03.340186+00:00"},{"alias_kind":"pith_short_8","alias_value":"K5NZJG3O","created_at":"2026-07-05T01:22:03.340186+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.12266","citing_title":"The Behavior Gap: Evaluating Zero-shot LLM Agents in Complex Task-Oriented Dialogs","ref_index":2020,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K5NZJG3O6CF7DKIEMOXLY65CPH","json":"https://pith.science/pith/K5NZJG3O6CF7DKIEMOXLY65CPH.json","graph_json":"https://pith.science/api/pith-number/K5NZJG3O6CF7DKIEMOXLY65CPH/graph.json","events_json":"https://pith.science/api/pith-number/K5NZJG3O6CF7DKIEMOXLY65CPH/events.json","paper":"https://pith.science/paper/K5NZJG3O"},"agent_actions":{"view_html":"https://pith.science/pith/K5NZJG3O6CF7DKIEMOXLY65CPH","download_json":"https://pith.science/pith/K5NZJG3O6CF7DKIEMOXLY65CPH.json","view_paper":"https://pith.science/paper/K5NZJG3O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2007.12720&json=true","fetch_graph":"https://pith.science/api/pith-number/K5NZJG3O6CF7DKIEMOXLY65CPH/graph.json","fetch_events":"https://pith.science/api/pith-number/K5NZJG3O6CF7DKIEMOXLY65CPH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K5NZJG3O6CF7DKIEMOXLY65CPH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K5NZJG3O6CF7DKIEMOXLY65CPH/action/storage_attestation","attest_author":"https://pith.science/pith/K5NZJG3O6CF7DKIEMOXLY65CPH/action/author_attestation","sign_citation":"https://pith.science/pith/K5NZJG3O6CF7DKIEMOXLY65CPH/action/citation_signature","submit_replication":"https://pith.science/pith/K5NZJG3O6CF7DKIEMOXLY65CPH/action/replication_record"}},"created_at":"2026-07-05T01:22:03.340186+00:00","updated_at":"2026-07-05T01:22:03.340186+00:00"}