{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:JSWMURCP2IHIJUH6O65KP6VOAS","short_pith_number":"pith:JSWMURCP","schema_version":"1.0","canonical_sha256":"4cacca444fd20e84d0fe77baa7faae049fb2749d0802843801c2fa7ad84fc5cc","source":{"kind":"arxiv","id":"2001.11770","version":1},"attestation_state":"computed","paper":{"title":"Break It Down: A Question Understanding Benchmark","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ankit Gupta, Daniel Deutch, Jonathan Berant, Matt Gardner, Mor Geva, Tomer Wolfson, Yoav Goldberg","submitted_at":"2020-01-31T11:04:52Z","abstract_excerpt":"Understanding natural language questions entails the ability to break down a question into the requisite steps for computing its answer. In this work, we introduce a Question Decomposition Meaning Representation (QDMR) for questions. QDMR constitutes the ordered list of steps, expressed through natural language, that are necessary for answering a question. We develop a crowdsourcing pipeline, showing that quality QDMRs can be annotated at scale, and release the Break dataset, containing over 83K pairs of questions and their QDMRs. We demonstrate the utility of QDMR by showing that (a) it can b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2001.11770","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-01-31T11:04:52Z","cross_cats_sorted":[],"title_canon_sha256":"7cebbd1da84c1c0d8387862b4021a31ef8640c97e23403f8b110744422bde7dd","abstract_canon_sha256":"0adea15ef2a842714051a7d590f1e85503d89700e8bd9c89d89c0c8f0f6e93da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:37:33.761162Z","signature_b64":"cJSFmjpbh66PqUBGJPveks/wKQhmc7cn/IkiWW+lP3oXM6rkK/FM2QY/XOWAtLVRao/p0brKPImQ1kv1PprnAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4cacca444fd20e84d0fe77baa7faae049fb2749d0802843801c2fa7ad84fc5cc","last_reissued_at":"2026-07-05T00:37:33.760704Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:37:33.760704Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Break It Down: A Question Understanding Benchmark","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ankit Gupta, Daniel Deutch, Jonathan Berant, Matt Gardner, Mor Geva, Tomer Wolfson, Yoav Goldberg","submitted_at":"2020-01-31T11:04:52Z","abstract_excerpt":"Understanding natural language questions entails the ability to break down a question into the requisite steps for computing its answer. In this work, we introduce a Question Decomposition Meaning Representation (QDMR) for questions. QDMR constitutes the ordered list of steps, expressed through natural language, that are necessary for answering a question. We develop a crowdsourcing pipeline, showing that quality QDMRs can be annotated at scale, and release the Break dataset, containing over 83K pairs of questions and their QDMRs. We demonstrate the utility of QDMR by showing that (a) it can b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2001.11770","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2001.11770/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2001.11770","created_at":"2026-07-05T00:37:33.760760+00:00"},{"alias_kind":"arxiv_version","alias_value":"2001.11770v1","created_at":"2026-07-05T00:37:33.760760+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2001.11770","created_at":"2026-07-05T00:37:33.760760+00:00"},{"alias_kind":"pith_short_12","alias_value":"JSWMURCP2IHI","created_at":"2026-07-05T00:37:33.760760+00:00"},{"alias_kind":"pith_short_16","alias_value":"JSWMURCP2IHIJUH6","created_at":"2026-07-05T00:37:33.760760+00:00"},{"alias_kind":"pith_short_8","alias_value":"JSWMURCP","created_at":"2026-07-05T00:37:33.760760+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2205.00445","citing_title":"MRKL Systems: A modular, neuro-symbolic architecture that combines large language models, external knowledge sources and discrete reasoning","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JSWMURCP2IHIJUH6O65KP6VOAS","json":"https://pith.science/pith/JSWMURCP2IHIJUH6O65KP6VOAS.json","graph_json":"https://pith.science/api/pith-number/JSWMURCP2IHIJUH6O65KP6VOAS/graph.json","events_json":"https://pith.science/api/pith-number/JSWMURCP2IHIJUH6O65KP6VOAS/events.json","paper":"https://pith.science/paper/JSWMURCP"},"agent_actions":{"view_html":"https://pith.science/pith/JSWMURCP2IHIJUH6O65KP6VOAS","download_json":"https://pith.science/pith/JSWMURCP2IHIJUH6O65KP6VOAS.json","view_paper":"https://pith.science/paper/JSWMURCP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2001.11770&json=true","fetch_graph":"https://pith.science/api/pith-number/JSWMURCP2IHIJUH6O65KP6VOAS/graph.json","fetch_events":"https://pith.science/api/pith-number/JSWMURCP2IHIJUH6O65KP6VOAS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JSWMURCP2IHIJUH6O65KP6VOAS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JSWMURCP2IHIJUH6O65KP6VOAS/action/storage_attestation","attest_author":"https://pith.science/pith/JSWMURCP2IHIJUH6O65KP6VOAS/action/author_attestation","sign_citation":"https://pith.science/pith/JSWMURCP2IHIJUH6O65KP6VOAS/action/citation_signature","submit_replication":"https://pith.science/pith/JSWMURCP2IHIJUH6O65KP6VOAS/action/replication_record"}},"created_at":"2026-07-05T00:37:33.760760+00:00","updated_at":"2026-07-05T00:37:33.760760+00:00"}