{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:QYSZHH2ZDIUA5HPFRXHOSQ2T4K","short_pith_number":"pith:QYSZHH2Z","schema_version":"1.0","canonical_sha256":"8625939f591a280e9de58dcee94353e28fcf85f0f421c891e998ff5b71998cd5","source":{"kind":"arxiv","id":"2108.06712","version":3},"attestation_state":"computed","paper":{"title":"HiTab: A Hierarchical Table Dataset for Question Answering and Natural Language Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Dongmei Zhang, Haoyu Dong, Jian-Guang Lou, Jiaqi Guo, Ran Jia, Shi Han, Yan Gao, Zhiruo Wang, Zhoujun Cheng","submitted_at":"2021-08-15T10:14:21Z","abstract_excerpt":"Tables are often created with hierarchies, but existing works on table reasoning mainly focus on flat tables and neglect hierarchical tables. Hierarchical tables challenge existing methods by hierarchical indexing, as well as implicit relationships of calculation and semantics. This work presents HiTab, a free and open dataset to study question answering (QA) and natural language generation (NLG) over hierarchical tables. HiTab is a cross-domain dataset constructed from a wealth of statistical reports (analyses) and Wikipedia pages, and has unique characteristics: (1) nearly all tables are hie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.06712","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-08-15T10:14:21Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"5dff6ae6400c1b0b03f8508a27c2900f869f56602886fdaaee675cee0aae57f1","abstract_canon_sha256":"7d5c8d1bca445a13dbbb14da21d877894d2b1c0be96791a27c00c7b8ccc777d9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:08:39.931968Z","signature_b64":"88JFOrdYu8DkMDeZ13NI8KK8TZyMUs/yyiTsfplpLCU3+KF8BZiZs78Xb8RBf6Kb81Cqe+IV8c+CBj1gvWCrCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8625939f591a280e9de58dcee94353e28fcf85f0f421c891e998ff5b71998cd5","last_reissued_at":"2026-07-05T04:08:39.931536Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:08:39.931536Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HiTab: A Hierarchical Table Dataset for Question Answering and Natural Language Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Dongmei Zhang, Haoyu Dong, Jian-Guang Lou, Jiaqi Guo, Ran Jia, Shi Han, Yan Gao, Zhiruo Wang, Zhoujun Cheng","submitted_at":"2021-08-15T10:14:21Z","abstract_excerpt":"Tables are often created with hierarchies, but existing works on table reasoning mainly focus on flat tables and neglect hierarchical tables. Hierarchical tables challenge existing methods by hierarchical indexing, as well as implicit relationships of calculation and semantics. This work presents HiTab, a free and open dataset to study question answering (QA) and natural language generation (NLG) over hierarchical tables. HiTab is a cross-domain dataset constructed from a wealth of statistical reports (analyses) and Wikipedia pages, and has unique characteristics: (1) nearly all tables are hie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.06712","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.06712/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.06712","created_at":"2026-07-05T04:08:39.931589+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.06712v3","created_at":"2026-07-05T04:08:39.931589+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.06712","created_at":"2026-07-05T04:08:39.931589+00:00"},{"alias_kind":"pith_short_12","alias_value":"QYSZHH2ZDIUA","created_at":"2026-07-05T04:08:39.931589+00:00"},{"alias_kind":"pith_short_16","alias_value":"QYSZHH2ZDIUA5HPF","created_at":"2026-07-05T04:08:39.931589+00:00"},{"alias_kind":"pith_short_8","alias_value":"QYSZHH2Z","created_at":"2026-07-05T04:08:39.931589+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10316","citing_title":"TabClaw: An Interactive and Self-Evolving Agent for Spreadsheet Manipulation and Table Reasoning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2504.09925","citing_title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00309","citing_title":"Retrieval-Augmented Generation with Graphs (GraphRAG)","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":126,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QYSZHH2ZDIUA5HPFRXHOSQ2T4K","json":"https://pith.science/pith/QYSZHH2ZDIUA5HPFRXHOSQ2T4K.json","graph_json":"https://pith.science/api/pith-number/QYSZHH2ZDIUA5HPFRXHOSQ2T4K/graph.json","events_json":"https://pith.science/api/pith-number/QYSZHH2ZDIUA5HPFRXHOSQ2T4K/events.json","paper":"https://pith.science/paper/QYSZHH2Z"},"agent_actions":{"view_html":"https://pith.science/pith/QYSZHH2ZDIUA5HPFRXHOSQ2T4K","download_json":"https://pith.science/pith/QYSZHH2ZDIUA5HPFRXHOSQ2T4K.json","view_paper":"https://pith.science/paper/QYSZHH2Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.06712&json=true","fetch_graph":"https://pith.science/api/pith-number/QYSZHH2ZDIUA5HPFRXHOSQ2T4K/graph.json","fetch_events":"https://pith.science/api/pith-number/QYSZHH2ZDIUA5HPFRXHOSQ2T4K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QYSZHH2ZDIUA5HPFRXHOSQ2T4K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QYSZHH2ZDIUA5HPFRXHOSQ2T4K/action/storage_attestation","attest_author":"https://pith.science/pith/QYSZHH2ZDIUA5HPFRXHOSQ2T4K/action/author_attestation","sign_citation":"https://pith.science/pith/QYSZHH2ZDIUA5HPFRXHOSQ2T4K/action/citation_signature","submit_replication":"https://pith.science/pith/QYSZHH2ZDIUA5HPFRXHOSQ2T4K/action/replication_record"}},"created_at":"2026-07-05T04:08:39.931589+00:00","updated_at":"2026-07-05T04:08:39.931589+00:00"}