{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:PJLWESZBJAIQ6N5JCNU3FYXT5V","short_pith_number":"pith:PJLWESZB","schema_version":"1.0","canonical_sha256":"7a57624b2148110f37a91369b2e2f3ed7c2914701222941d726a166e39975dc6","source":{"kind":"arxiv","id":"2210.03849","version":1},"attestation_state":"computed","paper":{"title":"ConvFinQA: Exploring the Chain of Numerical Reasoning in Conversational Finance Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Charese Smiley, Sameena Shah, Shiyang Li, William Yang Wang, Zhiqiang Ma, Zhiyu Chen","submitted_at":"2022-10-07T23:48:50Z","abstract_excerpt":"With the recent advance in large pre-trained language models, researchers have achieved record performances in NLP tasks that mostly focus on language pattern matching. The community is experiencing the shift of the challenge from how to model language to the imitation of complex reasoning abilities like human beings. In this work, we investigate the application domain of finance that involves real-world, complex numerical reasoning. We propose a new large-scale dataset, ConvFinQA, aiming to study the chain of numerical reasoning in conversational question answering. Our dataset poses great ch"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.03849","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-10-07T23:48:50Z","cross_cats_sorted":[],"title_canon_sha256":"07a20de7eadf12547bf59e98537eab387683780b70f8ff789c558d96ceb47fbd","abstract_canon_sha256":"0e15d89a8cf75b8e5e06485cd18373299282ef5887cd9ae98fc269060712c473"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:04:32.994618Z","signature_b64":"s++14De3qhif5Am0/cz6RzXA+b1J5kXV0SBuRQTJ0lq4mnGnSAZNl824mUH4zJbrnC+ZsLSfdFsjdHmVvGrMCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7a57624b2148110f37a91369b2e2f3ed7c2914701222941d726a166e39975dc6","last_reissued_at":"2026-07-05T05:04:32.994174Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:04:32.994174Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ConvFinQA: Exploring the Chain of Numerical Reasoning in Conversational Finance Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Charese Smiley, Sameena Shah, Shiyang Li, William Yang Wang, Zhiqiang Ma, Zhiyu Chen","submitted_at":"2022-10-07T23:48:50Z","abstract_excerpt":"With the recent advance in large pre-trained language models, researchers have achieved record performances in NLP tasks that mostly focus on language pattern matching. The community is experiencing the shift of the challenge from how to model language to the imitation of complex reasoning abilities like human beings. In this work, we investigate the application domain of finance that involves real-world, complex numerical reasoning. We propose a new large-scale dataset, ConvFinQA, aiming to study the chain of numerical reasoning in conversational question answering. Our dataset poses great ch"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.03849","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.03849/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.03849","created_at":"2026-07-05T05:04:32.994229+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.03849v1","created_at":"2026-07-05T05:04:32.994229+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.03849","created_at":"2026-07-05T05:04:32.994229+00:00"},{"alias_kind":"pith_short_12","alias_value":"PJLWESZBJAIQ","created_at":"2026-07-05T05:04:32.994229+00:00"},{"alias_kind":"pith_short_16","alias_value":"PJLWESZBJAIQ6N5J","created_at":"2026-07-05T05:04:32.994229+00:00"},{"alias_kind":"pith_short_8","alias_value":"PJLWESZB","created_at":"2026-07-05T05:04:32.994229+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25984","citing_title":"InvestPhilBench: A Multi-Layer Benchmark for Evaluating Large Language Model Procedural Reasoning in Expert Investment Philosophy","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31608","citing_title":"CLExEval: A Human-in-the-Loop Framework for Qualitative Evaluation of LLM Clinical Reasoning","ref_index":129,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15482","citing_title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2503.22693","citing_title":"Bridging Language Models and Financial Analysis","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20650","citing_title":"FinTagging: Benchmarking LLMs for Extracting and Structuring Financial Information","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2510.08886","citing_title":"FinAuditing: A Financial Taxonomy-Structured Multi-Document Benchmark for Evaluating LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15482","citing_title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2211.12588","citing_title":"Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning Tasks","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21495","citing_title":"Generalizing Numerical Reasoning in Table Data through Operation Sketches and Self-Supervised Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24334","citing_title":"Reducing Redundancy in Retrieval-Augmented Generation through Chunk Filtering","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PJLWESZBJAIQ6N5JCNU3FYXT5V","json":"https://pith.science/pith/PJLWESZBJAIQ6N5JCNU3FYXT5V.json","graph_json":"https://pith.science/api/pith-number/PJLWESZBJAIQ6N5JCNU3FYXT5V/graph.json","events_json":"https://pith.science/api/pith-number/PJLWESZBJAIQ6N5JCNU3FYXT5V/events.json","paper":"https://pith.science/paper/PJLWESZB"},"agent_actions":{"view_html":"https://pith.science/pith/PJLWESZBJAIQ6N5JCNU3FYXT5V","download_json":"https://pith.science/pith/PJLWESZBJAIQ6N5JCNU3FYXT5V.json","view_paper":"https://pith.science/paper/PJLWESZB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.03849&json=true","fetch_graph":"https://pith.science/api/pith-number/PJLWESZBJAIQ6N5JCNU3FYXT5V/graph.json","fetch_events":"https://pith.science/api/pith-number/PJLWESZBJAIQ6N5JCNU3FYXT5V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PJLWESZBJAIQ6N5JCNU3FYXT5V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PJLWESZBJAIQ6N5JCNU3FYXT5V/action/storage_attestation","attest_author":"https://pith.science/pith/PJLWESZBJAIQ6N5JCNU3FYXT5V/action/author_attestation","sign_citation":"https://pith.science/pith/PJLWESZBJAIQ6N5JCNU3FYXT5V/action/citation_signature","submit_replication":"https://pith.science/pith/PJLWESZBJAIQ6N5JCNU3FYXT5V/action/replication_record"}},"created_at":"2026-07-05T05:04:32.994229+00:00","updated_at":"2026-07-05T05:04:32.994229+00:00"}