{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2A3Q6NSKT4M7ROBXTYUVZSVFFY","short_pith_number":"pith:2A3Q6NSK","schema_version":"1.0","canonical_sha256":"d0370f364a9f19f8b8379e295ccaa52e239c9ba9c6b8b9f286cc272cd4c0357e","source":{"kind":"arxiv","id":"2310.17784","version":2},"attestation_state":"computed","paper":{"title":"Data-Centric Financial Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Fei Yu, Hong Chen, Huaiyu Guo, Jun Zhou, Longfei Li, Qing Cui, Sheng Li, Wanqing Xu, Xin Lu, Xinyuan Zhou, Yijia Wang, Zhixuan Chu","submitted_at":"2023-10-07T04:53:31Z","abstract_excerpt":"Large language models (LLMs) show promise for natural language tasks but struggle when applied directly to complex domains like finance. LLMs have difficulty reasoning about and integrating all relevant information. We propose a data-centric approach to enable LLMs to better handle financial tasks. Our key insight is that rather than overloading the LLM with everything at once, it is more effective to preprocess and pre-understand the data. We create a financial LLM (FLLM) using multitask prompt-based finetuning to achieve data pre-processing and pre-understanding. However, labeled data is sca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.17784","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-07T04:53:31Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"fc163ab318a3b76ed6b72addab355b93a9cc4b14602948b9468c72cee00b5d3a","abstract_canon_sha256":"860ae753fe011f28fddfa19689354197cf9de856d76fef244a07a3d40ae4ebcd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:19.401409Z","signature_b64":"UU+FjMiLz+SdyUSOuH8ljiFD2aHZKVYzzYqSGT8BL1T3K/Me3HcKqJMPaxPQ07/kldDXCrAng0G1dyM3oJFbDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d0370f364a9f19f8b8379e295ccaa52e239c9ba9c6b8b9f286cc272cd4c0357e","last_reissued_at":"2026-07-05T07:12:19.400772Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:19.400772Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Data-Centric Financial Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Fei Yu, Hong Chen, Huaiyu Guo, Jun Zhou, Longfei Li, Qing Cui, Sheng Li, Wanqing Xu, Xin Lu, Xinyuan Zhou, Yijia Wang, Zhixuan Chu","submitted_at":"2023-10-07T04:53:31Z","abstract_excerpt":"Large language models (LLMs) show promise for natural language tasks but struggle when applied directly to complex domains like finance. LLMs have difficulty reasoning about and integrating all relevant information. We propose a data-centric approach to enable LLMs to better handle financial tasks. Our key insight is that rather than overloading the LLM with everything at once, it is more effective to preprocess and pre-understand the data. We create a financial LLM (FLLM) using multitask prompt-based finetuning to achieve data pre-processing and pre-understanding. However, labeled data is sca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.17784","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.17784/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.17784","created_at":"2026-07-05T07:12:19.400873+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.17784v2","created_at":"2026-07-05T07:12:19.400873+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.17784","created_at":"2026-07-05T07:12:19.400873+00:00"},{"alias_kind":"pith_short_12","alias_value":"2A3Q6NSKT4M7","created_at":"2026-07-05T07:12:19.400873+00:00"},{"alias_kind":"pith_short_16","alias_value":"2A3Q6NSKT4M7ROBX","created_at":"2026-07-05T07:12:19.400873+00:00"},{"alias_kind":"pith_short_8","alias_value":"2A3Q6NSK","created_at":"2026-07-05T07:12:19.400873+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27316","citing_title":"LLM-Based Examination of Eligibility Criteria from Securities Prospectuses at the German Central Bank","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2503.22693","citing_title":"Bridging Language Models and Financial Analysis","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2507.13334","citing_title":"A Survey of Context Engineering for Large Language Models","ref_index":183,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2A3Q6NSKT4M7ROBXTYUVZSVFFY","json":"https://pith.science/pith/2A3Q6NSKT4M7ROBXTYUVZSVFFY.json","graph_json":"https://pith.science/api/pith-number/2A3Q6NSKT4M7ROBXTYUVZSVFFY/graph.json","events_json":"https://pith.science/api/pith-number/2A3Q6NSKT4M7ROBXTYUVZSVFFY/events.json","paper":"https://pith.science/paper/2A3Q6NSK"},"agent_actions":{"view_html":"https://pith.science/pith/2A3Q6NSKT4M7ROBXTYUVZSVFFY","download_json":"https://pith.science/pith/2A3Q6NSKT4M7ROBXTYUVZSVFFY.json","view_paper":"https://pith.science/paper/2A3Q6NSK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.17784&json=true","fetch_graph":"https://pith.science/api/pith-number/2A3Q6NSKT4M7ROBXTYUVZSVFFY/graph.json","fetch_events":"https://pith.science/api/pith-number/2A3Q6NSKT4M7ROBXTYUVZSVFFY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2A3Q6NSKT4M7ROBXTYUVZSVFFY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2A3Q6NSKT4M7ROBXTYUVZSVFFY/action/storage_attestation","attest_author":"https://pith.science/pith/2A3Q6NSKT4M7ROBXTYUVZSVFFY/action/author_attestation","sign_citation":"https://pith.science/pith/2A3Q6NSKT4M7ROBXTYUVZSVFFY/action/citation_signature","submit_replication":"https://pith.science/pith/2A3Q6NSKT4M7ROBXTYUVZSVFFY/action/replication_record"}},"created_at":"2026-07-05T07:12:19.400873+00:00","updated_at":"2026-07-05T07:12:19.400873+00:00"}