{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZXZ66MMDDC5V7TJ2GX7BNW57UE","short_pith_number":"pith:ZXZ66MMD","schema_version":"1.0","canonical_sha256":"cdf3ef318318bb5fcd3a35fe16dbbfa100a9599c20e0c9fec209b0e485be5a86","source":{"kind":"arxiv","id":"2307.10485","version":2},"attestation_state":"computed","paper":{"title":"FinGPT: Democratizing Internet-scale Data for Financial Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","q-fin.GN"],"primary_cat":"cs.CL","authors_text":"Daochen Zha, Guoxuan Wang, Hongyang Yang, Xiao-Yang Liu","submitted_at":"2023-07-19T22:43:57Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable proficiency in understanding and generating human-like texts, which may potentially revolutionize the finance industry. However, existing LLMs often fall short in the financial field, which is mainly attributed to the disparities between general text data and financial text data. Unfortunately, there is only a limited number of financial text datasets available, and BloombergGPT, the first financial LLM (FinLLM), is close-sourced (only the training logs were released). In light of this, we aim to democratize Internet-scale financial dat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.10485","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-07-19T22:43:57Z","cross_cats_sorted":["cs.LG","q-fin.GN"],"title_canon_sha256":"f6c970c3ffbfbd1c09ee0bb32e9216c711daee09982bccb4ad7b4ecbb3ecd5dc","abstract_canon_sha256":"b3077e8428835f9f6fb251b8b2ef47f25698bc6eed4dfcc50c10a6c1316df488"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:36.052243Z","signature_b64":"lyqRaBRCWhkX+1pU5Yq5bFcvIUHRu14PWWo+DBcUWZe6SVsPgAaHq/eTdgsFhVzPScN2PYYbWL1y13yFSSrFDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cdf3ef318318bb5fcd3a35fe16dbbfa100a9599c20e0c9fec209b0e485be5a86","last_reissued_at":"2026-07-05T07:12:36.051770Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:36.051770Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FinGPT: Democratizing Internet-scale Data for Financial Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","q-fin.GN"],"primary_cat":"cs.CL","authors_text":"Daochen Zha, Guoxuan Wang, Hongyang Yang, Xiao-Yang Liu","submitted_at":"2023-07-19T22:43:57Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable proficiency in understanding and generating human-like texts, which may potentially revolutionize the finance industry. However, existing LLMs often fall short in the financial field, which is mainly attributed to the disparities between general text data and financial text data. Unfortunately, there is only a limited number of financial text datasets available, and BloombergGPT, the first financial LLM (FinLLM), is close-sourced (only the training logs were released). In light of this, we aim to democratize Internet-scale financial dat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.10485","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.10485/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.10485","created_at":"2026-07-05T07:12:36.051827+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.10485v2","created_at":"2026-07-05T07:12:36.051827+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.10485","created_at":"2026-07-05T07:12:36.051827+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZXZ66MMDDC5V","created_at":"2026-07-05T07:12:36.051827+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZXZ66MMDDC5V7TJ2","created_at":"2026-07-05T07:12:36.051827+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZXZ66MMD","created_at":"2026-07-05T07:12:36.051827+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20553","citing_title":"From Efficiency to Leakage -- Privacy Backdoor in Federated Language Model Fine-Tuning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31608","citing_title":"CLExEval: A Human-in-the-Loop Framework for Qualitative Evaluation of LLM Clinical Reasoning","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31461","citing_title":"CSTrader: A Testbed for Language-Grounded Trading in a Community-Driven Virtual Asset Market","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13779","citing_title":"MinT: Managed Infrastructure for Training and Serving Millions of LLMs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25030","citing_title":"MimirRAG: A Multi-Agent RAG Framework for Financial Data Retrieval with Metadata Integration","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02429","citing_title":"MulFSA: Multi-level Financial Sentiment Analysis Framework for Bond Market","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2506.05640","citing_title":"FedShield-LLM: A Secure and Scalable Federated Fine-Tuned Large Language Model","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18028","citing_title":"FedSDR: Federated Self-Distillation with Rectification","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09544","citing_title":"MetaGraph: A Large-Scale Meta-Analysis of GenAI in Financial NLP (2022-2025)","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13779","citing_title":"MinT: Managed Infrastructure for Training and Serving Millions of LLMs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09106","citing_title":"Fin-Bias: Comprehensive Evaluation for LLM Decision-Making under human bias in Finance Domain","ref_index":67,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZXZ66MMDDC5V7TJ2GX7BNW57UE","json":"https://pith.science/pith/ZXZ66MMDDC5V7TJ2GX7BNW57UE.json","graph_json":"https://pith.science/api/pith-number/ZXZ66MMDDC5V7TJ2GX7BNW57UE/graph.json","events_json":"https://pith.science/api/pith-number/ZXZ66MMDDC5V7TJ2GX7BNW57UE/events.json","paper":"https://pith.science/paper/ZXZ66MMD"},"agent_actions":{"view_html":"https://pith.science/pith/ZXZ66MMDDC5V7TJ2GX7BNW57UE","download_json":"https://pith.science/pith/ZXZ66MMDDC5V7TJ2GX7BNW57UE.json","view_paper":"https://pith.science/paper/ZXZ66MMD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.10485&json=true","fetch_graph":"https://pith.science/api/pith-number/ZXZ66MMDDC5V7TJ2GX7BNW57UE/graph.json","fetch_events":"https://pith.science/api/pith-number/ZXZ66MMDDC5V7TJ2GX7BNW57UE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZXZ66MMDDC5V7TJ2GX7BNW57UE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZXZ66MMDDC5V7TJ2GX7BNW57UE/action/storage_attestation","attest_author":"https://pith.science/pith/ZXZ66MMDDC5V7TJ2GX7BNW57UE/action/author_attestation","sign_citation":"https://pith.science/pith/ZXZ66MMDDC5V7TJ2GX7BNW57UE/action/citation_signature","submit_replication":"https://pith.science/pith/ZXZ66MMDDC5V7TJ2GX7BNW57UE/action/replication_record"}},"created_at":"2026-07-05T07:12:36.051827+00:00","updated_at":"2026-07-05T07:12:36.051827+00:00"}