{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RUBQPXMF3RNZ3MQGLQGVPJ7NJ6","short_pith_number":"pith:RUBQPXMF","schema_version":"1.0","canonical_sha256":"8d0307dd85dc5b9db2065c0d57a7ed4fbfa7f484b4d24c0d69819bcbb5b11b10","source":{"kind":"arxiv","id":"2503.16974","version":4},"attestation_state":"computed","paper":{"title":"Assessing Consistency and Reproducibility in the Outputs of Large Language Models: Evidence Across Diverse Finance and Accounting Tasks","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CE","cs.CL","cs.LG"],"primary_cat":"q-fin.GN","authors_text":"Julian Junyan Wang, Victor Xiaoqi Wang","submitted_at":"2025-03-21T09:43:37Z","abstract_excerpt":"This study provides the first comprehensive assessment of consistency and reproducibility in Large Language Model (LLM) outputs in finance and accounting research. We evaluate how consistently LLMs produce outputs given identical inputs through extensive experimentation with 50 independent runs across five common tasks: classification, sentiment analysis, summarization, text generation, and prediction. Using three OpenAI models (GPT-3.5-turbo, GPT-4o-mini, and GPT-4o), we generate over 3.4 million outputs from diverse financial source texts and data, covering MD&As, FOMC statements, finance ne"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.16974","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"q-fin.GN","submitted_at":"2025-03-21T09:43:37Z","cross_cats_sorted":["cs.AI","cs.CE","cs.CL","cs.LG"],"title_canon_sha256":"ee3b43a42ecaf41a62a3cb62a340e7cb8b92a7b2fd7ec986cd53fa1f98f29f8f","abstract_canon_sha256":"6406c0c84391cc607e0961095d91cac7b14d375de0229ce5c34883436f9c8cb6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:11:16.291690Z","signature_b64":"BqGtdl9PUpxxs/MVOIYiQ53CGa7DAUBzvh/8RmO+1hGMIlhR8GjqYia5l9A/TKRwfLDoBZaJjvuuo17W8+QoAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8d0307dd85dc5b9db2065c0d57a7ed4fbfa7f484b4d24c0d69819bcbb5b11b10","last_reissued_at":"2026-07-05T12:11:16.291109Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:11:16.291109Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Assessing Consistency and Reproducibility in the Outputs of Large Language Models: Evidence Across Diverse Finance and Accounting Tasks","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CE","cs.CL","cs.LG"],"primary_cat":"q-fin.GN","authors_text":"Julian Junyan Wang, Victor Xiaoqi Wang","submitted_at":"2025-03-21T09:43:37Z","abstract_excerpt":"This study provides the first comprehensive assessment of consistency and reproducibility in Large Language Model (LLM) outputs in finance and accounting research. We evaluate how consistently LLMs produce outputs given identical inputs through extensive experimentation with 50 independent runs across five common tasks: classification, sentiment analysis, summarization, text generation, and prediction. Using three OpenAI models (GPT-3.5-turbo, GPT-4o-mini, and GPT-4o), we generate over 3.4 million outputs from diverse financial source texts and data, covering MD&As, FOMC statements, finance ne"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.16974","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.16974/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.16974","created_at":"2026-07-05T12:11:16.291182+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.16974v4","created_at":"2026-07-05T12:11:16.291182+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.16974","created_at":"2026-07-05T12:11:16.291182+00:00"},{"alias_kind":"pith_short_12","alias_value":"RUBQPXMF3RNZ","created_at":"2026-07-05T12:11:16.291182+00:00"},{"alias_kind":"pith_short_16","alias_value":"RUBQPXMF3RNZ3MQG","created_at":"2026-07-05T12:11:16.291182+00:00"},{"alias_kind":"pith_short_8","alias_value":"RUBQPXMF","created_at":"2026-07-05T12:11:16.291182+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00856","citing_title":"Shapley in Context: Explaining Financial Language with Domain Expertise","ref_index":284,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23955","citing_title":"From Accuracy to Auditability: A Survey of Determinism in Financial AI Systems","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15134","citing_title":"The Unreasonable Effectiveness of Entropy Minimization in LLM Reasoning","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13413","citing_title":"Dataset-Level Metrics Attenuate Non-Determinism: A Fine-Grained Non-Determinism Evaluation in Diffusion Language Models","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6","json":"https://pith.science/pith/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6.json","graph_json":"https://pith.science/api/pith-number/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6/graph.json","events_json":"https://pith.science/api/pith-number/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6/events.json","paper":"https://pith.science/paper/RUBQPXMF"},"agent_actions":{"view_html":"https://pith.science/pith/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6","download_json":"https://pith.science/pith/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6.json","view_paper":"https://pith.science/paper/RUBQPXMF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.16974&json=true","fetch_graph":"https://pith.science/api/pith-number/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6/graph.json","fetch_events":"https://pith.science/api/pith-number/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6/action/storage_attestation","attest_author":"https://pith.science/pith/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6/action/author_attestation","sign_citation":"https://pith.science/pith/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6/action/citation_signature","submit_replication":"https://pith.science/pith/RUBQPXMF3RNZ3MQGLQGVPJ7NJ6/action/replication_record"}},"created_at":"2026-07-05T12:11:16.291182+00:00","updated_at":"2026-07-05T12:11:16.291182+00:00"}