{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FFWK224RQLGOKU5U4HYJJPZ25J","short_pith_number":"pith:FFWK224R","schema_version":"1.0","canonical_sha256":"296cad6b9182cce553b4e1f094bf3aea4035c12b4dae758a4aec0cd13ac7694f","source":{"kind":"arxiv","id":"2412.18174","version":1},"attestation_state":"computed","paper":{"title":"INVESTORBENCH: A Benchmark for Financial Decision-Making Tasks with LLM-based Agent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","q-fin.CP"],"primary_cat":"cs.CE","authors_text":"Guojun Xiong, Haohang Li, Jimin Huang, Jordan W. Suchow, Koduvayur Subbalakshmi, Lingfei Qian, Qianqian Xie, Shashidhar Reddy Javaji, Xueqing Peng, Yangyang Yu, Yuechen Jiang, Yueru He, Yupeng Cao, Zhiyang Deng, Zining Zhu","submitted_at":"2024-12-24T05:22:33Z","abstract_excerpt":"Recent advancements have underscored the potential of large language model (LLM)-based agents in financial decision-making. Despite this progress, the field currently encounters two main challenges: (1) the lack of a comprehensive LLM agent framework adaptable to a variety of financial tasks, and (2) the absence of standardized benchmarks and consistent datasets for assessing agent performance. To tackle these issues, we introduce \\textsc{InvestorBench}, the first benchmark specifically designed for evaluating LLM-based agents in diverse financial decision-making contexts. InvestorBench enhanc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.18174","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CE","submitted_at":"2024-12-24T05:22:33Z","cross_cats_sorted":["cs.AI","q-fin.CP"],"title_canon_sha256":"5c94f03f19ed4991b3be6ffc6256789b6fd58550e9a154993e836ba3dcf6af4f","abstract_canon_sha256":"b993a931000a63ccef2c2f5660373aee39b36ab8dfbc9965fcbdd64ce2daff98"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:53:51.089440Z","signature_b64":"4kakyMtlMpzUBQfeh+c8JXxrvaDx9xnidZzr9fTptCEhKQgyuR0XRWZNSuK92OS3xwqNf6RAV08Se0W175yaBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"296cad6b9182cce553b4e1f094bf3aea4035c12b4dae758a4aec0cd13ac7694f","last_reissued_at":"2026-07-05T09:53:51.088966Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:53:51.088966Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"INVESTORBENCH: A Benchmark for Financial Decision-Making Tasks with LLM-based Agent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","q-fin.CP"],"primary_cat":"cs.CE","authors_text":"Guojun Xiong, Haohang Li, Jimin Huang, Jordan W. Suchow, Koduvayur Subbalakshmi, Lingfei Qian, Qianqian Xie, Shashidhar Reddy Javaji, Xueqing Peng, Yangyang Yu, Yuechen Jiang, Yueru He, Yupeng Cao, Zhiyang Deng, Zining Zhu","submitted_at":"2024-12-24T05:22:33Z","abstract_excerpt":"Recent advancements have underscored the potential of large language model (LLM)-based agents in financial decision-making. Despite this progress, the field currently encounters two main challenges: (1) the lack of a comprehensive LLM agent framework adaptable to a variety of financial tasks, and (2) the absence of standardized benchmarks and consistent datasets for assessing agent performance. To tackle these issues, we introduce \\textsc{InvestorBench}, the first benchmark specifically designed for evaluating LLM-based agents in diverse financial decision-making contexts. InvestorBench enhanc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.18174","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.18174/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.18174","created_at":"2026-07-05T09:53:51.089026+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.18174v1","created_at":"2026-07-05T09:53:51.089026+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.18174","created_at":"2026-07-05T09:53:51.089026+00:00"},{"alias_kind":"pith_short_12","alias_value":"FFWK224RQLGO","created_at":"2026-07-05T09:53:51.089026+00:00"},{"alias_kind":"pith_short_16","alias_value":"FFWK224RQLGOKU5U","created_at":"2026-07-05T09:53:51.089026+00:00"},{"alias_kind":"pith_short_8","alias_value":"FFWK224R","created_at":"2026-07-05T09:53:51.089026+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25984","citing_title":"InvestPhilBench: A Multi-Layer Benchmark for Evaluating Large Language Model Procedural Reasoning in Expert Investment Philosophy","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22719","citing_title":"Leakage-Aware Benchmarking of LLM Forecasting: Real-Time Nowcasts as the Decision-Time Input for Macro Factor Ranking","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08285","citing_title":"Beyond Agent Architecture: Execution Assumptions and Reproducibility in LLM-Based Trading Systems","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31608","citing_title":"CLExEval: A Human-in-the-Loop Framework for Qualitative Evaluation of LLM Clinical Reasoning","ref_index":167,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29771","citing_title":"CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17962","citing_title":"FinDocMRE: A Benchmark for Document-Level Financial Multimodal Reasoning Evaluation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27771","citing_title":"Emergent Social Intelligence Risks in Generative Multi-Agent Systems","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18500","citing_title":"QRAFTI: An Agentic Framework for Empirical Research in Quantitative Finance","ref_index":67,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FFWK224RQLGOKU5U4HYJJPZ25J","json":"https://pith.science/pith/FFWK224RQLGOKU5U4HYJJPZ25J.json","graph_json":"https://pith.science/api/pith-number/FFWK224RQLGOKU5U4HYJJPZ25J/graph.json","events_json":"https://pith.science/api/pith-number/FFWK224RQLGOKU5U4HYJJPZ25J/events.json","paper":"https://pith.science/paper/FFWK224R"},"agent_actions":{"view_html":"https://pith.science/pith/FFWK224RQLGOKU5U4HYJJPZ25J","download_json":"https://pith.science/pith/FFWK224RQLGOKU5U4HYJJPZ25J.json","view_paper":"https://pith.science/paper/FFWK224R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.18174&json=true","fetch_graph":"https://pith.science/api/pith-number/FFWK224RQLGOKU5U4HYJJPZ25J/graph.json","fetch_events":"https://pith.science/api/pith-number/FFWK224RQLGOKU5U4HYJJPZ25J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FFWK224RQLGOKU5U4HYJJPZ25J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FFWK224RQLGOKU5U4HYJJPZ25J/action/storage_attestation","attest_author":"https://pith.science/pith/FFWK224RQLGOKU5U4HYJJPZ25J/action/author_attestation","sign_citation":"https://pith.science/pith/FFWK224RQLGOKU5U4HYJJPZ25J/action/citation_signature","submit_replication":"https://pith.science/pith/FFWK224RQLGOKU5U4HYJJPZ25J/action/replication_record"}},"created_at":"2026-07-05T09:53:51.089026+00:00","updated_at":"2026-07-05T09:53:51.089026+00:00"}