{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:3VOTKBFQEUGK3FA3GLZRQRNWCB","short_pith_number":"pith:3VOTKBFQ","schema_version":"1.0","canonical_sha256":"dd5d3504b0250cad941b32f31845b61077459e348c549caad1c40ab26c971a53","source":{"kind":"arxiv","id":"2607.24889","version":1},"attestation_state":"computed","paper":{"title":"GAUGE: Grading Agent-Built Financial Models Without a Golden Answer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CE"],"primary_cat":"cs.LG","authors_text":"Beidi Luan, Cheng Hua, Daxin Jiang, Haibing Guan, Hui Cai, Jiacheng Lu, Jing Li, Lingjing Teng, Rui Sun, Sinuo Wang, Tao Song, Wentao Zhao, Yijia He, Zhengze Wu, Zuo Bai","submitted_at":"2026-07-27T13:03:19Z","abstract_excerpt":"Financial models combine public disclosures with analyst assumptions to produce forecasts and valuations. While some components can be checked mechanically, forecasts, discount rates, and target prices often admit multiple reasonable answers. Existing benchmarks nevertheless tend to grade such outputs against a single expert reference. Using independently built analyst models for the same companies, we find that across 108 directed pairs covering 65 companies, the median single-reference score is 0.33, 92.6% score below 0.70, and no same-vintage pair agrees on implied price within 10%. Point-t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.24889","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-27T13:03:19Z","cross_cats_sorted":["cs.AI","cs.CE"],"title_canon_sha256":"ea21b6b85984eae1d6e0884bce571cfb9b7998416394af85d8c4d48922cce959","abstract_canon_sha256":"ac727d0116dc0f59dbb954bb54f87215938df85470d31b2614d2ce15a439013f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-29T00:24:50.106202Z","signature_b64":"5NwbDIuCIeH177CQBIxTvrTnGPEYG1HmDyVVqJp10/koHemJp669F+5tgKi5c0SF6IojgCUkUbDHVPA3SYT5Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd5d3504b0250cad941b32f31845b61077459e348c549caad1c40ab26c971a53","last_reissued_at":"2026-07-29T00:24:50.105299Z","signature_status":"signed_v1","first_computed_at":"2026-07-29T00:24:50.105299Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GAUGE: Grading Agent-Built Financial Models Without a Golden Answer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CE"],"primary_cat":"cs.LG","authors_text":"Beidi Luan, Cheng Hua, Daxin Jiang, Haibing Guan, Hui Cai, Jiacheng Lu, Jing Li, Lingjing Teng, Rui Sun, Sinuo Wang, Tao Song, Wentao Zhao, Yijia He, Zhengze Wu, Zuo Bai","submitted_at":"2026-07-27T13:03:19Z","abstract_excerpt":"Financial models combine public disclosures with analyst assumptions to produce forecasts and valuations. While some components can be checked mechanically, forecasts, discount rates, and target prices often admit multiple reasonable answers. Existing benchmarks nevertheless tend to grade such outputs against a single expert reference. Using independently built analyst models for the same companies, we find that across 108 directed pairs covering 65 companies, the median single-reference score is 0.33, 92.6% score below 0.70, and no same-vintage pair agrees on implied price within 10%. Point-t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.24889","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.24889/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.24889","created_at":"2026-07-29T00:24:50.105757+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.24889v1","created_at":"2026-07-29T00:24:50.105757+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.24889","created_at":"2026-07-29T00:24:50.105757+00:00"},{"alias_kind":"pith_short_12","alias_value":"3VOTKBFQEUGK","created_at":"2026-07-29T00:24:50.105757+00:00"},{"alias_kind":"pith_short_16","alias_value":"3VOTKBFQEUGK3FA3","created_at":"2026-07-29T00:24:50.105757+00:00"},{"alias_kind":"pith_short_8","alias_value":"3VOTKBFQ","created_at":"2026-07-29T00:24:50.105757+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3VOTKBFQEUGK3FA3GLZRQRNWCB","json":"https://pith.science/pith/3VOTKBFQEUGK3FA3GLZRQRNWCB.json","graph_json":"https://pith.science/api/pith-number/3VOTKBFQEUGK3FA3GLZRQRNWCB/graph.json","events_json":"https://pith.science/api/pith-number/3VOTKBFQEUGK3FA3GLZRQRNWCB/events.json","paper":"https://pith.science/paper/3VOTKBFQ"},"agent_actions":{"view_html":"https://pith.science/pith/3VOTKBFQEUGK3FA3GLZRQRNWCB","download_json":"https://pith.science/pith/3VOTKBFQEUGK3FA3GLZRQRNWCB.json","view_paper":"https://pith.science/paper/3VOTKBFQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.24889&json=true","fetch_graph":"https://pith.science/api/pith-number/3VOTKBFQEUGK3FA3GLZRQRNWCB/graph.json","fetch_events":"https://pith.science/api/pith-number/3VOTKBFQEUGK3FA3GLZRQRNWCB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3VOTKBFQEUGK3FA3GLZRQRNWCB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3VOTKBFQEUGK3FA3GLZRQRNWCB/action/storage_attestation","attest_author":"https://pith.science/pith/3VOTKBFQEUGK3FA3GLZRQRNWCB/action/author_attestation","sign_citation":"https://pith.science/pith/3VOTKBFQEUGK3FA3GLZRQRNWCB/action/citation_signature","submit_replication":"https://pith.science/pith/3VOTKBFQEUGK3FA3GLZRQRNWCB/action/replication_record"}},"created_at":"2026-07-29T00:24:50.105757+00:00","updated_at":"2026-07-29T00:24:50.105757+00:00"}