{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:7ZYPOSCVX23PMRBZCSTB5K3SHB","short_pith_number":"pith:7ZYPOSCV","schema_version":"1.0","canonical_sha256":"fe70f74855beb6f6443914a61eab723857b63293505753dda83ac5f01b900187","source":{"kind":"arxiv","id":"2607.27189","version":1},"attestation_state":"computed","paper":{"title":"APEX-Accounting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Austin Bennett, Bertie Vidgen, Brendan Foody, Charis Ching, Felix Mercier, Hayley Popiel, Jasmin Kern, Julien Benchek, Rene Sultan, Ryan Stevens, Vaibhav Mittal","submitted_at":"2026-07-29T17:56:49Z","abstract_excerpt":"We introduce APEX-Accounting, a benchmark built by Mercor in partnership with Ramp, to assess whether frontier models can do the real work of accountants. Tasks include reconciling accounts, accruing expenses, posting transactions, and producing reports. The private eval set comprises 160 tasks, split across 10 worlds. Each world contains an accounting system, as well as spreadsheets, PDFs, and other files. Every task was authored and solved by experts in accounting and bookkeeping, who also wrote grading rubrics. Across nine frontier models, Claude-Fable-5 (Max) leads with 56.4% Mean Criteria"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.27189","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-07-29T17:56:49Z","cross_cats_sorted":["cs.AI","cs.HC"],"title_canon_sha256":"6c8f30bb6200d64b52221f62711867df5efbf320d6ce283a4f9768a18ae06fdb","abstract_canon_sha256":"3480db57b6116a40e58cb930df13aa0adc24ff1bbdfec65c08aa09d56485be00"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fe70f74855beb6f6443914a61eab723857b63293505753dda83ac5f01b900187","last_reissued_at":"2026-07-30T01:24:01.008805Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-30T01:24:01.008805Z"},"graph_snapshot":{"paper":{"title":"APEX-Accounting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Austin Bennett, Bertie Vidgen, Brendan Foody, Charis Ching, Felix Mercier, Hayley Popiel, Jasmin Kern, Julien Benchek, Rene Sultan, Ryan Stevens, Vaibhav Mittal","submitted_at":"2026-07-29T17:56:49Z","abstract_excerpt":"We introduce APEX-Accounting, a benchmark built by Mercor in partnership with Ramp, to assess whether frontier models can do the real work of accountants. Tasks include reconciling accounts, accruing expenses, posting transactions, and producing reports. The private eval set comprises 160 tasks, split across 10 worlds. Each world contains an accounting system, as well as spreadsheets, PDFs, and other files. Every task was authored and solved by experts in accounting and bookkeeping, who also wrote grading rubrics. Across nine frontier models, Claude-Fable-5 (Max) leads with 56.4% Mean Criteria"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.27189","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.27189/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.27189","created_at":"2026-07-30T01:24:01.013847+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.27189v1","created_at":"2026-07-30T01:24:01.013847+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.27189","created_at":"2026-07-30T01:24:01.013847+00:00"},{"alias_kind":"pith_short_12","alias_value":"7ZYPOSCVX23P","created_at":"2026-07-30T01:24:01.013847+00:00"},{"alias_kind":"pith_short_16","alias_value":"7ZYPOSCVX23PMRBZ","created_at":"2026-07-30T01:24:01.013847+00:00"},{"alias_kind":"pith_short_8","alias_value":"7ZYPOSCV","created_at":"2026-07-30T01:24:01.013847+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7ZYPOSCVX23PMRBZCSTB5K3SHB","json":"https://pith.science/pith/7ZYPOSCVX23PMRBZCSTB5K3SHB.json","graph_json":"https://pith.science/api/pith-number/7ZYPOSCVX23PMRBZCSTB5K3SHB/graph.json","events_json":"https://pith.science/api/pith-number/7ZYPOSCVX23PMRBZCSTB5K3SHB/events.json","paper":"https://pith.science/paper/7ZYPOSCV"},"agent_actions":{"view_html":"https://pith.science/pith/7ZYPOSCVX23PMRBZCSTB5K3SHB","download_json":"https://pith.science/pith/7ZYPOSCVX23PMRBZCSTB5K3SHB.json","view_paper":"https://pith.science/paper/7ZYPOSCV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.27189&json=true","fetch_graph":"https://pith.science/api/pith-number/7ZYPOSCVX23PMRBZCSTB5K3SHB/graph.json","fetch_events":"https://pith.science/api/pith-number/7ZYPOSCVX23PMRBZCSTB5K3SHB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7ZYPOSCVX23PMRBZCSTB5K3SHB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7ZYPOSCVX23PMRBZCSTB5K3SHB/action/storage_attestation","attest_author":"https://pith.science/pith/7ZYPOSCVX23PMRBZCSTB5K3SHB/action/author_attestation","sign_citation":"https://pith.science/pith/7ZYPOSCVX23PMRBZCSTB5K3SHB/action/citation_signature","submit_replication":"https://pith.science/pith/7ZYPOSCVX23PMRBZCSTB5K3SHB/action/replication_record"}},"created_at":"2026-07-30T01:24:01.013847+00:00","updated_at":"2026-07-30T01:24:01.013847+00:00"}