{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5RHF46A3SLP4D35ATRD6I5NCSV","short_pith_number":"pith:5RHF46A3","schema_version":"1.0","canonical_sha256":"ec4e5e781b92dfc1efa09c47e475a2955f37f44ced641d8f2b25668d649e83a3","source":{"kind":"arxiv","id":"2502.06329","version":1},"attestation_state":"computed","paper":{"title":"Expect the Unexpected: FailSafe Long Context QA for Finance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dmytro Mozolevskyi, Kiran Kamble, Mateusz Russak, Melisa Russak, Muayad Ali, Waseem Alshikh","submitted_at":"2025-02-10T10:29:28Z","abstract_excerpt":"We propose a new long-context financial benchmark, FailSafeQA, designed to test the robustness and context-awareness of LLMs against six variations in human-interface interactions in LLM-based query-answer systems within finance. We concentrate on two case studies: Query Failure and Context Failure. In the Query Failure scenario, we perturb the original query to vary in domain expertise, completeness, and linguistic accuracy. In the Context Failure case, we simulate the uploads of degraded, irrelevant, and empty documents. We employ the LLM-as-a-Judge methodology with Qwen2.5-72B-Instruct and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06329","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-10T10:29:28Z","cross_cats_sorted":[],"title_canon_sha256":"d941963342174ef44dd53788a2af00136e210be89f2df922200c4c4ba87efccf","abstract_canon_sha256":"fbe056d9506b42e42ed2492890dbd448aee5dfa84ed562797262210160a3e8a4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:05.147077Z","signature_b64":"Wh8wBJHikgvN0mjuFXDAocnPP8oQyTwGFzfpdf5tdwlswTJtRFwyBv6267XT2JU4fWPxKgYBRNttMJcodYq0Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec4e5e781b92dfc1efa09c47e475a2955f37f44ced641d8f2b25668d649e83a3","last_reissued_at":"2026-07-05T10:12:05.146568Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:05.146568Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Expect the Unexpected: FailSafe Long Context QA for Finance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dmytro Mozolevskyi, Kiran Kamble, Mateusz Russak, Melisa Russak, Muayad Ali, Waseem Alshikh","submitted_at":"2025-02-10T10:29:28Z","abstract_excerpt":"We propose a new long-context financial benchmark, FailSafeQA, designed to test the robustness and context-awareness of LLMs against six variations in human-interface interactions in LLM-based query-answer systems within finance. We concentrate on two case studies: Query Failure and Context Failure. In the Query Failure scenario, we perturb the original query to vary in domain expertise, completeness, and linguistic accuracy. In the Context Failure case, we simulate the uploads of degraded, irrelevant, and empty documents. We employ the LLM-as-a-Judge methodology with Qwen2.5-72B-Instruct and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06329","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06329","created_at":"2026-07-05T10:12:05.146631+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06329v1","created_at":"2026-07-05T10:12:05.146631+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06329","created_at":"2026-07-05T10:12:05.146631+00:00"},{"alias_kind":"pith_short_12","alias_value":"5RHF46A3SLP4","created_at":"2026-07-05T10:12:05.146631+00:00"},{"alias_kind":"pith_short_16","alias_value":"5RHF46A3SLP4D35A","created_at":"2026-07-05T10:12:05.146631+00:00"},{"alias_kind":"pith_short_8","alias_value":"5RHF46A3","created_at":"2026-07-05T10:12:05.146631+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.19906","citing_title":"GTAC: A Generative Transformer for Approximate Circuits","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5RHF46A3SLP4D35ATRD6I5NCSV","json":"https://pith.science/pith/5RHF46A3SLP4D35ATRD6I5NCSV.json","graph_json":"https://pith.science/api/pith-number/5RHF46A3SLP4D35ATRD6I5NCSV/graph.json","events_json":"https://pith.science/api/pith-number/5RHF46A3SLP4D35ATRD6I5NCSV/events.json","paper":"https://pith.science/paper/5RHF46A3"},"agent_actions":{"view_html":"https://pith.science/pith/5RHF46A3SLP4D35ATRD6I5NCSV","download_json":"https://pith.science/pith/5RHF46A3SLP4D35ATRD6I5NCSV.json","view_paper":"https://pith.science/paper/5RHF46A3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06329&json=true","fetch_graph":"https://pith.science/api/pith-number/5RHF46A3SLP4D35ATRD6I5NCSV/graph.json","fetch_events":"https://pith.science/api/pith-number/5RHF46A3SLP4D35ATRD6I5NCSV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5RHF46A3SLP4D35ATRD6I5NCSV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5RHF46A3SLP4D35ATRD6I5NCSV/action/storage_attestation","attest_author":"https://pith.science/pith/5RHF46A3SLP4D35ATRD6I5NCSV/action/author_attestation","sign_citation":"https://pith.science/pith/5RHF46A3SLP4D35ATRD6I5NCSV/action/citation_signature","submit_replication":"https://pith.science/pith/5RHF46A3SLP4D35ATRD6I5NCSV/action/replication_record"}},"created_at":"2026-07-05T10:12:05.146631+00:00","updated_at":"2026-07-05T10:12:05.146631+00:00"}