{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:XTBCM7Y25G5COC5ZKGIZTAAWPF","short_pith_number":"pith:XTBCM7Y2","canonical_record":{"source":{"id":"2604.27374","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-30T03:39:14Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"12843edef8b6ca23984ba8607a83c15dab27476bbb66df1de9186456acb742c0","abstract_canon_sha256":"889b924a68da930bea0140058a11da647e6ab00355c43b270d8220a7b07503a7"},"schema_version":"1.0"},"canonical_sha256":"bcc2267f1ae9ba270bb9519199801679637ce9177a2817917872a932d14d3bb1","source":{"kind":"arxiv","id":"2604.27374","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.27374","created_at":"2026-07-15T01:21:56Z"},{"alias_kind":"arxiv_version","alias_value":"2604.27374v2","created_at":"2026-07-15T01:21:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.27374","created_at":"2026-07-15T01:21:56Z"},{"alias_kind":"pith_short_12","alias_value":"XTBCM7Y25G5C","created_at":"2026-07-15T01:21:56Z"},{"alias_kind":"pith_short_16","alias_value":"XTBCM7Y25G5COC5Z","created_at":"2026-07-15T01:21:56Z"},{"alias_kind":"pith_short_8","alias_value":"XTBCM7Y2","created_at":"2026-07-15T01:21:56Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:XTBCM7Y25G5COC5ZKGIZTAAWPF","target":"record","payload":{"canonical_record":{"source":{"id":"2604.27374","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-30T03:39:14Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"12843edef8b6ca23984ba8607a83c15dab27476bbb66df1de9186456acb742c0","abstract_canon_sha256":"889b924a68da930bea0140058a11da647e6ab00355c43b270d8220a7b07503a7"},"schema_version":"1.0"},"canonical_sha256":"bcc2267f1ae9ba270bb9519199801679637ce9177a2817917872a932d14d3bb1","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-15T01:21:56.617598Z","signature_b64":"ZIrsSAlprVtiSblJk9n/6KZ8BcGAKTKzTggx2iU6EHXL6NTKk9dN1oFdYKS/zxECso1FanWywQD7Th/PhIhPAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bcc2267f1ae9ba270bb9519199801679637ce9177a2817917872a932d14d3bb1","last_reissued_at":"2026-07-15T01:21:56.616755Z","signature_status":"signed_v1","first_computed_at":"2026-07-15T01:21:56.616755Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.27374","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-15T01:21:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"GA+/ZhLP11KtoCKTmkbE44UAmu8BoC7OxBTMDT45NkSm9s/xJzI2m8uNkB42URwzOoYI7S7eCq8zqxWjWFYpCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-22T03:16:53.441902Z"},"content_sha256":"0cc681ab0de319de6ba88a0cf39b908260b810f560a008284137ecb121c0187a","schema_version":"1.0","event_id":"sha256:0cc681ab0de319de6ba88a0cf39b908260b810f560a008284137ecb121c0187a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:XTBCM7Y25G5COC5ZKGIZTAAWPF","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Measurement Risk in Supervised Financial NLP: Rubric and Metric Sensitivity on JF-ICR","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Rubric wording changes model labels on financial implicit-commitment data, with agreement between variants ranging from 70 to 83 percent.","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Peiying Zhu, Rongdong Chai, Sidi Chang, Yuxiao Chen","submitted_at":"2026-04-30T03:39:14Z","abstract_excerpt":"As LLMs become credible readers of earnings calls, investor-relations Q\\&A, guidance, and disclosure language, supervised financial NLP benchmarks increasingly function as decision evidence for model selection and deployment. A hidden assumption is that gold labels make such evidence objective. This assumption breaks down when the benchmark ruler itself is sensitive to rubric wording, metric choice, or aggregation policy. We study this measurement risk on Japanese Financial Implicit-Commitment Recognition (JF-ICR; a pinned 253-item test split x 4 frontier LLMs x 5 rubrics x 3 temperatures x 5 "},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Rubric wording materially changes model-assigned labels: R2--R3 agreement ranges from 70.0% to 83.4%, with the dominant movement near the +1 / 0 implicit-commitment boundary; ranking claims become more defensible only after this metric-identifiability audit.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The rubric variants are representative of real measurement risk despite confounding semantics, examples, and verbosity, and the observed patterns generalize beyond the pinned 253-item JF-ICR split.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Rubric changes shift implicit-commitment labels by 70-83% agreement and make some metrics uninformative, so model rankings only stabilize after a metric-identifiability audit.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Rubric wording changes model labels on financial implicit-commitment data, with agreement between variants ranging from 70 to 83 percent.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"78130072d01a1ba49bab66325293502b3273a03398607e6b4b8c26ee88d4a331"},"source":{"id":"2604.27374","kind":"arxiv","version":2},"verdict":{"id":"c3544c9b-470f-48b1-81ef-a4386c9aa217","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-07T08:59:34.473610Z","strongest_claim":"Rubric wording materially changes model-assigned labels: R2--R3 agreement ranges from 70.0% to 83.4%, with the dominant movement near the +1 / 0 implicit-commitment boundary; ranking claims become more defensible only after this metric-identifiability audit.","one_line_summary":"Rubric changes shift implicit-commitment labels by 70-83% agreement and make some metrics uninformative, so model rankings only stabilize after a metric-identifiability audit.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The rubric variants are representative of real measurement risk despite confounding semantics, examples, and verbosity, and the observed patterns generalize beyond the pinned 253-item JF-ICR split.","pith_extraction_headline":"Rubric wording changes model labels on financial implicit-commitment data, with agreement between variants ranging from 70 to 83 percent."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.27374/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-20T22:39:03.387127Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T19:16:37.841879Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"42521d27ad3498e1216dd3ce5bbd963004282d1313b7cf267665183bd710c34f"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"c3544c9b-470f-48b1-81ef-a4386c9aa217"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-15T01:21:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"cnKO7RKFKBXCzMJvTWCw8gA2nlFGYf3yuJGcr8sn8ttUqrPcFCJrdz02aF24qzB+MVBvbTdEv3RpNUSudSE3AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-22T03:16:53.442394Z"},"content_sha256":"cc9a012132362a57fba2024fbd1b04acd625467580029d85c42f9f29ea7604e3","schema_version":"1.0","event_id":"sha256:cc9a012132362a57fba2024fbd1b04acd625467580029d85c42f9f29ea7604e3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XTBCM7Y25G5COC5ZKGIZTAAWPF/bundle.json","state_url":"https://pith.science/pith/XTBCM7Y25G5COC5ZKGIZTAAWPF/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XTBCM7Y25G5COC5ZKGIZTAAWPF/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-22T03:16:53Z","links":{"resolver":"https://pith.science/pith/XTBCM7Y25G5COC5ZKGIZTAAWPF","bundle":"https://pith.science/pith/XTBCM7Y25G5COC5ZKGIZTAAWPF/bundle.json","state":"https://pith.science/pith/XTBCM7Y25G5COC5ZKGIZTAAWPF/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XTBCM7Y25G5COC5ZKGIZTAAWPF/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:XTBCM7Y25G5COC5ZKGIZTAAWPF","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"889b924a68da930bea0140058a11da647e6ab00355c43b270d8220a7b07503a7","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-30T03:39:14Z","title_canon_sha256":"12843edef8b6ca23984ba8607a83c15dab27476bbb66df1de9186456acb742c0"},"schema_version":"1.0","source":{"id":"2604.27374","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.27374","created_at":"2026-07-15T01:21:56Z"},{"alias_kind":"arxiv_version","alias_value":"2604.27374v2","created_at":"2026-07-15T01:21:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.27374","created_at":"2026-07-15T01:21:56Z"},{"alias_kind":"pith_short_12","alias_value":"XTBCM7Y25G5C","created_at":"2026-07-15T01:21:56Z"},{"alias_kind":"pith_short_16","alias_value":"XTBCM7Y25G5COC5Z","created_at":"2026-07-15T01:21:56Z"},{"alias_kind":"pith_short_8","alias_value":"XTBCM7Y2","created_at":"2026-07-15T01:21:56Z"}],"graph_snapshots":[{"event_id":"sha256:cc9a012132362a57fba2024fbd1b04acd625467580029d85c42f9f29ea7604e3","target":"graph","created_at":"2026-07-15T01:21:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Rubric wording materially changes model-assigned labels: R2--R3 agreement ranges from 70.0% to 83.4%, with the dominant movement near the +1 / 0 implicit-commitment boundary; ranking claims become more defensible only after this metric-identifiability audit."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The rubric variants are representative of real measurement risk despite confounding semantics, examples, and verbosity, and the observed patterns generalize beyond the pinned 253-item JF-ICR split."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Rubric changes shift implicit-commitment labels by 70-83% agreement and make some metrics uninformative, so model rankings only stabilize after a metric-identifiability audit."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Rubric wording changes model labels on financial implicit-commitment data, with agreement between variants ranging from 70 to 83 percent."}],"snapshot_sha256":"78130072d01a1ba49bab66325293502b3273a03398607e6b4b8c26ee88d4a331"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-20T22:39:03.387127Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T19:16:37.841879Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2604.27374/integrity.json","findings":[],"snapshot_sha256":"42521d27ad3498e1216dd3ce5bbd963004282d1313b7cf267665183bd710c34f","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"As LLMs become credible readers of earnings calls, investor-relations Q\\&A, guidance, and disclosure language, supervised financial NLP benchmarks increasingly function as decision evidence for model selection and deployment. A hidden assumption is that gold labels make such evidence objective. This assumption breaks down when the benchmark ruler itself is sensitive to rubric wording, metric choice, or aggregation policy. We study this measurement risk on Japanese Financial Implicit-Commitment Recognition (JF-ICR; a pinned 253-item test split x 4 frontier LLMs x 5 rubrics x 3 temperatures x 5 ","authors_text":"Peiying Zhu, Rongdong Chai, Sidi Chang, Yuxiao Chen","cross_cats":["cs.CL"],"headline":"Rubric wording changes model labels on financial implicit-commitment data, with agreement between variants ranging from 70 to 83 percent.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-30T03:39:14Z","title":"Measurement Risk in Supervised Financial NLP: Rubric and Metric Sensitivity on JF-ICR"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.27374","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-07T08:59:34.473610Z","id":"c3544c9b-470f-48b1-81ef-a4386c9aa217","model_set":{"reader":"grok-4.3"},"one_line_summary":"Rubric changes shift implicit-commitment labels by 70-83% agreement and make some metrics uninformative, so model rankings only stabilize after a metric-identifiability audit.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Rubric wording changes model labels on financial implicit-commitment data, with agreement between variants ranging from 70 to 83 percent.","strongest_claim":"Rubric wording materially changes model-assigned labels: R2--R3 agreement ranges from 70.0% to 83.4%, with the dominant movement near the +1 / 0 implicit-commitment boundary; ranking claims become more defensible only after this metric-identifiability audit.","weakest_assumption":"The rubric variants are representative of real measurement risk despite confounding semantics, examples, and verbosity, and the observed patterns generalize beyond the pinned 253-item JF-ICR split."}},"verdict_id":"c3544c9b-470f-48b1-81ef-a4386c9aa217"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:0cc681ab0de319de6ba88a0cf39b908260b810f560a008284137ecb121c0187a","target":"record","created_at":"2026-07-15T01:21:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"889b924a68da930bea0140058a11da647e6ab00355c43b270d8220a7b07503a7","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-30T03:39:14Z","title_canon_sha256":"12843edef8b6ca23984ba8607a83c15dab27476bbb66df1de9186456acb742c0"},"schema_version":"1.0","source":{"id":"2604.27374","kind":"arxiv","version":2}},"canonical_sha256":"bcc2267f1ae9ba270bb9519199801679637ce9177a2817917872a932d14d3bb1","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"bcc2267f1ae9ba270bb9519199801679637ce9177a2817917872a932d14d3bb1","first_computed_at":"2026-07-15T01:21:56.616755Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-15T01:21:56.616755Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ZIrsSAlprVtiSblJk9n/6KZ8BcGAKTKzTggx2iU6EHXL6NTKk9dN1oFdYKS/zxECso1FanWywQD7Th/PhIhPAA==","signature_status":"signed_v1","signed_at":"2026-07-15T01:21:56.617598Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.27374","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:0cc681ab0de319de6ba88a0cf39b908260b810f560a008284137ecb121c0187a","sha256:cc9a012132362a57fba2024fbd1b04acd625467580029d85c42f9f29ea7604e3"],"state_sha256":"73e86b1cc6ccef3b28c0b0e6006f995d3c9ba427189adc5653f19feb1369c6ac"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"wfFo6aEJs+sJSKFRqyildwlwtp3pm+gHeW4hQ2rGBvb+bj55duSj/R5nr8TmQ4sr6DKg8qfeIwfBXeO9f9MDDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-22T03:16:53.444796Z","bundle_sha256":"98b1888d8324fb7a43edce5bccae76ce2dc6b241655c29c0f574470c801e8eca"}}