{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EWTS5MUAJZYYOBPBF25QMC56YL","short_pith_number":"pith:EWTS5MUA","schema_version":"1.0","canonical_sha256":"25a72eb2804e718705e12ebb060bbec2f70dc89f9e7e51f7c19c69833331c852","source":{"kind":"arxiv","id":"2502.16838","version":2},"attestation_state":"computed","paper":{"title":"REGen: A Reliable Evaluation Framework for Generative Event Argument Extraction","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Joseph Gatto, Madhusudan Basak, Omar Sharif, Sarah M. Preum","submitted_at":"2025-02-24T04:49:49Z","abstract_excerpt":"Event argument extraction identifies arguments for predefined event roles in text. Existing work evaluates this task with exact match (EM), where predicted arguments must align exactly with annotated spans. While suitable for span-based models, this approach falls short for large language models (LLMs), which often generate diverse yet semantically accurate arguments. EM severely underestimates performance by disregarding valid variations. Furthermore, EM evaluation fails to capture implicit arguments (unstated but inferable) and scattered arguments (distributed across a document). These limit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.16838","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-24T04:49:49Z","cross_cats_sorted":[],"title_canon_sha256":"28147a9bf8548f4e2d61eec0b15687cc56f3eca01907d9415df35e49761804b7","abstract_canon_sha256":"5a10bd6d7a3b121b2a5ae1f12ea3fd1c27a448ddb8108839dd911115ef2fd7a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:08:00.070947Z","signature_b64":"MlZiNkeNgO+AMUZUWHoNxAxi7ummYb8sUWK7X2REzBR/cgk5lMdGLAdsBsNAUR8xQvOr6xVZWTldkw4lEyhxDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25a72eb2804e718705e12ebb060bbec2f70dc89f9e7e51f7c19c69833331c852","last_reissued_at":"2026-07-05T12:08:00.070432Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:08:00.070432Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"REGen: A Reliable Evaluation Framework for Generative Event Argument Extraction","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Joseph Gatto, Madhusudan Basak, Omar Sharif, Sarah M. Preum","submitted_at":"2025-02-24T04:49:49Z","abstract_excerpt":"Event argument extraction identifies arguments for predefined event roles in text. Existing work evaluates this task with exact match (EM), where predicted arguments must align exactly with annotated spans. While suitable for span-based models, this approach falls short for large language models (LLMs), which often generate diverse yet semantically accurate arguments. EM severely underestimates performance by disregarding valid variations. Furthermore, EM evaluation fails to capture implicit arguments (unstated but inferable) and scattered arguments (distributed across a document). These limit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16838","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.16838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.16838","created_at":"2026-07-05T12:08:00.070491+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.16838v2","created_at":"2026-07-05T12:08:00.070491+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16838","created_at":"2026-07-05T12:08:00.070491+00:00"},{"alias_kind":"pith_short_12","alias_value":"EWTS5MUAJZYY","created_at":"2026-07-05T12:08:00.070491+00:00"},{"alias_kind":"pith_short_16","alias_value":"EWTS5MUAJZYYOBPB","created_at":"2026-07-05T12:08:00.070491+00:00"},{"alias_kind":"pith_short_8","alias_value":"EWTS5MUA","created_at":"2026-07-05T12:08:00.070491+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EWTS5MUAJZYYOBPBF25QMC56YL","json":"https://pith.science/pith/EWTS5MUAJZYYOBPBF25QMC56YL.json","graph_json":"https://pith.science/api/pith-number/EWTS5MUAJZYYOBPBF25QMC56YL/graph.json","events_json":"https://pith.science/api/pith-number/EWTS5MUAJZYYOBPBF25QMC56YL/events.json","paper":"https://pith.science/paper/EWTS5MUA"},"agent_actions":{"view_html":"https://pith.science/pith/EWTS5MUAJZYYOBPBF25QMC56YL","download_json":"https://pith.science/pith/EWTS5MUAJZYYOBPBF25QMC56YL.json","view_paper":"https://pith.science/paper/EWTS5MUA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.16838&json=true","fetch_graph":"https://pith.science/api/pith-number/EWTS5MUAJZYYOBPBF25QMC56YL/graph.json","fetch_events":"https://pith.science/api/pith-number/EWTS5MUAJZYYOBPBF25QMC56YL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EWTS5MUAJZYYOBPBF25QMC56YL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EWTS5MUAJZYYOBPBF25QMC56YL/action/storage_attestation","attest_author":"https://pith.science/pith/EWTS5MUAJZYYOBPBF25QMC56YL/action/author_attestation","sign_citation":"https://pith.science/pith/EWTS5MUAJZYYOBPBF25QMC56YL/action/citation_signature","submit_replication":"https://pith.science/pith/EWTS5MUAJZYYOBPBF25QMC56YL/action/replication_record"}},"created_at":"2026-07-05T12:08:00.070491+00:00","updated_at":"2026-07-05T12:08:00.070491+00:00"}