{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:K46K5CWL6WDCP7PJ4RAIE65YT6","short_pith_number":"pith:K46K5CWL","schema_version":"1.0","canonical_sha256":"573cae8acbf58627fde9e440827bb89fabb9150b85a8d45c7ec6104a07334e50","source":{"kind":"arxiv","id":"2205.06439","version":1},"attestation_state":"computed","paper":{"title":"AEON: A Method for Automatic Evaluation of NLP Test Cases","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Jen-tse Huang, Jianping Zhang, Michael R. Lyu, Pinjia He, Wenxuan Wang, Yuxin Su","submitted_at":"2022-05-13T03:47:13Z","abstract_excerpt":"Due to the labor-intensive nature of manual test oracle construction, various automated testing techniques have been proposed to enhance the reliability of Natural Language Processing (NLP) software. In theory, these techniques mutate an existing test case (e.g., a sentence with its label) and assume the generated one preserves an equivalent or similar semantic meaning and thus, the same label. However, in practice, many of the generated test cases fail to preserve similar semantic meaning and are unnatural (e.g., grammar errors), which leads to a high false alarm rate and unnatural test cases"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.06439","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2022-05-13T03:47:13Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8ea16a92cbfceab75c543a086b0d8721eaf46f9e52a8b35ff7a9698f56fb6d91","abstract_canon_sha256":"c9d42c26ec807d7a8ceb09014a2e1d1e5f35b8b477868b509c59aafaa139ae88"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:22:57.856500Z","signature_b64":"TVdoSOUrWyGuHWx7u9H5GVX0w6u+JoTFAExeUk9Xf2xaa3/T6ptb+zgFu59OArCiqWG/YiQgo146H92uYMIMAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"573cae8acbf58627fde9e440827bb89fabb9150b85a8d45c7ec6104a07334e50","last_reissued_at":"2026-07-05T04:22:57.856002Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:22:57.856002Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AEON: A Method for Automatic Evaluation of NLP Test Cases","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Jen-tse Huang, Jianping Zhang, Michael R. Lyu, Pinjia He, Wenxuan Wang, Yuxin Su","submitted_at":"2022-05-13T03:47:13Z","abstract_excerpt":"Due to the labor-intensive nature of manual test oracle construction, various automated testing techniques have been proposed to enhance the reliability of Natural Language Processing (NLP) software. In theory, these techniques mutate an existing test case (e.g., a sentence with its label) and assume the generated one preserves an equivalent or similar semantic meaning and thus, the same label. However, in practice, many of the generated test cases fail to preserve similar semantic meaning and are unnatural (e.g., grammar errors), which leads to a high false alarm rate and unnatural test cases"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.06439","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.06439/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.06439","created_at":"2026-07-05T04:22:57.856064+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.06439v1","created_at":"2026-07-05T04:22:57.856064+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.06439","created_at":"2026-07-05T04:22:57.856064+00:00"},{"alias_kind":"pith_short_12","alias_value":"K46K5CWL6WDC","created_at":"2026-07-05T04:22:57.856064+00:00"},{"alias_kind":"pith_short_16","alias_value":"K46K5CWL6WDCP7PJ","created_at":"2026-07-05T04:22:57.856064+00:00"},{"alias_kind":"pith_short_8","alias_value":"K46K5CWL","created_at":"2026-07-05T04:22:57.856064+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K46K5CWL6WDCP7PJ4RAIE65YT6","json":"https://pith.science/pith/K46K5CWL6WDCP7PJ4RAIE65YT6.json","graph_json":"https://pith.science/api/pith-number/K46K5CWL6WDCP7PJ4RAIE65YT6/graph.json","events_json":"https://pith.science/api/pith-number/K46K5CWL6WDCP7PJ4RAIE65YT6/events.json","paper":"https://pith.science/paper/K46K5CWL"},"agent_actions":{"view_html":"https://pith.science/pith/K46K5CWL6WDCP7PJ4RAIE65YT6","download_json":"https://pith.science/pith/K46K5CWL6WDCP7PJ4RAIE65YT6.json","view_paper":"https://pith.science/paper/K46K5CWL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.06439&json=true","fetch_graph":"https://pith.science/api/pith-number/K46K5CWL6WDCP7PJ4RAIE65YT6/graph.json","fetch_events":"https://pith.science/api/pith-number/K46K5CWL6WDCP7PJ4RAIE65YT6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K46K5CWL6WDCP7PJ4RAIE65YT6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K46K5CWL6WDCP7PJ4RAIE65YT6/action/storage_attestation","attest_author":"https://pith.science/pith/K46K5CWL6WDCP7PJ4RAIE65YT6/action/author_attestation","sign_citation":"https://pith.science/pith/K46K5CWL6WDCP7PJ4RAIE65YT6/action/citation_signature","submit_replication":"https://pith.science/pith/K46K5CWL6WDCP7PJ4RAIE65YT6/action/replication_record"}},"created_at":"2026-07-05T04:22:57.856064+00:00","updated_at":"2026-07-05T04:22:57.856064+00:00"}