{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:EE5JNNIFNXTCBBKTJHCTZMFCRZ","short_pith_number":"pith:EE5JNNIF","schema_version":"1.0","canonical_sha256":"213a96b5056de620855349c53cb0a28e7371565eaca76d389e337f5226d18517","source":{"kind":"arxiv","id":"2310.07856","version":1},"attestation_state":"computed","paper":{"title":"Assessing Evaluation Metrics for Neural Test Oracle Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CL","authors_text":"Hadi Hemmati, Jiho Shin, Moshi Wei, Song Wang","submitted_at":"2023-10-11T19:58:07Z","abstract_excerpt":"In this work, we revisit existing oracle generation studies plus ChatGPT to empirically investigate the current standing of their performance in both NLG-based and test adequacy metrics. Specifically, we train and run four state-of-the-art test oracle generation models on five NLG-based and two test adequacy metrics for our analysis. We apply two different correlation analyses between these two different sets of metrics. Surprisingly, we found no significant correlation between the NLG-based metrics and test adequacy metrics. For instance, oracles generated from ChatGPT on the project activemq"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.07856","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-11T19:58:07Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"5732a7dcd2f71a0064dd3b46fbe6efbe84829a2b088b7d4bfe195fc3bbca7fed","abstract_canon_sha256":"9212679a44282bca66bd9b3828b81c7e5117be00611b769baaad2ff8318bd880"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:29.116206Z","signature_b64":"KjrdNes4XBvmGIUUQT2DfeSvYCjuFTfrLZBZTCMZkrcEXXfHie1BBSlMwVCId0sDenGgwGu7tgrl/qWdnjxCAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"213a96b5056de620855349c53cb0a28e7371565eaca76d389e337f5226d18517","last_reissued_at":"2026-07-05T09:31:29.115735Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:29.115735Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Assessing Evaluation Metrics for Neural Test Oracle Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CL","authors_text":"Hadi Hemmati, Jiho Shin, Moshi Wei, Song Wang","submitted_at":"2023-10-11T19:58:07Z","abstract_excerpt":"In this work, we revisit existing oracle generation studies plus ChatGPT to empirically investigate the current standing of their performance in both NLG-based and test adequacy metrics. Specifically, we train and run four state-of-the-art test oracle generation models on five NLG-based and two test adequacy metrics for our analysis. We apply two different correlation analyses between these two different sets of metrics. Surprisingly, we found no significant correlation between the NLG-based metrics and test adequacy metrics. For instance, oracles generated from ChatGPT on the project activemq"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.07856","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.07856/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.07856","created_at":"2026-07-05T09:31:29.115793+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.07856v1","created_at":"2026-07-05T09:31:29.115793+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.07856","created_at":"2026-07-05T09:31:29.115793+00:00"},{"alias_kind":"pith_short_12","alias_value":"EE5JNNIFNXTC","created_at":"2026-07-05T09:31:29.115793+00:00"},{"alias_kind":"pith_short_16","alias_value":"EE5JNNIFNXTCBBKT","created_at":"2026-07-05T09:31:29.115793+00:00"},{"alias_kind":"pith_short_8","alias_value":"EE5JNNIF","created_at":"2026-07-05T09:31:29.115793+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EE5JNNIFNXTCBBKTJHCTZMFCRZ","json":"https://pith.science/pith/EE5JNNIFNXTCBBKTJHCTZMFCRZ.json","graph_json":"https://pith.science/api/pith-number/EE5JNNIFNXTCBBKTJHCTZMFCRZ/graph.json","events_json":"https://pith.science/api/pith-number/EE5JNNIFNXTCBBKTJHCTZMFCRZ/events.json","paper":"https://pith.science/paper/EE5JNNIF"},"agent_actions":{"view_html":"https://pith.science/pith/EE5JNNIFNXTCBBKTJHCTZMFCRZ","download_json":"https://pith.science/pith/EE5JNNIFNXTCBBKTJHCTZMFCRZ.json","view_paper":"https://pith.science/paper/EE5JNNIF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.07856&json=true","fetch_graph":"https://pith.science/api/pith-number/EE5JNNIFNXTCBBKTJHCTZMFCRZ/graph.json","fetch_events":"https://pith.science/api/pith-number/EE5JNNIFNXTCBBKTJHCTZMFCRZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EE5JNNIFNXTCBBKTJHCTZMFCRZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EE5JNNIFNXTCBBKTJHCTZMFCRZ/action/storage_attestation","attest_author":"https://pith.science/pith/EE5JNNIFNXTCBBKTJHCTZMFCRZ/action/author_attestation","sign_citation":"https://pith.science/pith/EE5JNNIFNXTCBBKTJHCTZMFCRZ/action/citation_signature","submit_replication":"https://pith.science/pith/EE5JNNIFNXTCBBKTJHCTZMFCRZ/action/replication_record"}},"created_at":"2026-07-05T09:31:29.115793+00:00","updated_at":"2026-07-05T09:31:29.115793+00:00"}