{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:E2PDNOTZT27AQ3F35OPZLBDHKA","short_pith_number":"pith:E2PDNOTZ","schema_version":"1.0","canonical_sha256":"269e36ba799ebe086cbbeb9f958467500632d4fbde35685d9cddef02abe90abd","source":{"kind":"arxiv","id":"2206.12664","version":2},"attestation_state":"computed","paper":{"title":"Evaluation of Semantic Answer Similarity Metrics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Farida Mustafazade, Peter F. Ebbinghaus","submitted_at":"2022-06-25T14:40:36Z","abstract_excerpt":"There are several issues with the existing general machine translation or natural language generation evaluation metrics, and question-answering (QA) systems are indifferent in that context. To build robust QA systems, we need the ability to have equivalently robust evaluation systems to verify whether model predictions to questions are similar to ground-truth annotations. The ability to compare similarity based on semantics as opposed to pure string overlap is important to compare models fairly and to indicate more realistic acceptance criteria in real-life applications. We build upon the fir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.12664","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-06-25T14:40:36Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3e3e1076ba0bbc7fa39d474a1fc4953779c7313159eaefa41f3f9277bfe58ebd","abstract_canon_sha256":"e959607d69c688a13f26c865756d1a7a1423c251a8562a15cfb9c0fb4f18a576"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:37:41.052211Z","signature_b64":"d1HNb6eNW71WVrS8B6s+9KLSxR4Hwgu92eYJV9fBtbVqd5413IyyX21FVa4QEcFRgKOvt+8nlYPcHfwT8iRpBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"269e36ba799ebe086cbbeb9f958467500632d4fbde35685d9cddef02abe90abd","last_reissued_at":"2026-07-05T04:37:41.051798Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:37:41.051798Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation of Semantic Answer Similarity Metrics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Farida Mustafazade, Peter F. Ebbinghaus","submitted_at":"2022-06-25T14:40:36Z","abstract_excerpt":"There are several issues with the existing general machine translation or natural language generation evaluation metrics, and question-answering (QA) systems are indifferent in that context. To build robust QA systems, we need the ability to have equivalently robust evaluation systems to verify whether model predictions to questions are similar to ground-truth annotations. The ability to compare similarity based on semantics as opposed to pure string overlap is important to compare models fairly and to indicate more realistic acceptance criteria in real-life applications. We build upon the fir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.12664","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.12664/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.12664","created_at":"2026-07-05T04:37:41.051854+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.12664v2","created_at":"2026-07-05T04:37:41.051854+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.12664","created_at":"2026-07-05T04:37:41.051854+00:00"},{"alias_kind":"pith_short_12","alias_value":"E2PDNOTZT27A","created_at":"2026-07-05T04:37:41.051854+00:00"},{"alias_kind":"pith_short_16","alias_value":"E2PDNOTZT27AQ3F3","created_at":"2026-07-05T04:37:41.051854+00:00"},{"alias_kind":"pith_short_8","alias_value":"E2PDNOTZ","created_at":"2026-07-05T04:37:41.051854+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.13171","citing_title":"Querying Large Automotive Software Models: Agentic vs. Direct LLM Approaches","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E2PDNOTZT27AQ3F35OPZLBDHKA","json":"https://pith.science/pith/E2PDNOTZT27AQ3F35OPZLBDHKA.json","graph_json":"https://pith.science/api/pith-number/E2PDNOTZT27AQ3F35OPZLBDHKA/graph.json","events_json":"https://pith.science/api/pith-number/E2PDNOTZT27AQ3F35OPZLBDHKA/events.json","paper":"https://pith.science/paper/E2PDNOTZ"},"agent_actions":{"view_html":"https://pith.science/pith/E2PDNOTZT27AQ3F35OPZLBDHKA","download_json":"https://pith.science/pith/E2PDNOTZT27AQ3F35OPZLBDHKA.json","view_paper":"https://pith.science/paper/E2PDNOTZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.12664&json=true","fetch_graph":"https://pith.science/api/pith-number/E2PDNOTZT27AQ3F35OPZLBDHKA/graph.json","fetch_events":"https://pith.science/api/pith-number/E2PDNOTZT27AQ3F35OPZLBDHKA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E2PDNOTZT27AQ3F35OPZLBDHKA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E2PDNOTZT27AQ3F35OPZLBDHKA/action/storage_attestation","attest_author":"https://pith.science/pith/E2PDNOTZT27AQ3F35OPZLBDHKA/action/author_attestation","sign_citation":"https://pith.science/pith/E2PDNOTZT27AQ3F35OPZLBDHKA/action/citation_signature","submit_replication":"https://pith.science/pith/E2PDNOTZT27AQ3F35OPZLBDHKA/action/replication_record"}},"created_at":"2026-07-05T04:37:41.051854+00:00","updated_at":"2026-07-05T04:37:41.051854+00:00"}