{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:R7QOVW2LFQX5FCUODKTIPZQSUM","short_pith_number":"pith:R7QOVW2L","schema_version":"1.0","canonical_sha256":"8fe0eadb4b2c2fd28a8e1aa687e612a32f392ca3e511fd017c86116cfb0c1440","source":{"kind":"arxiv","id":"2305.14341","version":4},"attestation_state":"computed","paper":{"title":"APPLS: Evaluating Evaluation Metrics for Plain Language Summarization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gondy Leroy, Lucy Lu Wang, Tal August, Trevor Cohen, Yue Guo","submitted_at":"2023-05-23T17:59:19Z","abstract_excerpt":"While there has been significant development of models for Plain Language Summarization (PLS), evaluation remains a challenge. PLS lacks a dedicated assessment metric, and the suitability of text generation evaluation metrics is unclear due to the unique transformations involved (e.g., adding background explanations, removing jargon). To address these questions, our study introduces a granular meta-evaluation testbed, APPLS, designed to evaluate metrics for PLS. We identify four PLS criteria from previous work -- informativeness, simplification, coherence, and faithfulness -- and define a set "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14341","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-23T17:59:19Z","cross_cats_sorted":[],"title_canon_sha256":"dee9df534287cfd1ea48fa5f6329fcdd74e9343ff2d54c8a6cea4c99244826ce","abstract_canon_sha256":"a0f56e94205a102d0d7f4b5938cae1761111df575b4d5d971561b5394e882bdd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:42:56.593937Z","signature_b64":"i6anVDpCaJfq2F8/9eo0GjhWQhgkv3NoLtgYujh6dJXXqsD/7j42cTMPt15/W6bQXhwqrTfBTEQSao4ZsxZbAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8fe0eadb4b2c2fd28a8e1aa687e612a32f392ca3e511fd017c86116cfb0c1440","last_reissued_at":"2026-07-05T10:42:56.593485Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:42:56.593485Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"APPLS: Evaluating Evaluation Metrics for Plain Language Summarization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gondy Leroy, Lucy Lu Wang, Tal August, Trevor Cohen, Yue Guo","submitted_at":"2023-05-23T17:59:19Z","abstract_excerpt":"While there has been significant development of models for Plain Language Summarization (PLS), evaluation remains a challenge. PLS lacks a dedicated assessment metric, and the suitability of text generation evaluation metrics is unclear due to the unique transformations involved (e.g., adding background explanations, removing jargon). To address these questions, our study introduces a granular meta-evaluation testbed, APPLS, designed to evaluate metrics for PLS. We identify four PLS criteria from previous work -- informativeness, simplification, coherence, and faithfulness -- and define a set "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14341","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14341/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14341","created_at":"2026-07-05T10:42:56.593546+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14341v4","created_at":"2026-07-05T10:42:56.593546+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14341","created_at":"2026-07-05T10:42:56.593546+00:00"},{"alias_kind":"pith_short_12","alias_value":"R7QOVW2LFQX5","created_at":"2026-07-05T10:42:56.593546+00:00"},{"alias_kind":"pith_short_16","alias_value":"R7QOVW2LFQX5FCUO","created_at":"2026-07-05T10:42:56.593546+00:00"},{"alias_kind":"pith_short_8","alias_value":"R7QOVW2L","created_at":"2026-07-05T10:42:56.593546+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.07419","citing_title":"MedReadCtrl: Personalizing medical text generation with readability-controlled instruction learning","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R7QOVW2LFQX5FCUODKTIPZQSUM","json":"https://pith.science/pith/R7QOVW2LFQX5FCUODKTIPZQSUM.json","graph_json":"https://pith.science/api/pith-number/R7QOVW2LFQX5FCUODKTIPZQSUM/graph.json","events_json":"https://pith.science/api/pith-number/R7QOVW2LFQX5FCUODKTIPZQSUM/events.json","paper":"https://pith.science/paper/R7QOVW2L"},"agent_actions":{"view_html":"https://pith.science/pith/R7QOVW2LFQX5FCUODKTIPZQSUM","download_json":"https://pith.science/pith/R7QOVW2LFQX5FCUODKTIPZQSUM.json","view_paper":"https://pith.science/paper/R7QOVW2L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14341&json=true","fetch_graph":"https://pith.science/api/pith-number/R7QOVW2LFQX5FCUODKTIPZQSUM/graph.json","fetch_events":"https://pith.science/api/pith-number/R7QOVW2LFQX5FCUODKTIPZQSUM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R7QOVW2LFQX5FCUODKTIPZQSUM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R7QOVW2LFQX5FCUODKTIPZQSUM/action/storage_attestation","attest_author":"https://pith.science/pith/R7QOVW2LFQX5FCUODKTIPZQSUM/action/author_attestation","sign_citation":"https://pith.science/pith/R7QOVW2LFQX5FCUODKTIPZQSUM/action/citation_signature","submit_replication":"https://pith.science/pith/R7QOVW2LFQX5FCUODKTIPZQSUM/action/replication_record"}},"created_at":"2026-07-05T10:42:56.593546+00:00","updated_at":"2026-07-05T10:42:56.593546+00:00"}