{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4S7UTKYXMLEA2AIQG5R5NWQOG3","short_pith_number":"pith:4S7UTKYX","schema_version":"1.0","canonical_sha256":"e4bf49ab1762c80d01103763d6da0e36dd46e252aa1bd58b67540665e315c7a6","source":{"kind":"arxiv","id":"2506.18105","version":1},"attestation_state":"computed","paper":{"title":"Chengyu-Bench: Benchmarking Large Language Models for Chinese Idiom Understanding and Use","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Liuxin Yang, Yicheng Fu, Yumeng Lu, Zhemin Huang, Zhongdongming Dai","submitted_at":"2025-06-22T17:26:09Z","abstract_excerpt":"Chinese idioms (Chengyu) are concise four-character expressions steeped in history and culture, whose literal translations often fail to capture their full meaning. This complexity makes them challenging for language models to interpret and use correctly. Existing benchmarks focus on narrow tasks - multiple-choice cloze tests, isolated translation, or simple paraphrasing. We introduce Chengyu-Bench, a comprehensive benchmark featuring three tasks: (1) Evaluative Connotation, classifying idioms as positive or negative; (2) Appropriateness, detecting incorrect idiom usage in context; and (3) Ope"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.18105","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-22T17:26:09Z","cross_cats_sorted":[],"title_canon_sha256":"3a1008fe5b4c09646905b7c8f8565d3adf552af22154b876e9bf432e888640ed","abstract_canon_sha256":"17f756503ac9a31a33e42a08188b95f9e8d033ad711d80760e8910cce058c66a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:28.620936Z","signature_b64":"gMaw/xuOy9CEc3R3XwKylROCaov2IYG5b4AO0dQ9HPSXaH+MP0F5wz8+Zz7JRqvE7vp8X71XdQZtyHhKwWvSBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e4bf49ab1762c80d01103763d6da0e36dd46e252aa1bd58b67540665e315c7a6","last_reissued_at":"2026-07-05T11:25:28.620470Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:28.620470Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Chengyu-Bench: Benchmarking Large Language Models for Chinese Idiom Understanding and Use","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Liuxin Yang, Yicheng Fu, Yumeng Lu, Zhemin Huang, Zhongdongming Dai","submitted_at":"2025-06-22T17:26:09Z","abstract_excerpt":"Chinese idioms (Chengyu) are concise four-character expressions steeped in history and culture, whose literal translations often fail to capture their full meaning. This complexity makes them challenging for language models to interpret and use correctly. Existing benchmarks focus on narrow tasks - multiple-choice cloze tests, isolated translation, or simple paraphrasing. We introduce Chengyu-Bench, a comprehensive benchmark featuring three tasks: (1) Evaluative Connotation, classifying idioms as positive or negative; (2) Appropriateness, detecting incorrect idiom usage in context; and (3) Ope"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.18105","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.18105/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.18105","created_at":"2026-07-05T11:25:28.620535+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.18105v1","created_at":"2026-07-05T11:25:28.620535+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.18105","created_at":"2026-07-05T11:25:28.620535+00:00"},{"alias_kind":"pith_short_12","alias_value":"4S7UTKYXMLEA","created_at":"2026-07-05T11:25:28.620535+00:00"},{"alias_kind":"pith_short_16","alias_value":"4S7UTKYXMLEA2AIQ","created_at":"2026-07-05T11:25:28.620535+00:00"},{"alias_kind":"pith_short_8","alias_value":"4S7UTKYX","created_at":"2026-07-05T11:25:28.620535+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4S7UTKYXMLEA2AIQG5R5NWQOG3","json":"https://pith.science/pith/4S7UTKYXMLEA2AIQG5R5NWQOG3.json","graph_json":"https://pith.science/api/pith-number/4S7UTKYXMLEA2AIQG5R5NWQOG3/graph.json","events_json":"https://pith.science/api/pith-number/4S7UTKYXMLEA2AIQG5R5NWQOG3/events.json","paper":"https://pith.science/paper/4S7UTKYX"},"agent_actions":{"view_html":"https://pith.science/pith/4S7UTKYXMLEA2AIQG5R5NWQOG3","download_json":"https://pith.science/pith/4S7UTKYXMLEA2AIQG5R5NWQOG3.json","view_paper":"https://pith.science/paper/4S7UTKYX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.18105&json=true","fetch_graph":"https://pith.science/api/pith-number/4S7UTKYXMLEA2AIQG5R5NWQOG3/graph.json","fetch_events":"https://pith.science/api/pith-number/4S7UTKYXMLEA2AIQG5R5NWQOG3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4S7UTKYXMLEA2AIQG5R5NWQOG3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4S7UTKYXMLEA2AIQG5R5NWQOG3/action/storage_attestation","attest_author":"https://pith.science/pith/4S7UTKYXMLEA2AIQG5R5NWQOG3/action/author_attestation","sign_citation":"https://pith.science/pith/4S7UTKYXMLEA2AIQG5R5NWQOG3/action/citation_signature","submit_replication":"https://pith.science/pith/4S7UTKYXMLEA2AIQG5R5NWQOG3/action/replication_record"}},"created_at":"2026-07-05T11:25:28.620535+00:00","updated_at":"2026-07-05T11:25:28.620535+00:00"}