{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GTGZU6VZEPD7G67OSCCAO7EJIA","short_pith_number":"pith:GTGZU6VZ","schema_version":"1.0","canonical_sha256":"34cd9a7ab923c7f37bee9084077c8940133bd286be6db6e2678350f7480728f1","source":{"kind":"arxiv","id":"2505.18690","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking and Rethinking Knowledge Editing for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aixin Sun, Futing Wang, Guoxiu He, Xin Song","submitted_at":"2025-05-24T13:32:03Z","abstract_excerpt":"Knowledge editing aims to update the embedded knowledge within Large Language Models (LLMs). However, existing approaches, whether through parameter modification or external memory integration, often suffer from inconsistent evaluation objectives and experimental setups. To address this gap, we conduct a comprehensive benchmarking study. In addition to fact-level datasets, we introduce more complex event-based datasets and general-purpose datasets drawn from other tasks. Our evaluation covers both instruction-tuned and reasoning-oriented LLMs, under a realistic autoregressive inference setting"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.18690","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-24T13:32:03Z","cross_cats_sorted":[],"title_canon_sha256":"8fc1a1f8356a8132adf1bd131cb2c8868a1c8b95f73f42f72f272b75bc2731bc","abstract_canon_sha256":"0f17c708a2d3c33fa4b7ea73161072fc56de0516c50b76a8d258a3286a60fe8a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:17.942118Z","signature_b64":"0KiC0+vXgx1TpnDLLaSUyPaU8WIWt31Zwv6FJHU0tHwqnnxEPgtd1qC5Gc+h9VVI/ExLPJ3lpZNu6gqx9Xc1Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"34cd9a7ab923c7f37bee9084077c8940133bd286be6db6e2678350f7480728f1","last_reissued_at":"2026-07-05T11:09:17.941632Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:17.941632Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking and Rethinking Knowledge Editing for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aixin Sun, Futing Wang, Guoxiu He, Xin Song","submitted_at":"2025-05-24T13:32:03Z","abstract_excerpt":"Knowledge editing aims to update the embedded knowledge within Large Language Models (LLMs). However, existing approaches, whether through parameter modification or external memory integration, often suffer from inconsistent evaluation objectives and experimental setups. To address this gap, we conduct a comprehensive benchmarking study. In addition to fact-level datasets, we introduce more complex event-based datasets and general-purpose datasets drawn from other tasks. Our evaluation covers both instruction-tuned and reasoning-oriented LLMs, under a realistic autoregressive inference setting"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18690","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18690/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.18690","created_at":"2026-07-05T11:09:17.941688+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.18690v1","created_at":"2026-07-05T11:09:17.941688+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18690","created_at":"2026-07-05T11:09:17.941688+00:00"},{"alias_kind":"pith_short_12","alias_value":"GTGZU6VZEPD7","created_at":"2026-07-05T11:09:17.941688+00:00"},{"alias_kind":"pith_short_16","alias_value":"GTGZU6VZEPD7G67O","created_at":"2026-07-05T11:09:17.941688+00:00"},{"alias_kind":"pith_short_8","alias_value":"GTGZU6VZ","created_at":"2026-07-05T11:09:17.941688+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GTGZU6VZEPD7G67OSCCAO7EJIA","json":"https://pith.science/pith/GTGZU6VZEPD7G67OSCCAO7EJIA.json","graph_json":"https://pith.science/api/pith-number/GTGZU6VZEPD7G67OSCCAO7EJIA/graph.json","events_json":"https://pith.science/api/pith-number/GTGZU6VZEPD7G67OSCCAO7EJIA/events.json","paper":"https://pith.science/paper/GTGZU6VZ"},"agent_actions":{"view_html":"https://pith.science/pith/GTGZU6VZEPD7G67OSCCAO7EJIA","download_json":"https://pith.science/pith/GTGZU6VZEPD7G67OSCCAO7EJIA.json","view_paper":"https://pith.science/paper/GTGZU6VZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.18690&json=true","fetch_graph":"https://pith.science/api/pith-number/GTGZU6VZEPD7G67OSCCAO7EJIA/graph.json","fetch_events":"https://pith.science/api/pith-number/GTGZU6VZEPD7G67OSCCAO7EJIA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GTGZU6VZEPD7G67OSCCAO7EJIA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GTGZU6VZEPD7G67OSCCAO7EJIA/action/storage_attestation","attest_author":"https://pith.science/pith/GTGZU6VZEPD7G67OSCCAO7EJIA/action/author_attestation","sign_citation":"https://pith.science/pith/GTGZU6VZEPD7G67OSCCAO7EJIA/action/citation_signature","submit_replication":"https://pith.science/pith/GTGZU6VZEPD7G67OSCCAO7EJIA/action/replication_record"}},"created_at":"2026-07-05T11:09:17.941688+00:00","updated_at":"2026-07-05T11:09:17.941688+00:00"}