{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2HCUPMYR6METPZWB6USL4CM4WR","short_pith_number":"pith:2HCUPMYR","schema_version":"1.0","canonical_sha256":"d1c547b311f30937e6c1f524be099cb47802d602efbae4d4e1ed1b4fb6fe9bcd","source":{"kind":"arxiv","id":"2410.03523","version":6},"attestation_state":"computed","paper":{"title":"A Probabilistic Perspective on Unlearning and Alignment for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Leo Schwinn, Stephan G\\\"unnemann, Yan Scholten","submitted_at":"2024-10-04T15:44:23Z","abstract_excerpt":"Comprehensive evaluation of Large Language Models (LLMs) is an open research problem. Existing evaluations rely on deterministic point estimates generated via greedy decoding. However, we find that deterministic evaluations fail to capture the whole output distribution of a model, yielding inaccurate estimations of model capabilities. This is particularly problematic in critical contexts such as unlearning and alignment, where precise model evaluations are crucial. To remedy this, we introduce the first formal probabilistic evaluation framework for LLMs. Namely, we propose novel metrics with h"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03523","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-04T15:44:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3efd2e89874e44c4b4f317653794533d9d5c60f3d68ef0c8df48aeb8262b5728","abstract_canon_sha256":"ffa67770eb4db7ff71b1a61eb772ec722d95afc6989a0f9307db8cf1c14cd207"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:05.904217Z","signature_b64":"IWew/WA9+kk8a1fCuYPB/mapowosu/QtVt/Qp+p1iIjCd7PudEs+rndG3WypsV1wHvT491v+fuOno+zHMFfgDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1c547b311f30937e6c1f524be099cb47802d602efbae4d4e1ed1b4fb6fe9bcd","last_reissued_at":"2026-07-05T10:22:05.903683Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:05.903683Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Probabilistic Perspective on Unlearning and Alignment for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Leo Schwinn, Stephan G\\\"unnemann, Yan Scholten","submitted_at":"2024-10-04T15:44:23Z","abstract_excerpt":"Comprehensive evaluation of Large Language Models (LLMs) is an open research problem. Existing evaluations rely on deterministic point estimates generated via greedy decoding. However, we find that deterministic evaluations fail to capture the whole output distribution of a model, yielding inaccurate estimations of model capabilities. This is particularly problematic in critical contexts such as unlearning and alignment, where precise model evaluations are crucial. To remedy this, we introduce the first formal probabilistic evaluation framework for LLMs. Namely, we propose novel metrics with h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03523","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03523/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03523","created_at":"2026-07-05T10:22:05.903751+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03523v6","created_at":"2026-07-05T10:22:05.903751+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03523","created_at":"2026-07-05T10:22:05.903751+00:00"},{"alias_kind":"pith_short_12","alias_value":"2HCUPMYR6MET","created_at":"2026-07-05T10:22:05.903751+00:00"},{"alias_kind":"pith_short_16","alias_value":"2HCUPMYR6METPZWB","created_at":"2026-07-05T10:22:05.903751+00:00"},{"alias_kind":"pith_short_8","alias_value":"2HCUPMYR","created_at":"2026-07-05T10:22:05.903751+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.02574","citing_title":"LLM-Safety Evaluations Lack Robustness","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2HCUPMYR6METPZWB6USL4CM4WR","json":"https://pith.science/pith/2HCUPMYR6METPZWB6USL4CM4WR.json","graph_json":"https://pith.science/api/pith-number/2HCUPMYR6METPZWB6USL4CM4WR/graph.json","events_json":"https://pith.science/api/pith-number/2HCUPMYR6METPZWB6USL4CM4WR/events.json","paper":"https://pith.science/paper/2HCUPMYR"},"agent_actions":{"view_html":"https://pith.science/pith/2HCUPMYR6METPZWB6USL4CM4WR","download_json":"https://pith.science/pith/2HCUPMYR6METPZWB6USL4CM4WR.json","view_paper":"https://pith.science/paper/2HCUPMYR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03523&json=true","fetch_graph":"https://pith.science/api/pith-number/2HCUPMYR6METPZWB6USL4CM4WR/graph.json","fetch_events":"https://pith.science/api/pith-number/2HCUPMYR6METPZWB6USL4CM4WR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2HCUPMYR6METPZWB6USL4CM4WR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2HCUPMYR6METPZWB6USL4CM4WR/action/storage_attestation","attest_author":"https://pith.science/pith/2HCUPMYR6METPZWB6USL4CM4WR/action/author_attestation","sign_citation":"https://pith.science/pith/2HCUPMYR6METPZWB6USL4CM4WR/action/citation_signature","submit_replication":"https://pith.science/pith/2HCUPMYR6METPZWB6USL4CM4WR/action/replication_record"}},"created_at":"2026-07-05T10:22:05.903751+00:00","updated_at":"2026-07-05T10:22:05.903751+00:00"}