{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HAZUTTIOJYL3RHIATGICQDJEMR","short_pith_number":"pith:HAZUTTIO","schema_version":"1.0","canonical_sha256":"383349cd0e4e17b89d009990280d24644b00fc1dbb15d9d0593f3263ec104c5d","source":{"kind":"arxiv","id":"2410.10864","version":1},"attestation_state":"computed","paper":{"title":"Fill In The Gaps: Model Calibration and Generalization with Synthetic Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Michelle V. Mancenido, Rong Pan, Yang Ba","submitted_at":"2024-10-07T23:06:42Z","abstract_excerpt":"As machine learning models continue to swiftly advance, calibrating their performance has become a major concern prior to practical and widespread implementation. Most existing calibration methods often negatively impact model accuracy due to the lack of diversity of validation data, resulting in reduced generalizability. To address this, we propose a calibration method that incorporates synthetic data without compromising accuracy. We derive the expected calibration error (ECE) bound using the Probably Approximately Correct (PAC) learning framework. Large language models (LLMs), known for the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.10864","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-07T23:06:42Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d45f3a0b94053360781666d5946c2b6a5f34cae60b9d554815b6275f230befc3","abstract_canon_sha256":"a49ec062069865bccad60e7d5473c9305c428662dea2fb90346bd6ad3b605529"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:26.153745Z","signature_b64":"0zDd0wPonmzufBq7ghxM4twurINHdsuzOvTm3a9Va9Iqc6ebb4tDC9AjJ5yV9p4cHcu9r5wTNr0T9RFTd2DhBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"383349cd0e4e17b89d009990280d24644b00fc1dbb15d9d0593f3263ec104c5d","last_reissued_at":"2026-07-05T09:20:26.153377Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:26.153377Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fill In The Gaps: Model Calibration and Generalization with Synthetic Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Michelle V. Mancenido, Rong Pan, Yang Ba","submitted_at":"2024-10-07T23:06:42Z","abstract_excerpt":"As machine learning models continue to swiftly advance, calibrating their performance has become a major concern prior to practical and widespread implementation. Most existing calibration methods often negatively impact model accuracy due to the lack of diversity of validation data, resulting in reduced generalizability. To address this, we propose a calibration method that incorporates synthetic data without compromising accuracy. We derive the expected calibration error (ECE) bound using the Probably Approximately Correct (PAC) learning framework. Large language models (LLMs), known for the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.10864","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.10864/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.10864","created_at":"2026-07-05T09:20:26.153441+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.10864v1","created_at":"2026-07-05T09:20:26.153441+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.10864","created_at":"2026-07-05T09:20:26.153441+00:00"},{"alias_kind":"pith_short_12","alias_value":"HAZUTTIOJYL3","created_at":"2026-07-05T09:20:26.153441+00:00"},{"alias_kind":"pith_short_16","alias_value":"HAZUTTIOJYL3RHIA","created_at":"2026-07-05T09:20:26.153441+00:00"},{"alias_kind":"pith_short_8","alias_value":"HAZUTTIO","created_at":"2026-07-05T09:20:26.153441+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.19570","citing_title":"Generative Models for Synthetic Data: Transforming Data Mining in the GenAI Era","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HAZUTTIOJYL3RHIATGICQDJEMR","json":"https://pith.science/pith/HAZUTTIOJYL3RHIATGICQDJEMR.json","graph_json":"https://pith.science/api/pith-number/HAZUTTIOJYL3RHIATGICQDJEMR/graph.json","events_json":"https://pith.science/api/pith-number/HAZUTTIOJYL3RHIATGICQDJEMR/events.json","paper":"https://pith.science/paper/HAZUTTIO"},"agent_actions":{"view_html":"https://pith.science/pith/HAZUTTIOJYL3RHIATGICQDJEMR","download_json":"https://pith.science/pith/HAZUTTIOJYL3RHIATGICQDJEMR.json","view_paper":"https://pith.science/paper/HAZUTTIO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.10864&json=true","fetch_graph":"https://pith.science/api/pith-number/HAZUTTIOJYL3RHIATGICQDJEMR/graph.json","fetch_events":"https://pith.science/api/pith-number/HAZUTTIOJYL3RHIATGICQDJEMR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HAZUTTIOJYL3RHIATGICQDJEMR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HAZUTTIOJYL3RHIATGICQDJEMR/action/storage_attestation","attest_author":"https://pith.science/pith/HAZUTTIOJYL3RHIATGICQDJEMR/action/author_attestation","sign_citation":"https://pith.science/pith/HAZUTTIOJYL3RHIATGICQDJEMR/action/citation_signature","submit_replication":"https://pith.science/pith/HAZUTTIOJYL3RHIATGICQDJEMR/action/replication_record"}},"created_at":"2026-07-05T09:20:26.153441+00:00","updated_at":"2026-07-05T09:20:26.153441+00:00"}