{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KSWILDZ3Q2EE2T74A3UAYTQGHM","short_pith_number":"pith:KSWILDZ3","schema_version":"1.0","canonical_sha256":"54ac858f3b86884d4ffc06e80c4e063b1c7d741471747c61f16e8fe7eec7e1a2","source":{"kind":"arxiv","id":"2402.05939","version":1},"attestation_state":"computed","paper":{"title":"Uncertainty Awareness of Large Language Models Under Code Distribution Shifts: A Benchmark Study","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Cong Liu, Simin Chen, Wei Yang, Yanghong Guo, Yue Dong, Yufei Li","submitted_at":"2024-01-12T00:00:32Z","abstract_excerpt":"Large Language Models (LLMs) have been widely employed in programming language analysis to enhance human productivity. Yet, their reliability can be compromised by various code distribution shifts, leading to inconsistent outputs. While probabilistic methods are known to mitigate such impact through uncertainty calibration and estimation, their efficacy in the language domain remains underexplored compared to their application in image-based tasks. In this work, we first introduce a large-scale benchmark dataset, incorporating three realistic patterns of code distribution shifts at varying int"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05939","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SE","submitted_at":"2024-01-12T00:00:32Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"bc9e04099efcb815b29e3cb369f722976d7ba3086cdf58c274147bbac1317c26","abstract_canon_sha256":"3ca5c574f8be73342542e98bf0454ab30420a2d5ec463989182b543d226a4f94"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:43:15.589671Z","signature_b64":"YqlmZfQj5eJhiCMMTqcnxItu5LzvL0Etp4EZRE6hcBzT/uSts44PRH028X+WG/FbKzW8nDcoTpaxU9tXmfe5Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54ac858f3b86884d4ffc06e80c4e063b1c7d741471747c61f16e8fe7eec7e1a2","last_reissued_at":"2026-07-05T07:43:15.589138Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:43:15.589138Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Uncertainty Awareness of Large Language Models Under Code Distribution Shifts: A Benchmark Study","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Cong Liu, Simin Chen, Wei Yang, Yanghong Guo, Yue Dong, Yufei Li","submitted_at":"2024-01-12T00:00:32Z","abstract_excerpt":"Large Language Models (LLMs) have been widely employed in programming language analysis to enhance human productivity. Yet, their reliability can be compromised by various code distribution shifts, leading to inconsistent outputs. While probabilistic methods are known to mitigate such impact through uncertainty calibration and estimation, their efficacy in the language domain remains underexplored compared to their application in image-based tasks. In this work, we first introduce a large-scale benchmark dataset, incorporating three realistic patterns of code distribution shifts at varying int"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05939","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05939/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05939","created_at":"2026-07-05T07:43:15.589200+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05939v1","created_at":"2026-07-05T07:43:15.589200+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05939","created_at":"2026-07-05T07:43:15.589200+00:00"},{"alias_kind":"pith_short_12","alias_value":"KSWILDZ3Q2EE","created_at":"2026-07-05T07:43:15.589200+00:00"},{"alias_kind":"pith_short_16","alias_value":"KSWILDZ3Q2EE2T74","created_at":"2026-07-05T07:43:15.589200+00:00"},{"alias_kind":"pith_short_8","alias_value":"KSWILDZ3","created_at":"2026-07-05T07:43:15.589200+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.21276","citing_title":"LeMix: Unified Scheduling for LLM Training and Inference on Multi-GPU Systems","ref_index":67,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KSWILDZ3Q2EE2T74A3UAYTQGHM","json":"https://pith.science/pith/KSWILDZ3Q2EE2T74A3UAYTQGHM.json","graph_json":"https://pith.science/api/pith-number/KSWILDZ3Q2EE2T74A3UAYTQGHM/graph.json","events_json":"https://pith.science/api/pith-number/KSWILDZ3Q2EE2T74A3UAYTQGHM/events.json","paper":"https://pith.science/paper/KSWILDZ3"},"agent_actions":{"view_html":"https://pith.science/pith/KSWILDZ3Q2EE2T74A3UAYTQGHM","download_json":"https://pith.science/pith/KSWILDZ3Q2EE2T74A3UAYTQGHM.json","view_paper":"https://pith.science/paper/KSWILDZ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05939&json=true","fetch_graph":"https://pith.science/api/pith-number/KSWILDZ3Q2EE2T74A3UAYTQGHM/graph.json","fetch_events":"https://pith.science/api/pith-number/KSWILDZ3Q2EE2T74A3UAYTQGHM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KSWILDZ3Q2EE2T74A3UAYTQGHM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KSWILDZ3Q2EE2T74A3UAYTQGHM/action/storage_attestation","attest_author":"https://pith.science/pith/KSWILDZ3Q2EE2T74A3UAYTQGHM/action/author_attestation","sign_citation":"https://pith.science/pith/KSWILDZ3Q2EE2T74A3UAYTQGHM/action/citation_signature","submit_replication":"https://pith.science/pith/KSWILDZ3Q2EE2T74A3UAYTQGHM/action/replication_record"}},"created_at":"2026-07-05T07:43:15.589200+00:00","updated_at":"2026-07-05T07:43:15.589200+00:00"}