{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AK5JLCFDIPQKZQSDP2SQCKB53J","short_pith_number":"pith:AK5JLCFD","schema_version":"1.0","canonical_sha256":"02ba9588a343e0acc2437ea501283dda72f24346a52e9759b246cd011f0f237f","source":{"kind":"arxiv","id":"2405.18780","version":3},"attestation_state":"computed","paper":{"title":"Certifying Counterfactual Bias in LLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Gagandeep Singh, Isha Chaudhary, Manoj Kumar, Morteza Ziyadi, Qian Hu, Rahul Gupta","submitted_at":"2024-05-29T05:39:37Z","abstract_excerpt":"Large Language Models (LLMs) can produce biased responses that can cause representational harms. However, conventional studies are insufficient to thoroughly evaluate biases across LLM responses for different demographic groups (a.k.a. counterfactual bias), as they do not scale to large number of inputs and do not provide guarantees. Therefore, we propose the first framework, LLMCert-B that certifies LLMs for counterfactual bias on distributions of prompts. A certificate consists of high-confidence bounds on the probability of unbiased LLM responses for any set of counterfactual prompts - prom"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.18780","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-05-29T05:39:37Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"64d5a2e2438b5a6eab503a4c97142e3e51a91886982be196a823065172212dfb","abstract_canon_sha256":"617d12f67f4b3a0311171c363ecd8f2abf6e2ebee111d1be9d603ca31b6b9195"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:12.129275Z","signature_b64":"O7gElROnjfralFTXHBZYUBMwv2Ww2mlo9oM47teSXKhUYToAymkdq4JlrZvdFd/RO98RfRf+U1vk+SKPDCiZAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"02ba9588a343e0acc2437ea501283dda72f24346a52e9759b246cd011f0f237f","last_reissued_at":"2026-07-05T10:52:12.128773Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:12.128773Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Certifying Counterfactual Bias in LLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Gagandeep Singh, Isha Chaudhary, Manoj Kumar, Morteza Ziyadi, Qian Hu, Rahul Gupta","submitted_at":"2024-05-29T05:39:37Z","abstract_excerpt":"Large Language Models (LLMs) can produce biased responses that can cause representational harms. However, conventional studies are insufficient to thoroughly evaluate biases across LLM responses for different demographic groups (a.k.a. counterfactual bias), as they do not scale to large number of inputs and do not provide guarantees. Therefore, we propose the first framework, LLMCert-B that certifies LLMs for counterfactual bias on distributions of prompts. A certificate consists of high-confidence bounds on the probability of unbiased LLM responses for any set of counterfactual prompts - prom"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.18780","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.18780/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.18780","created_at":"2026-07-05T10:52:12.128839+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.18780v3","created_at":"2026-07-05T10:52:12.128839+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.18780","created_at":"2026-07-05T10:52:12.128839+00:00"},{"alias_kind":"pith_short_12","alias_value":"AK5JLCFDIPQK","created_at":"2026-07-05T10:52:12.128839+00:00"},{"alias_kind":"pith_short_16","alias_value":"AK5JLCFDIPQKZQSD","created_at":"2026-07-05T10:52:12.128839+00:00"},{"alias_kind":"pith_short_8","alias_value":"AK5JLCFD","created_at":"2026-07-05T10:52:12.128839+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2408.09049","citing_title":"Inertia in Moral and Value Judgments of Large Language Models","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AK5JLCFDIPQKZQSDP2SQCKB53J","json":"https://pith.science/pith/AK5JLCFDIPQKZQSDP2SQCKB53J.json","graph_json":"https://pith.science/api/pith-number/AK5JLCFDIPQKZQSDP2SQCKB53J/graph.json","events_json":"https://pith.science/api/pith-number/AK5JLCFDIPQKZQSDP2SQCKB53J/events.json","paper":"https://pith.science/paper/AK5JLCFD"},"agent_actions":{"view_html":"https://pith.science/pith/AK5JLCFDIPQKZQSDP2SQCKB53J","download_json":"https://pith.science/pith/AK5JLCFDIPQKZQSDP2SQCKB53J.json","view_paper":"https://pith.science/paper/AK5JLCFD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.18780&json=true","fetch_graph":"https://pith.science/api/pith-number/AK5JLCFDIPQKZQSDP2SQCKB53J/graph.json","fetch_events":"https://pith.science/api/pith-number/AK5JLCFDIPQKZQSDP2SQCKB53J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AK5JLCFDIPQKZQSDP2SQCKB53J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AK5JLCFDIPQKZQSDP2SQCKB53J/action/storage_attestation","attest_author":"https://pith.science/pith/AK5JLCFDIPQKZQSDP2SQCKB53J/action/author_attestation","sign_citation":"https://pith.science/pith/AK5JLCFDIPQKZQSDP2SQCKB53J/action/citation_signature","submit_replication":"https://pith.science/pith/AK5JLCFDIPQKZQSDP2SQCKB53J/action/replication_record"}},"created_at":"2026-07-05T10:52:12.128839+00:00","updated_at":"2026-07-05T10:52:12.128839+00:00"}