{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2VIMFTNKFQXLMFW67IET2XTQ72","short_pith_number":"pith:2VIMFTNK","schema_version":"1.0","canonical_sha256":"d550c2cdaa2c2eb616defa093d5e70fe8ecf1052b0d0fcad15837edb4673129c","source":{"kind":"arxiv","id":"2402.15481","version":5},"attestation_state":"computed","paper":{"title":"Bias and Volatility: A Statistical Framework for Evaluating Large Language Model's Stereotypes and the Associated Generation Inconsistency","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.CL","authors_text":"ChengXiang Zhai, Ke Yang, Xiao Liu, Yang Yu, Yiran Liu, Zehan Qi","submitted_at":"2024-02-23T18:15:56Z","abstract_excerpt":"We present a novel statistical framework for analyzing stereotypes in large language models (LLMs) by systematically estimating the bias and variation in their generation. Current alignment evaluation metrics often overlook stereotypes' randomness caused by LLMs' inconsistent generative behavior. For instance, LLMs may display contradictory stereotypes, such as those related to gender or race, for identical professions in different contexts. Ignoring this inconsistency risks misleading conclusions in alignment assessments and undermines efforts to evaluate the potential of LLMs to perpetuate o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.15481","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-23T18:15:56Z","cross_cats_sorted":["cs.CY"],"title_canon_sha256":"cc4a4f89d6eb86b5e4bfd3a4131b696824b7823cfdce8d4535ae6eb6edcd51df","abstract_canon_sha256":"f5519018ec93a4b144c7dc0cbf50ea8b78dc02097b0715e90a6100c2c76f4026"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:41.936389Z","signature_b64":"7kwldHfOF5liRlsHuXv1Wo325iCu/fTdjyPcJYZtA8zKXY7Y50TejXtE2lRu6tOfcCA6Dsq0jG4APBYv4zIZDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d550c2cdaa2c2eb616defa093d5e70fe8ecf1052b0d0fcad15837edb4673129c","last_reissued_at":"2026-07-05T11:09:41.935894Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:41.935894Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bias and Volatility: A Statistical Framework for Evaluating Large Language Model's Stereotypes and the Associated Generation Inconsistency","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.CL","authors_text":"ChengXiang Zhai, Ke Yang, Xiao Liu, Yang Yu, Yiran Liu, Zehan Qi","submitted_at":"2024-02-23T18:15:56Z","abstract_excerpt":"We present a novel statistical framework for analyzing stereotypes in large language models (LLMs) by systematically estimating the bias and variation in their generation. Current alignment evaluation metrics often overlook stereotypes' randomness caused by LLMs' inconsistent generative behavior. For instance, LLMs may display contradictory stereotypes, such as those related to gender or race, for identical professions in different contexts. Ignoring this inconsistency risks misleading conclusions in alignment assessments and undermines efforts to evaluate the potential of LLMs to perpetuate o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.15481","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.15481/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.15481","created_at":"2026-07-05T11:09:41.935951+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.15481v5","created_at":"2026-07-05T11:09:41.935951+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.15481","created_at":"2026-07-05T11:09:41.935951+00:00"},{"alias_kind":"pith_short_12","alias_value":"2VIMFTNKFQXL","created_at":"2026-07-05T11:09:41.935951+00:00"},{"alias_kind":"pith_short_16","alias_value":"2VIMFTNKFQXLMFW6","created_at":"2026-07-05T11:09:41.935951+00:00"},{"alias_kind":"pith_short_8","alias_value":"2VIMFTNK","created_at":"2026-07-05T11:09:41.935951+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2VIMFTNKFQXLMFW67IET2XTQ72","json":"https://pith.science/pith/2VIMFTNKFQXLMFW67IET2XTQ72.json","graph_json":"https://pith.science/api/pith-number/2VIMFTNKFQXLMFW67IET2XTQ72/graph.json","events_json":"https://pith.science/api/pith-number/2VIMFTNKFQXLMFW67IET2XTQ72/events.json","paper":"https://pith.science/paper/2VIMFTNK"},"agent_actions":{"view_html":"https://pith.science/pith/2VIMFTNKFQXLMFW67IET2XTQ72","download_json":"https://pith.science/pith/2VIMFTNKFQXLMFW67IET2XTQ72.json","view_paper":"https://pith.science/paper/2VIMFTNK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.15481&json=true","fetch_graph":"https://pith.science/api/pith-number/2VIMFTNKFQXLMFW67IET2XTQ72/graph.json","fetch_events":"https://pith.science/api/pith-number/2VIMFTNKFQXLMFW67IET2XTQ72/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2VIMFTNKFQXLMFW67IET2XTQ72/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2VIMFTNKFQXLMFW67IET2XTQ72/action/storage_attestation","attest_author":"https://pith.science/pith/2VIMFTNKFQXLMFW67IET2XTQ72/action/author_attestation","sign_citation":"https://pith.science/pith/2VIMFTNKFQXLMFW67IET2XTQ72/action/citation_signature","submit_replication":"https://pith.science/pith/2VIMFTNKFQXLMFW67IET2XTQ72/action/replication_record"}},"created_at":"2026-07-05T11:09:41.935951+00:00","updated_at":"2026-07-05T11:09:41.935951+00:00"}