{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AO27AMZCIXMPHSHI7F2UFBTEEQ","short_pith_number":"pith:AO27AMZC","schema_version":"1.0","canonical_sha256":"03b5f0332245d8f3c8e8f975428664240e59c49cf69533a813a57a19c15d885e","source":{"kind":"arxiv","id":"2402.14889","version":4},"attestation_state":"computed","paper":{"title":"COBIAS: Assessing the Contextual Reliability of Bias Benchmarks for Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aman Chadha, Hemang Jain, Manas Gaur, Ponnurangam Kumaraguru, Priyanshul Govil, Sanorita Dey, Vamshi Krishna Bonagiri","submitted_at":"2024-02-22T10:46:11Z","abstract_excerpt":"Large Language Models (LLMs) often inherit biases from the web data they are trained on, which contains stereotypes and prejudices. Current methods for evaluating and mitigating these biases rely on bias-benchmark datasets. These benchmarks measure bias by observing an LLM's behavior on biased statements. However, these statements lack contextual considerations of the situations they try to present. To address this, we introduce a contextual reliability framework, which evaluates model robustness to biased statements by considering the various contexts in which they may appear. We develop the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14889","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-22T10:46:11Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4f7dc1987db388d422557b9dccc25fbfb594c3140a353ed531f69f12b8aa56ed","abstract_canon_sha256":"5f973aca00d9095425954f1ff98f30f1a0ed3ee3ba7371686add33b0fc361be9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:49.052173Z","signature_b64":"U5V73Df8mz9RmaBwYUrwcaFNKqkgBR60GUHv5I5RoI1G9h4HSl+6nyi/Kk7siOq/r+pYtLT81ehKlXJ05YHgBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03b5f0332245d8f3c8e8f975428664240e59c49cf69533a813a57a19c15d885e","last_reissued_at":"2026-07-05T11:03:49.051721Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:49.051721Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"COBIAS: Assessing the Contextual Reliability of Bias Benchmarks for Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aman Chadha, Hemang Jain, Manas Gaur, Ponnurangam Kumaraguru, Priyanshul Govil, Sanorita Dey, Vamshi Krishna Bonagiri","submitted_at":"2024-02-22T10:46:11Z","abstract_excerpt":"Large Language Models (LLMs) often inherit biases from the web data they are trained on, which contains stereotypes and prejudices. Current methods for evaluating and mitigating these biases rely on bias-benchmark datasets. These benchmarks measure bias by observing an LLM's behavior on biased statements. However, these statements lack contextual considerations of the situations they try to present. To address this, we introduce a contextual reliability framework, which evaluates model robustness to biased statements by considering the various contexts in which they may appear. We develop the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14889","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14889/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14889","created_at":"2026-07-05T11:03:49.051778+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14889v4","created_at":"2026-07-05T11:03:49.051778+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14889","created_at":"2026-07-05T11:03:49.051778+00:00"},{"alias_kind":"pith_short_12","alias_value":"AO27AMZCIXMP","created_at":"2026-07-05T11:03:49.051778+00:00"},{"alias_kind":"pith_short_16","alias_value":"AO27AMZCIXMPHSHI","created_at":"2026-07-05T11:03:49.051778+00:00"},{"alias_kind":"pith_short_8","alias_value":"AO27AMZC","created_at":"2026-07-05T11:03:49.051778+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.08979","citing_title":"PRISM: Reducing Spurious Implicit Biases in Vision-Language Models with LLM-Guided Embedding Projection","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AO27AMZCIXMPHSHI7F2UFBTEEQ","json":"https://pith.science/pith/AO27AMZCIXMPHSHI7F2UFBTEEQ.json","graph_json":"https://pith.science/api/pith-number/AO27AMZCIXMPHSHI7F2UFBTEEQ/graph.json","events_json":"https://pith.science/api/pith-number/AO27AMZCIXMPHSHI7F2UFBTEEQ/events.json","paper":"https://pith.science/paper/AO27AMZC"},"agent_actions":{"view_html":"https://pith.science/pith/AO27AMZCIXMPHSHI7F2UFBTEEQ","download_json":"https://pith.science/pith/AO27AMZCIXMPHSHI7F2UFBTEEQ.json","view_paper":"https://pith.science/paper/AO27AMZC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14889&json=true","fetch_graph":"https://pith.science/api/pith-number/AO27AMZCIXMPHSHI7F2UFBTEEQ/graph.json","fetch_events":"https://pith.science/api/pith-number/AO27AMZCIXMPHSHI7F2UFBTEEQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AO27AMZCIXMPHSHI7F2UFBTEEQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AO27AMZCIXMPHSHI7F2UFBTEEQ/action/storage_attestation","attest_author":"https://pith.science/pith/AO27AMZCIXMPHSHI7F2UFBTEEQ/action/author_attestation","sign_citation":"https://pith.science/pith/AO27AMZCIXMPHSHI7F2UFBTEEQ/action/citation_signature","submit_replication":"https://pith.science/pith/AO27AMZCIXMPHSHI7F2UFBTEEQ/action/replication_record"}},"created_at":"2026-07-05T11:03:49.051778+00:00","updated_at":"2026-07-05T11:03:49.051778+00:00"}