{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KYQNWJGU3KLQHDMOGNNVQJE5DK","short_pith_number":"pith:KYQNWJGU","schema_version":"1.0","canonical_sha256":"5620db24d4da97038d8e335b58249d1a90aab62b8fad9b1e14d737958983a62c","source":{"kind":"arxiv","id":"2311.08472","version":1},"attestation_state":"computed","paper":{"title":"Selecting Shots for Demographic Fairness in Few-Shot Learning with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Carlos Aguirre, Isabel Cachola, Kuleen Sasse, Mark Dredze","submitted_at":"2023-11-14T19:02:03Z","abstract_excerpt":"Recently, work in NLP has shifted to few-shot (in-context) learning, with large language models (LLMs) performing well across a range of tasks. However, while fairness evaluations have become a standard for supervised methods, little is known about the fairness of LLMs as prediction systems. Further, common standard methods for fairness involve access to models weights or are applied during finetuning, which are not applicable in few-shot learning. Do LLMs exhibit prediction biases when used for standard NLP tasks? In this work, we explore the effect of shots, which directly affect the perform"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.08472","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-14T19:02:03Z","cross_cats_sorted":[],"title_canon_sha256":"8d0755cf847a3b01320fdef001baaa6cb45705471b566d081e030cfc52b91292","abstract_canon_sha256":"91062c37860402b50615625635902dc5033ed193b50c198cee797609d705c64f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:57.583052Z","signature_b64":"z358fwWOwLzlnUQGi5a23IpoYfawx1Y3khbf4Zsz5w43XgPGdoR7CNf42VKrTyrVMKWOCEhXgMa7578Ov0iTBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5620db24d4da97038d8e335b58249d1a90aab62b8fad9b1e14d737958983a62c","last_reissued_at":"2026-07-05T07:12:57.582584Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:57.582584Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Selecting Shots for Demographic Fairness in Few-Shot Learning with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Carlos Aguirre, Isabel Cachola, Kuleen Sasse, Mark Dredze","submitted_at":"2023-11-14T19:02:03Z","abstract_excerpt":"Recently, work in NLP has shifted to few-shot (in-context) learning, with large language models (LLMs) performing well across a range of tasks. However, while fairness evaluations have become a standard for supervised methods, little is known about the fairness of LLMs as prediction systems. Further, common standard methods for fairness involve access to models weights or are applied during finetuning, which are not applicable in few-shot learning. Do LLMs exhibit prediction biases when used for standard NLP tasks? In this work, we explore the effect of shots, which directly affect the perform"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.08472","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.08472/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.08472","created_at":"2026-07-05T07:12:57.582651+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.08472v1","created_at":"2026-07-05T07:12:57.582651+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.08472","created_at":"2026-07-05T07:12:57.582651+00:00"},{"alias_kind":"pith_short_12","alias_value":"KYQNWJGU3KLQ","created_at":"2026-07-05T07:12:57.582651+00:00"},{"alias_kind":"pith_short_16","alias_value":"KYQNWJGU3KLQHDMO","created_at":"2026-07-05T07:12:57.582651+00:00"},{"alias_kind":"pith_short_8","alias_value":"KYQNWJGU","created_at":"2026-07-05T07:12:57.582651+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.18754","citing_title":"Few-Shot Optimization for Sensor Data Using Large Language Models: A Case Study on Fatigue Detection","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KYQNWJGU3KLQHDMOGNNVQJE5DK","json":"https://pith.science/pith/KYQNWJGU3KLQHDMOGNNVQJE5DK.json","graph_json":"https://pith.science/api/pith-number/KYQNWJGU3KLQHDMOGNNVQJE5DK/graph.json","events_json":"https://pith.science/api/pith-number/KYQNWJGU3KLQHDMOGNNVQJE5DK/events.json","paper":"https://pith.science/paper/KYQNWJGU"},"agent_actions":{"view_html":"https://pith.science/pith/KYQNWJGU3KLQHDMOGNNVQJE5DK","download_json":"https://pith.science/pith/KYQNWJGU3KLQHDMOGNNVQJE5DK.json","view_paper":"https://pith.science/paper/KYQNWJGU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.08472&json=true","fetch_graph":"https://pith.science/api/pith-number/KYQNWJGU3KLQHDMOGNNVQJE5DK/graph.json","fetch_events":"https://pith.science/api/pith-number/KYQNWJGU3KLQHDMOGNNVQJE5DK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KYQNWJGU3KLQHDMOGNNVQJE5DK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KYQNWJGU3KLQHDMOGNNVQJE5DK/action/storage_attestation","attest_author":"https://pith.science/pith/KYQNWJGU3KLQHDMOGNNVQJE5DK/action/author_attestation","sign_citation":"https://pith.science/pith/KYQNWJGU3KLQHDMOGNNVQJE5DK/action/citation_signature","submit_replication":"https://pith.science/pith/KYQNWJGU3KLQHDMOGNNVQJE5DK/action/replication_record"}},"created_at":"2026-07-05T07:12:57.582651+00:00","updated_at":"2026-07-05T07:12:57.582651+00:00"}