{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JQDGEE2VDX6CZD3Y5XLUCJEPSF","short_pith_number":"pith:JQDGEE2V","schema_version":"1.0","canonical_sha256":"4c066213551dfc2c8f78edd741248f9147d20981dd8aa8d69396ff3af995b937","source":{"kind":"arxiv","id":"2502.00561","version":2},"attestation_state":"computed","paper":{"title":"Position: Evaluating Generative AI Systems Is a Social Science Measurement Challenge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Abigail Z. Jacobs, A. Feder Cooper, Alexandra Chouldechova, Alexandra Olteanu, Angelina Wang, Chad Atalla, Dan Vann, Emily Corvi, Emily Sheng, Hannah Washington, Hanna Wallach, Jean Garcia-Gathright, Jennifer Wortman Vaughan, Matthew Vogel, Meera Desai, Nicholas Pangakis, P. Alex Dow, Solon Barocas, Stefanie Reed, Su Lin Blodgett","submitted_at":"2025-02-01T21:09:51Z","abstract_excerpt":"The measurement tasks involved in evaluating generative AI (GenAI) systems lack sufficient scientific rigor, leading to what has been described as \"a tangle of sloppy tests [and] apples-to-oranges comparisons\" (Roose, 2024). In this position paper, we argue that the ML community would benefit from learning from and drawing on the social sciences when developing and using measurement instruments for evaluating GenAI systems. Specifically, our position is that evaluating GenAI systems is a social science measurement challenge. We present a four-level framework, grounded in measurement theory fro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.00561","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CY","submitted_at":"2025-02-01T21:09:51Z","cross_cats_sorted":[],"title_canon_sha256":"782aa48e08e58ad0e7432375ae6fbb712128ef92b8ce93889a7b4978e436c6ca","abstract_canon_sha256":"c4e66f9e34438b578c345efd7b8656e69cef34f59320226bbaa310846915232a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:40.326461Z","signature_b64":"ObM5f4o2NY/5huXmeS2989hCKOkwYIwpFwlpICRo6dEUlZ5C4tphqG/C8qXMVDuf0IrVEIFSeLtyrIEDBVf9AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c066213551dfc2c8f78edd741248f9147d20981dd8aa8d69396ff3af995b937","last_reissued_at":"2026-07-05T11:17:40.325960Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:40.325960Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Position: Evaluating Generative AI Systems Is a Social Science Measurement Challenge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Abigail Z. Jacobs, A. Feder Cooper, Alexandra Chouldechova, Alexandra Olteanu, Angelina Wang, Chad Atalla, Dan Vann, Emily Corvi, Emily Sheng, Hannah Washington, Hanna Wallach, Jean Garcia-Gathright, Jennifer Wortman Vaughan, Matthew Vogel, Meera Desai, Nicholas Pangakis, P. Alex Dow, Solon Barocas, Stefanie Reed, Su Lin Blodgett","submitted_at":"2025-02-01T21:09:51Z","abstract_excerpt":"The measurement tasks involved in evaluating generative AI (GenAI) systems lack sufficient scientific rigor, leading to what has been described as \"a tangle of sloppy tests [and] apples-to-oranges comparisons\" (Roose, 2024). In this position paper, we argue that the ML community would benefit from learning from and drawing on the social sciences when developing and using measurement instruments for evaluating GenAI systems. Specifically, our position is that evaluating GenAI systems is a social science measurement challenge. We present a four-level framework, grounded in measurement theory fro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00561","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.00561/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.00561","created_at":"2026-07-05T11:17:40.326018+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.00561v2","created_at":"2026-07-05T11:17:40.326018+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00561","created_at":"2026-07-05T11:17:40.326018+00:00"},{"alias_kind":"pith_short_12","alias_value":"JQDGEE2VDX6C","created_at":"2026-07-05T11:17:40.326018+00:00"},{"alias_kind":"pith_short_16","alias_value":"JQDGEE2VDX6CZD3Y","created_at":"2026-07-05T11:17:40.326018+00:00"},{"alias_kind":"pith_short_8","alias_value":"JQDGEE2V","created_at":"2026-07-05T11:17:40.326018+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17799","citing_title":"Position: Coding Benchmarks Are Misaligned with Agentic Software Engineering","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11018","citing_title":"Measuring Human Value Expression in Social Media Texts: Calibrated LLM Annotation and Encoder Transfer","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02293","citing_title":"AI as a Tool for Simulation-Based Experiments in Literary Studies","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14528","citing_title":"Why Johnny Can't Use Agents: Industry Aspirations vs. User Realities with AI Agents","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2602.00065","citing_title":"Responsible Evaluation of AI for Mental Health","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03238","citing_title":"RLHF May Not Reflect Genuine Preferences","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06811","citing_title":"Making AI Evaluation Deployment Relevant Through Context Specification","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07591","citing_title":"From Ground Truth to Measurement: A Statistical Framework for Human Labeling","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07986","citing_title":"Towards Apples to Apples for AI Evaluations: From Real-World Use Cases to Evaluation Scenarios","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14266","citing_title":"\"I Just Don't Want My Work Being Fed Into The AI Blender\": Queer Artists on Refusing and Resisting Generative AI","ref_index":138,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JQDGEE2VDX6CZD3Y5XLUCJEPSF","json":"https://pith.science/pith/JQDGEE2VDX6CZD3Y5XLUCJEPSF.json","graph_json":"https://pith.science/api/pith-number/JQDGEE2VDX6CZD3Y5XLUCJEPSF/graph.json","events_json":"https://pith.science/api/pith-number/JQDGEE2VDX6CZD3Y5XLUCJEPSF/events.json","paper":"https://pith.science/paper/JQDGEE2V"},"agent_actions":{"view_html":"https://pith.science/pith/JQDGEE2VDX6CZD3Y5XLUCJEPSF","download_json":"https://pith.science/pith/JQDGEE2VDX6CZD3Y5XLUCJEPSF.json","view_paper":"https://pith.science/paper/JQDGEE2V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.00561&json=true","fetch_graph":"https://pith.science/api/pith-number/JQDGEE2VDX6CZD3Y5XLUCJEPSF/graph.json","fetch_events":"https://pith.science/api/pith-number/JQDGEE2VDX6CZD3Y5XLUCJEPSF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JQDGEE2VDX6CZD3Y5XLUCJEPSF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JQDGEE2VDX6CZD3Y5XLUCJEPSF/action/storage_attestation","attest_author":"https://pith.science/pith/JQDGEE2VDX6CZD3Y5XLUCJEPSF/action/author_attestation","sign_citation":"https://pith.science/pith/JQDGEE2VDX6CZD3Y5XLUCJEPSF/action/citation_signature","submit_replication":"https://pith.science/pith/JQDGEE2VDX6CZD3Y5XLUCJEPSF/action/replication_record"}},"created_at":"2026-07-05T11:17:40.326018+00:00","updated_at":"2026-07-05T11:17:40.326018+00:00"}