{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ATPEOXGBXWXRQYB3JELL2LCR2C","short_pith_number":"pith:ATPEOXGB","schema_version":"1.0","canonical_sha256":"04de475cc1bdaf18603b4916bd2c51d086049b3efe1765789d8a988cfd213ef1","source":{"kind":"arxiv","id":"2504.16778","version":2},"attestation_state":"computed","paper":{"title":"Evaluation Framework for AI Systems in \"the Wild\"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Anindya Das Antar, Bryan Goodman, David Jurgens, Elizabeth Bondi-Kelly, Insu Jang, Jae-Won Chung, Jiachen Liu, Joseph Peper, Kevin Samy, Lu Wang, Michael Wellman, Mosharaf Chowdhury, Rada Mihalcea, Sarah Jabbour, Shiqi He, Trenton Chang","submitted_at":"2025-04-23T14:52:39Z","abstract_excerpt":"Generative AI (GenAI) models have become vital across industries, yet current evaluation methods have not adapted to their widespread use. Traditional evaluations often rely on benchmarks and fixed datasets, frequently failing to reflect real-world performance, which creates a gap between lab-tested outcomes and practical applications. This white paper proposes a comprehensive framework for how we should evaluate real-world GenAI systems, emphasizing diverse, evolving inputs and holistic, dynamic, and ongoing assessment approaches. The paper offers guidance for practitioners on how to design e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.16778","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-23T14:52:39Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"a95926e7adbfca33b71381a2be14fcc3856a2d3c3cf062b586e0c98a83ae36ca","abstract_canon_sha256":"297b0ea990c13b48cc37e86b393c4e21d25c97f660474759ab0507105173b1bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:47.823499Z","signature_b64":"n9aJqg1vQhjq2hxpWy/4Hs9tzlfcWgB6c7x0tijWPkz9EEJP5xvQQvDo14cjZEL3etiVBAaXTCY8B+hLcHszCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04de475cc1bdaf18603b4916bd2c51d086049b3efe1765789d8a988cfd213ef1","last_reissued_at":"2026-07-05T10:54:47.823018Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:47.823018Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation Framework for AI Systems in \"the Wild\"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Anindya Das Antar, Bryan Goodman, David Jurgens, Elizabeth Bondi-Kelly, Insu Jang, Jae-Won Chung, Jiachen Liu, Joseph Peper, Kevin Samy, Lu Wang, Michael Wellman, Mosharaf Chowdhury, Rada Mihalcea, Sarah Jabbour, Shiqi He, Trenton Chang","submitted_at":"2025-04-23T14:52:39Z","abstract_excerpt":"Generative AI (GenAI) models have become vital across industries, yet current evaluation methods have not adapted to their widespread use. Traditional evaluations often rely on benchmarks and fixed datasets, frequently failing to reflect real-world performance, which creates a gap between lab-tested outcomes and practical applications. This white paper proposes a comprehensive framework for how we should evaluate real-world GenAI systems, emphasizing diverse, evolving inputs and holistic, dynamic, and ongoing assessment approaches. The paper offers guidance for practitioners on how to design e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.16778","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.16778/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.16778","created_at":"2026-07-05T10:54:47.823072+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.16778v2","created_at":"2026-07-05T10:54:47.823072+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.16778","created_at":"2026-07-05T10:54:47.823072+00:00"},{"alias_kind":"pith_short_12","alias_value":"ATPEOXGBXWXR","created_at":"2026-07-05T10:54:47.823072+00:00"},{"alias_kind":"pith_short_16","alias_value":"ATPEOXGBXWXRQYB3","created_at":"2026-07-05T10:54:47.823072+00:00"},{"alias_kind":"pith_short_8","alias_value":"ATPEOXGB","created_at":"2026-07-05T10:54:47.823072+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17843","citing_title":"Learning from AVA: Early Lessons from a Curated and Trustworthy Generative AI for Policy and Development Research","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ATPEOXGBXWXRQYB3JELL2LCR2C","json":"https://pith.science/pith/ATPEOXGBXWXRQYB3JELL2LCR2C.json","graph_json":"https://pith.science/api/pith-number/ATPEOXGBXWXRQYB3JELL2LCR2C/graph.json","events_json":"https://pith.science/api/pith-number/ATPEOXGBXWXRQYB3JELL2LCR2C/events.json","paper":"https://pith.science/paper/ATPEOXGB"},"agent_actions":{"view_html":"https://pith.science/pith/ATPEOXGBXWXRQYB3JELL2LCR2C","download_json":"https://pith.science/pith/ATPEOXGBXWXRQYB3JELL2LCR2C.json","view_paper":"https://pith.science/paper/ATPEOXGB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.16778&json=true","fetch_graph":"https://pith.science/api/pith-number/ATPEOXGBXWXRQYB3JELL2LCR2C/graph.json","fetch_events":"https://pith.science/api/pith-number/ATPEOXGBXWXRQYB3JELL2LCR2C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ATPEOXGBXWXRQYB3JELL2LCR2C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ATPEOXGBXWXRQYB3JELL2LCR2C/action/storage_attestation","attest_author":"https://pith.science/pith/ATPEOXGBXWXRQYB3JELL2LCR2C/action/author_attestation","sign_citation":"https://pith.science/pith/ATPEOXGBXWXRQYB3JELL2LCR2C/action/citation_signature","submit_replication":"https://pith.science/pith/ATPEOXGBXWXRQYB3JELL2LCR2C/action/replication_record"}},"created_at":"2026-07-05T10:54:47.823072+00:00","updated_at":"2026-07-05T10:54:47.823072+00:00"}