{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:YA3MVVKXHTTYLEA4GAEQBRX6D5","short_pith_number":"pith:YA3MVVKX","schema_version":"1.0","canonical_sha256":"c036cad5573ce785901c300900c6fe1f4bcbf8da3f93a11143a58f6b010f6b15","source":{"kind":"arxiv","id":"2607.14673","version":1},"attestation_state":"computed","paper":{"title":"Project Kaleidoscope: Contextual, Human-Aligned Evaluation for Real-World AI Applications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.AI","authors_text":"Leanne Tan, Rohan Jaggi, Roy Ka-Wei Lee, Shaun Khoo","submitted_at":"2026-07-16T07:38:39Z","abstract_excerpt":"Evaluations (Evals) are a deployment bottleneck for real-world AI applications: public benchmarks rarely match a team's users, context, or policies, and human review is often tedious to scale. Motivated by our work with AI applications in the public sector, this project addresses recurring evaluation challenges encountered when applications must satisfy local policy and governance requirements. We present Kaleidoscope, an integrated workflow for contextual functional evaluation that links persona-based test generation, contextualized rubrics, and human review for reliability-gated automated sc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.14673","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-16T07:38:39Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"1d9a3e7620d6dc5bd4452424f8d8c77a253cf9c2658b22fe9c2ec4582b2be7d4","abstract_canon_sha256":"53706836d3e536e788261e9a7246c59bcc8e5f0ee7ef0e44e3045a559f09c699"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-17T01:21:24.315939Z","signature_b64":"uVkFv8B98nHE8/0rRkgSDNyUxjlP5xR2w1ritW9V9S7RSMLB83KkRlcGxjalA0+H5Osc5jG6GqSb3QT+zHrUBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c036cad5573ce785901c300900c6fe1f4bcbf8da3f93a11143a58f6b010f6b15","last_reissued_at":"2026-07-17T01:21:24.315119Z","signature_status":"signed_v1","first_computed_at":"2026-07-17T01:21:24.315119Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Project Kaleidoscope: Contextual, Human-Aligned Evaluation for Real-World AI Applications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.AI","authors_text":"Leanne Tan, Rohan Jaggi, Roy Ka-Wei Lee, Shaun Khoo","submitted_at":"2026-07-16T07:38:39Z","abstract_excerpt":"Evaluations (Evals) are a deployment bottleneck for real-world AI applications: public benchmarks rarely match a team's users, context, or policies, and human review is often tedious to scale. Motivated by our work with AI applications in the public sector, this project addresses recurring evaluation challenges encountered when applications must satisfy local policy and governance requirements. We present Kaleidoscope, an integrated workflow for contextual functional evaluation that links persona-based test generation, contextualized rubrics, and human review for reliability-gated automated sc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.14673","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.14673/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.14673","created_at":"2026-07-17T01:21:24.315541+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.14673v1","created_at":"2026-07-17T01:21:24.315541+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.14673","created_at":"2026-07-17T01:21:24.315541+00:00"},{"alias_kind":"pith_short_12","alias_value":"YA3MVVKXHTTY","created_at":"2026-07-17T01:21:24.315541+00:00"},{"alias_kind":"pith_short_16","alias_value":"YA3MVVKXHTTYLEA4","created_at":"2026-07-17T01:21:24.315541+00:00"},{"alias_kind":"pith_short_8","alias_value":"YA3MVVKX","created_at":"2026-07-17T01:21:24.315541+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YA3MVVKXHTTYLEA4GAEQBRX6D5","json":"https://pith.science/pith/YA3MVVKXHTTYLEA4GAEQBRX6D5.json","graph_json":"https://pith.science/api/pith-number/YA3MVVKXHTTYLEA4GAEQBRX6D5/graph.json","events_json":"https://pith.science/api/pith-number/YA3MVVKXHTTYLEA4GAEQBRX6D5/events.json","paper":"https://pith.science/paper/YA3MVVKX"},"agent_actions":{"view_html":"https://pith.science/pith/YA3MVVKXHTTYLEA4GAEQBRX6D5","download_json":"https://pith.science/pith/YA3MVVKXHTTYLEA4GAEQBRX6D5.json","view_paper":"https://pith.science/paper/YA3MVVKX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.14673&json=true","fetch_graph":"https://pith.science/api/pith-number/YA3MVVKXHTTYLEA4GAEQBRX6D5/graph.json","fetch_events":"https://pith.science/api/pith-number/YA3MVVKXHTTYLEA4GAEQBRX6D5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YA3MVVKXHTTYLEA4GAEQBRX6D5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YA3MVVKXHTTYLEA4GAEQBRX6D5/action/storage_attestation","attest_author":"https://pith.science/pith/YA3MVVKXHTTYLEA4GAEQBRX6D5/action/author_attestation","sign_citation":"https://pith.science/pith/YA3MVVKXHTTYLEA4GAEQBRX6D5/action/citation_signature","submit_replication":"https://pith.science/pith/YA3MVVKXHTTYLEA4GAEQBRX6D5/action/replication_record"}},"created_at":"2026-07-17T01:21:24.315541+00:00","updated_at":"2026-07-17T01:21:24.315541+00:00"}