{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OGFAI2PGMTQ3OYA6LVLQ426CKN","short_pith_number":"pith:OGFAI2PG","schema_version":"1.0","canonical_sha256":"718a0469e664e1b7601e5d570e6bc253473b6b8714aa3e82e385f94ed3a01eac","source":{"kind":"arxiv","id":"2507.02306","version":1},"attestation_state":"computed","paper":{"title":"Synthetic Heuristic Evaluation: A Comparison between AI- and Human-Powered Usability Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.HC","authors_text":"David W. McDonald, Gary Hsieh, Ruican Zhong","submitted_at":"2025-07-03T04:27:16Z","abstract_excerpt":"Usability evaluation is crucial in human-centered design but can be costly, requiring expert time and user compensation. In this work, we developed a method for synthetic heuristic evaluation using multimodal LLMs' ability to analyze images and provide design feedback. Comparing our synthetic evaluations to those by experienced UX practitioners across two apps, we found our evaluation identified 73% and 77% of usability issues, which exceeded the performance of 5 experienced human evaluators (57% and 63%). Compared to human evaluators, the synthetic evaluation's performance maintained consiste"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.02306","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.HC","submitted_at":"2025-07-03T04:27:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"759c744c83ac2b871ce3cf00d751db4b3e301af0a135313eb399bf216931fd5f","abstract_canon_sha256":"0928b48107d8289266b77bd705ebfd4624dc6f2baadb8e50b884a55056e0a3ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:15.674931Z","signature_b64":"k+uNXMXZqc8kNTz9TQFjiGkCKQ0tDkotR+TkD7f8B75fC01IhKAYsSphFT3evHyvIuky0nmLEaNRSo4EllKOBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"718a0469e664e1b7601e5d570e6bc253473b6b8714aa3e82e385f94ed3a01eac","last_reissued_at":"2026-07-05T11:31:15.674447Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:15.674447Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Synthetic Heuristic Evaluation: A Comparison between AI- and Human-Powered Usability Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.HC","authors_text":"David W. McDonald, Gary Hsieh, Ruican Zhong","submitted_at":"2025-07-03T04:27:16Z","abstract_excerpt":"Usability evaluation is crucial in human-centered design but can be costly, requiring expert time and user compensation. In this work, we developed a method for synthetic heuristic evaluation using multimodal LLMs' ability to analyze images and provide design feedback. Comparing our synthetic evaluations to those by experienced UX practitioners across two apps, we found our evaluation identified 73% and 77% of usability issues, which exceeded the performance of 5 experienced human evaluators (57% and 63%). Compared to human evaluators, the synthetic evaluation's performance maintained consiste"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.02306","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.02306/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.02306","created_at":"2026-07-05T11:31:15.674505+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.02306v1","created_at":"2026-07-05T11:31:15.674505+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.02306","created_at":"2026-07-05T11:31:15.674505+00:00"},{"alias_kind":"pith_short_12","alias_value":"OGFAI2PGMTQ3","created_at":"2026-07-05T11:31:15.674505+00:00"},{"alias_kind":"pith_short_16","alias_value":"OGFAI2PGMTQ3OYA6","created_at":"2026-07-05T11:31:15.674505+00:00"},{"alias_kind":"pith_short_8","alias_value":"OGFAI2PG","created_at":"2026-07-05T11:31:15.674505+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29456","citing_title":"Usability Analysis of Configurator User Interfaces with Multimodal Large Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26020","citing_title":"Training Computer Use Agents to Assess the Usability of Graphical User Interfaces","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25420","citing_title":"Recommending Usability Improvements with Multimodal Large Language Models","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OGFAI2PGMTQ3OYA6LVLQ426CKN","json":"https://pith.science/pith/OGFAI2PGMTQ3OYA6LVLQ426CKN.json","graph_json":"https://pith.science/api/pith-number/OGFAI2PGMTQ3OYA6LVLQ426CKN/graph.json","events_json":"https://pith.science/api/pith-number/OGFAI2PGMTQ3OYA6LVLQ426CKN/events.json","paper":"https://pith.science/paper/OGFAI2PG"},"agent_actions":{"view_html":"https://pith.science/pith/OGFAI2PGMTQ3OYA6LVLQ426CKN","download_json":"https://pith.science/pith/OGFAI2PGMTQ3OYA6LVLQ426CKN.json","view_paper":"https://pith.science/paper/OGFAI2PG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.02306&json=true","fetch_graph":"https://pith.science/api/pith-number/OGFAI2PGMTQ3OYA6LVLQ426CKN/graph.json","fetch_events":"https://pith.science/api/pith-number/OGFAI2PGMTQ3OYA6LVLQ426CKN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OGFAI2PGMTQ3OYA6LVLQ426CKN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OGFAI2PGMTQ3OYA6LVLQ426CKN/action/storage_attestation","attest_author":"https://pith.science/pith/OGFAI2PGMTQ3OYA6LVLQ426CKN/action/author_attestation","sign_citation":"https://pith.science/pith/OGFAI2PGMTQ3OYA6LVLQ426CKN/action/citation_signature","submit_replication":"https://pith.science/pith/OGFAI2PGMTQ3OYA6LVLQ426CKN/action/replication_record"}},"created_at":"2026-07-05T11:31:15.674505+00:00","updated_at":"2026-07-05T11:31:15.674505+00:00"}