{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:OCST54C7AW75KG5HL7ZXHVXVV6","short_pith_number":"pith:OCST54C7","schema_version":"1.0","canonical_sha256":"70a53ef05f05bfd51ba75ff373d6f5af9632c58ed749d16891bc0bac43fe0cca","source":{"kind":"arxiv","id":"2604.25420","version":1},"attestation_state":"computed","paper":{"title":"Recommending Usability Improvements with Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Multimodal large language models identify usability issues in screen recordings and suggest ranked fixes.","cross_cats":["cs.HC"],"primary_cat":"cs.SE","authors_text":"Alexander Felfernig, Damian Garber, Manuel Henrich, Sebastian Lubos, Viet-Man Le","submitted_at":"2026-04-28T09:29:13Z","abstract_excerpt":"Usability describes quality attributes of application user interfaces that determine how effectively users can interact with them. Traditional usability evaluation methods require considerable expertise and resources, which can be challenging, especially for small teams and organizations. Automating usability evaluation could make it more accessible and help to improve the user experience. The recent emergence of powerful multimodal large language models (MLLMs) has opened new opportunities for automating usability evaluation and recommendation of improvements. These models can process visual "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":true,"formal_links_present":false},"canonical_record":{"source":{"id":"2604.25420","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2026-04-28T09:29:13Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"b2711cfca3e279f68c6175bbd4b9e0cd989c304c647686ec1f6cda82335a7231","abstract_canon_sha256":"99863a0a8e037dac0600849cbe98cc001a2410d94651c0f469003aa5eda86e9f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-12T01:09:28.100452Z","signature_b64":"bDyJxiw4PoTmvj5ZQMOWVl/iwWMBcd+DfezsF9VZtfVhTtaRcjZC7XJbL/PpscYP3NnfGGWUXoSqTiw/kDyuAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"70a53ef05f05bfd51ba75ff373d6f5af9632c58ed749d16891bc0bac43fe0cca","last_reissued_at":"2026-06-12T01:09:28.100026Z","signature_status":"signed_v1","first_computed_at":"2026-06-12T01:09:28.100026Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Recommending Usability Improvements with Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Multimodal large language models identify usability issues in screen recordings and suggest ranked fixes.","cross_cats":["cs.HC"],"primary_cat":"cs.SE","authors_text":"Alexander Felfernig, Damian Garber, Manuel Henrich, Sebastian Lubos, Viet-Man Le","submitted_at":"2026-04-28T09:29:13Z","abstract_excerpt":"Usability describes quality attributes of application user interfaces that determine how effectively users can interact with them. Traditional usability evaluation methods require considerable expertise and resources, which can be challenging, especially for small teams and organizations. Automating usability evaluation could make it more accessible and help to improve the user experience. The recent emergence of powerful multimodal large language models (MLLMs) has opened new opportunities for automating usability evaluation and recommendation of improvements. These models can process visual "},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"The results demonstrate the potential of our approach to provide low-effort usability improvement recommendations. This makes it a promising complement to traditional evaluation methods, especially in settings with limited access to usability experts.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the MLLM can reliably identify real usability issues and produce actionable, correctly ranked recommendations from only limited context and screen recordings, and that feedback from a small group of software engineers in the user study is sufficient to establish practical usefulness.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Multimodal LLMs can detect usability issues from screen recordings, explain them via Nielsen's heuristics, and rank improvement recommendations, with engineer feedback indicating practical usefulness for teams lacking experts.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Multimodal large language models identify usability issues in screen recordings and suggest ranked fixes.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"5598b8e37432bd4b710e5b07a2e778814dfe3a58ec36a019ed6fca91ec59bd01"},"source":{"id":"2604.25420","kind":"arxiv","version":1},"verdict":{"id":"20eecc75-ef09-47ae-afb3-f80edd4b495e","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-07T16:19:35.347285Z","strongest_claim":"The results demonstrate the potential of our approach to provide low-effort usability improvement recommendations. This makes it a promising complement to traditional evaluation methods, especially in settings with limited access to usability experts.","one_line_summary":"Multimodal LLMs can detect usability issues from screen recordings, explain them via Nielsen's heuristics, and rank improvement recommendations, with engineer feedback indicating practical usefulness for teams lacking experts.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the MLLM can reliably identify real usability issues and produce actionable, correctly ranked recommendations from only limited context and screen recordings, and that feedback from a small group of software engineers in the user study is sufficient to establish practical usefulness.","pith_extraction_headline":"Multimodal large language models identify usability issues in screen recordings and suggest ranked fixes."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.25420/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-21T04:40:20.878874Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T21:09:24.676869Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"bb2fdfb1bea6394b323515233e22e8680f52364de97a6339ee764c5e477331ce"},"references":{"count":33,"sample":[{"doi":"10.1145/3540250.3549118","year":2022,"title":"Abdulaziz Alshayban and Sam Malek. 2022. AccessiText: automated detection of text accessibility issues in Android apps. InProceedings of the 30th ACM Joint European Software Engineering Conference and","work_id":"ae1a0c73-8cf6-48b5-b0f6-61a6ef0e68c9","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"10.1109/tse.2013.29","year":2013,"title":"Moreno, María-Isabel Sánchez-Segura, and Ahmed Sef- fah","work_id":"ed77506b-33eb-4f93-9fde-91ec5a23accd","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2022,"title":"Castro, Ignacio Garnica, and Luis A","work_id":"dfc3a5a1-7b4a-4b27-b80c-60dd8ed1a9ee","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2023,"title":"Xiang Deng, Yu Gu, Boyuan Zheng, Shijie Chen, Samuel Stevens, Boshi Wang, Huan Sun, and Yu Su. 2023. MIND2WEB: towards a generalist agent for the web. InProceedings of the 37th International Conferenc","work_id":"3ee8df5a-ae38-46bc-a724-ef3308e48dde","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"10.1145/3613904.3642782","year":2024,"title":"Peitong Duan, Jeremy Warner, Yang Li, and Bjoern Hartmann. 2024. Generating Automatic Feedback on UI Mockups with Large Language Models. InProceedings of the 2024 CHI Conference on Human Factors in Co","work_id":"57aa781b-f03e-410f-889b-61aa5b53e99e","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":33,"snapshot_sha256":"8ba3d9dfa4c355880357b45f6f724488da634a7455357dec2cc286e2a12b838a","internal_anchors":3},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2604.25420","created_at":"2026-06-12T01:09:28.100083+00:00"},{"alias_kind":"arxiv_version","alias_value":"2604.25420v1","created_at":"2026-06-12T01:09:28.100083+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.25420","created_at":"2026-06-12T01:09:28.100083+00:00"},{"alias_kind":"pith_short_12","alias_value":"OCST54C7AW75","created_at":"2026-06-12T01:09:28.100083+00:00"},{"alias_kind":"pith_short_16","alias_value":"OCST54C7AW75KG5H","created_at":"2026-06-12T01:09:28.100083+00:00"},{"alias_kind":"pith_short_8","alias_value":"OCST54C7","created_at":"2026-06-12T01:09:28.100083+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OCST54C7AW75KG5HL7ZXHVXVV6","json":"https://pith.science/pith/OCST54C7AW75KG5HL7ZXHVXVV6.json","graph_json":"https://pith.science/api/pith-number/OCST54C7AW75KG5HL7ZXHVXVV6/graph.json","events_json":"https://pith.science/api/pith-number/OCST54C7AW75KG5HL7ZXHVXVV6/events.json","paper":"https://pith.science/paper/OCST54C7"},"agent_actions":{"view_html":"https://pith.science/pith/OCST54C7AW75KG5HL7ZXHVXVV6","download_json":"https://pith.science/pith/OCST54C7AW75KG5HL7ZXHVXVV6.json","view_paper":"https://pith.science/paper/OCST54C7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2604.25420&json=true","fetch_graph":"https://pith.science/api/pith-number/OCST54C7AW75KG5HL7ZXHVXVV6/graph.json","fetch_events":"https://pith.science/api/pith-number/OCST54C7AW75KG5HL7ZXHVXVV6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OCST54C7AW75KG5HL7ZXHVXVV6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OCST54C7AW75KG5HL7ZXHVXVV6/action/storage_attestation","attest_author":"https://pith.science/pith/OCST54C7AW75KG5HL7ZXHVXVV6/action/author_attestation","sign_citation":"https://pith.science/pith/OCST54C7AW75KG5HL7ZXHVXVV6/action/citation_signature","submit_replication":"https://pith.science/pith/OCST54C7AW75KG5HL7ZXHVXVV6/action/replication_record"}},"created_at":"2026-06-12T01:09:28.100083+00:00","updated_at":"2026-06-12T01:09:28.100083+00:00"}