{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:F42IR3YAJWF7IEKBGO2HJAWEUG","short_pith_number":"pith:F42IR3YA","schema_version":"1.0","canonical_sha256":"2f3488ef004d8bf4114133b47482c4a185f121f5448ca76b048e51c7bc2e2dc4","source":{"kind":"arxiv","id":"2206.12390","version":2},"attestation_state":"computed","paper":{"title":"A Test for Evaluating Performance in Human-Computer Systems","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.HC","authors_text":"Abdullah Almaatouq, Andres Campero, Haoran Wen, Jaeyoon Song, Michelle Vaccaro, Thomas W. Malone","submitted_at":"2022-06-24T17:44:58Z","abstract_excerpt":"The Turing test for comparing computer performance to that of humans is well known, but, surprisingly, there is no widely used test for comparing how much better human-computer systems perform relative to humans alone, computers alone, or other baselines. Here, we show how to perform such a test using the ratio of means as a measure of effect size. Then we demonstrate the use of this test in three ways. First, in an analysis of 79 recently published experimental results, we find that, surprisingly, over half of the studies find a decrease in performance, the mean and median ratios of performan"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.12390","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.HC","submitted_at":"2022-06-24T17:44:58Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"74257ff5109f952ad8cc45593f6e4569258d1de888d6ca7fe935171bf99d5c98","abstract_canon_sha256":"92580edf67b0860223fbfb356c46a43c2e9fd753a23ca7bcda9c72a67c70ebbb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:36:01.704939Z","signature_b64":"PqsSlGM8tw9zFH8XcQUBfOuvZAZSueCYSFh4ljIJdWWzfqss9uRA5K9b8/J7uUJ4FQuJqnkL2HyGCZjGem92Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f3488ef004d8bf4114133b47482c4a185f121f5448ca76b048e51c7bc2e2dc4","last_reissued_at":"2026-07-05T04:36:01.704460Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:36:01.704460Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Test for Evaluating Performance in Human-Computer Systems","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.HC","authors_text":"Abdullah Almaatouq, Andres Campero, Haoran Wen, Jaeyoon Song, Michelle Vaccaro, Thomas W. Malone","submitted_at":"2022-06-24T17:44:58Z","abstract_excerpt":"The Turing test for comparing computer performance to that of humans is well known, but, surprisingly, there is no widely used test for comparing how much better human-computer systems perform relative to humans alone, computers alone, or other baselines. Here, we show how to perform such a test using the ratio of means as a measure of effect size. Then we demonstrate the use of this test in three ways. First, in an analysis of 79 recently published experimental results, we find that, surprisingly, over half of the studies find a decrease in performance, the mean and median ratios of performan"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.12390","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.12390/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.12390","created_at":"2026-07-05T04:36:01.704520+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.12390v2","created_at":"2026-07-05T04:36:01.704520+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.12390","created_at":"2026-07-05T04:36:01.704520+00:00"},{"alias_kind":"pith_short_12","alias_value":"F42IR3YAJWF7","created_at":"2026-07-05T04:36:01.704520+00:00"},{"alias_kind":"pith_short_16","alias_value":"F42IR3YAJWF7IEKB","created_at":"2026-07-05T04:36:01.704520+00:00"},{"alias_kind":"pith_short_8","alias_value":"F42IR3YA","created_at":"2026-07-05T04:36:01.704520+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10125","citing_title":"Useful for Exploration, Risky for Precision: Evaluating AI Tools in Academic Research","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10125","citing_title":"Useful for Exploration, Risky for Precision: Evaluating AI Tools in Academic Research","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F42IR3YAJWF7IEKBGO2HJAWEUG","json":"https://pith.science/pith/F42IR3YAJWF7IEKBGO2HJAWEUG.json","graph_json":"https://pith.science/api/pith-number/F42IR3YAJWF7IEKBGO2HJAWEUG/graph.json","events_json":"https://pith.science/api/pith-number/F42IR3YAJWF7IEKBGO2HJAWEUG/events.json","paper":"https://pith.science/paper/F42IR3YA"},"agent_actions":{"view_html":"https://pith.science/pith/F42IR3YAJWF7IEKBGO2HJAWEUG","download_json":"https://pith.science/pith/F42IR3YAJWF7IEKBGO2HJAWEUG.json","view_paper":"https://pith.science/paper/F42IR3YA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.12390&json=true","fetch_graph":"https://pith.science/api/pith-number/F42IR3YAJWF7IEKBGO2HJAWEUG/graph.json","fetch_events":"https://pith.science/api/pith-number/F42IR3YAJWF7IEKBGO2HJAWEUG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F42IR3YAJWF7IEKBGO2HJAWEUG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F42IR3YAJWF7IEKBGO2HJAWEUG/action/storage_attestation","attest_author":"https://pith.science/pith/F42IR3YAJWF7IEKBGO2HJAWEUG/action/author_attestation","sign_citation":"https://pith.science/pith/F42IR3YAJWF7IEKBGO2HJAWEUG/action/citation_signature","submit_replication":"https://pith.science/pith/F42IR3YAJWF7IEKBGO2HJAWEUG/action/replication_record"}},"created_at":"2026-07-05T04:36:01.704520+00:00","updated_at":"2026-07-05T04:36:01.704520+00:00"}