{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5UU5FHVVPQIVV2R42MNMXMIV2Z","short_pith_number":"pith:5UU5FHVV","schema_version":"1.0","canonical_sha256":"ed29d29eb57c115aea3cd31acbb115d6586111f773c6994656d682ff2e598283","source":{"kind":"arxiv","id":"2302.04434","version":1},"attestation_state":"computed","paper":{"title":"Real-Time Visual Feedback to Guide Benchmark Creation: A Human-and-Metric-in-the-Loop Workflow","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anjana Arunkumar, Bhavdeep Sachdeva, Chitta Baral, Chris Bryan, Swaroop Mishra","submitted_at":"2023-02-09T04:43:10Z","abstract_excerpt":"Recent research has shown that language models exploit `artifacts' in benchmarks to solve tasks, rather than truly learning them, leading to inflated model performance. In pursuit of creating better benchmarks, we propose VAIDA, a novel benchmark creation paradigm for NLP, that focuses on guiding crowdworkers, an under-explored facet of addressing benchmark idiosyncrasies. VAIDA facilitates sample correction by providing realtime visual feedback and recommendations to improve sample quality. Our approach is domain, model, task, and metric agnostic, and constitutes a paradigm shift for robust, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.04434","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-02-09T04:43:10Z","cross_cats_sorted":["cs.AI","cs.HC","cs.LG"],"title_canon_sha256":"e9ee1fdb45ef08db5d88c678deeec45214291840f00de5d7007a85036ef7029d","abstract_canon_sha256":"76fb4e85d35cf3cc2eab2be9051c4c028c3e136a36529267b22153983e19ac07"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:40:13.779441Z","signature_b64":"h/sAjmCpF3Fy2BrfbhUmAbcN+7qLMpo3kwikYPfRCKypHlZlg7ovXT8FjKf4QRZ/z5IFlRHN0ympkh9doaF6Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed29d29eb57c115aea3cd31acbb115d6586111f773c6994656d682ff2e598283","last_reissued_at":"2026-07-05T05:40:13.778972Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:40:13.778972Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Real-Time Visual Feedback to Guide Benchmark Creation: A Human-and-Metric-in-the-Loop Workflow","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anjana Arunkumar, Bhavdeep Sachdeva, Chitta Baral, Chris Bryan, Swaroop Mishra","submitted_at":"2023-02-09T04:43:10Z","abstract_excerpt":"Recent research has shown that language models exploit `artifacts' in benchmarks to solve tasks, rather than truly learning them, leading to inflated model performance. In pursuit of creating better benchmarks, we propose VAIDA, a novel benchmark creation paradigm for NLP, that focuses on guiding crowdworkers, an under-explored facet of addressing benchmark idiosyncrasies. VAIDA facilitates sample correction by providing realtime visual feedback and recommendations to improve sample quality. Our approach is domain, model, task, and metric agnostic, and constitutes a paradigm shift for robust, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.04434","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.04434/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.04434","created_at":"2026-07-05T05:40:13.779030+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.04434v1","created_at":"2026-07-05T05:40:13.779030+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.04434","created_at":"2026-07-05T05:40:13.779030+00:00"},{"alias_kind":"pith_short_12","alias_value":"5UU5FHVVPQIV","created_at":"2026-07-05T05:40:13.779030+00:00"},{"alias_kind":"pith_short_16","alias_value":"5UU5FHVVPQIVV2R4","created_at":"2026-07-05T05:40:13.779030+00:00"},{"alias_kind":"pith_short_8","alias_value":"5UU5FHVV","created_at":"2026-07-05T05:40:13.779030+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5UU5FHVVPQIVV2R42MNMXMIV2Z","json":"https://pith.science/pith/5UU5FHVVPQIVV2R42MNMXMIV2Z.json","graph_json":"https://pith.science/api/pith-number/5UU5FHVVPQIVV2R42MNMXMIV2Z/graph.json","events_json":"https://pith.science/api/pith-number/5UU5FHVVPQIVV2R42MNMXMIV2Z/events.json","paper":"https://pith.science/paper/5UU5FHVV"},"agent_actions":{"view_html":"https://pith.science/pith/5UU5FHVVPQIVV2R42MNMXMIV2Z","download_json":"https://pith.science/pith/5UU5FHVVPQIVV2R42MNMXMIV2Z.json","view_paper":"https://pith.science/paper/5UU5FHVV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.04434&json=true","fetch_graph":"https://pith.science/api/pith-number/5UU5FHVVPQIVV2R42MNMXMIV2Z/graph.json","fetch_events":"https://pith.science/api/pith-number/5UU5FHVVPQIVV2R42MNMXMIV2Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5UU5FHVVPQIVV2R42MNMXMIV2Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5UU5FHVVPQIVV2R42MNMXMIV2Z/action/storage_attestation","attest_author":"https://pith.science/pith/5UU5FHVVPQIVV2R42MNMXMIV2Z/action/author_attestation","sign_citation":"https://pith.science/pith/5UU5FHVVPQIVV2R42MNMXMIV2Z/action/citation_signature","submit_replication":"https://pith.science/pith/5UU5FHVVPQIVV2R42MNMXMIV2Z/action/replication_record"}},"created_at":"2026-07-05T05:40:13.779030+00:00","updated_at":"2026-07-05T05:40:13.779030+00:00"}