{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XT2YWUJHUWBJ7LS4RLETQSOGBK","short_pith_number":"pith:XT2YWUJH","schema_version":"1.0","canonical_sha256":"bcf58b5127a5829fae5c8ac93849c60ab0a67888a996f0659524fdc5c94bdee8","source":{"kind":"arxiv","id":"2112.09062","version":3},"attestation_state":"computed","paper":{"title":"Models in the Loop: Aiding Crowdworkers with Generative Annotation Assistants","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Douwe Kiela, Max Bartolo, Pontus Stenetorp, Robin Jia, Sebastian Riedel, Tristan Thrush","submitted_at":"2021-12-16T17:59:39Z","abstract_excerpt":"In Dynamic Adversarial Data Collection (DADC), human annotators are tasked with finding examples that models struggle to predict correctly. Models trained on DADC-collected training data have been shown to be more robust in adversarial and out-of-domain settings, and are considerably harder for humans to fool. However, DADC is more time-consuming than traditional data collection and thus more costly per annotated example. In this work, we examine whether we can maintain the advantages of DADC, without incurring the additional cost. To that end, we introduce Generative Annotation Assistants (GA"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.09062","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-12-16T17:59:39Z","cross_cats_sorted":[],"title_canon_sha256":"29f85b02b81d798bd8f31e691bb8226e293907a235240fe67529a9ae8eb9b146","abstract_canon_sha256":"e8c0793e71561e8ec1ea5619d06916a550cf468ccee6c8856927e013c3d55021"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:23:48.610430Z","signature_b64":"K2zW3OkUlcgcRLw7qlUxPNmu6qAOP1qWVP6pN728f+hP8+2SODjsTesjurVLv0juiT9dkk2NyKWhS7nV/zf0AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bcf58b5127a5829fae5c8ac93849c60ab0a67888a996f0659524fdc5c94bdee8","last_reissued_at":"2026-07-05T04:23:48.609896Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:23:48.609896Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Models in the Loop: Aiding Crowdworkers with Generative Annotation Assistants","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Douwe Kiela, Max Bartolo, Pontus Stenetorp, Robin Jia, Sebastian Riedel, Tristan Thrush","submitted_at":"2021-12-16T17:59:39Z","abstract_excerpt":"In Dynamic Adversarial Data Collection (DADC), human annotators are tasked with finding examples that models struggle to predict correctly. Models trained on DADC-collected training data have been shown to be more robust in adversarial and out-of-domain settings, and are considerably harder for humans to fool. However, DADC is more time-consuming than traditional data collection and thus more costly per annotated example. In this work, we examine whether we can maintain the advantages of DADC, without incurring the additional cost. To that end, we introduce Generative Annotation Assistants (GA"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.09062","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.09062/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.09062","created_at":"2026-07-05T04:23:48.609961+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.09062v3","created_at":"2026-07-05T04:23:48.609961+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.09062","created_at":"2026-07-05T04:23:48.609961+00:00"},{"alias_kind":"pith_short_12","alias_value":"XT2YWUJHUWBJ","created_at":"2026-07-05T04:23:48.609961+00:00"},{"alias_kind":"pith_short_16","alias_value":"XT2YWUJHUWBJ7LS4","created_at":"2026-07-05T04:23:48.609961+00:00"},{"alias_kind":"pith_short_8","alias_value":"XT2YWUJH","created_at":"2026-07-05T04:23:48.609961+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2406.10162","citing_title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2303.09014","citing_title":"ART: Automatic multi-step reasoning and tool-use for large language models","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2212.09251","citing_title":"Discovering Language Model Behaviors with Model-Written Evaluations","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2310.08419","citing_title":"Jailbreaking Black Box Large Language Models in Twenty Queries","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XT2YWUJHUWBJ7LS4RLETQSOGBK","json":"https://pith.science/pith/XT2YWUJHUWBJ7LS4RLETQSOGBK.json","graph_json":"https://pith.science/api/pith-number/XT2YWUJHUWBJ7LS4RLETQSOGBK/graph.json","events_json":"https://pith.science/api/pith-number/XT2YWUJHUWBJ7LS4RLETQSOGBK/events.json","paper":"https://pith.science/paper/XT2YWUJH"},"agent_actions":{"view_html":"https://pith.science/pith/XT2YWUJHUWBJ7LS4RLETQSOGBK","download_json":"https://pith.science/pith/XT2YWUJHUWBJ7LS4RLETQSOGBK.json","view_paper":"https://pith.science/paper/XT2YWUJH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.09062&json=true","fetch_graph":"https://pith.science/api/pith-number/XT2YWUJHUWBJ7LS4RLETQSOGBK/graph.json","fetch_events":"https://pith.science/api/pith-number/XT2YWUJHUWBJ7LS4RLETQSOGBK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XT2YWUJHUWBJ7LS4RLETQSOGBK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XT2YWUJHUWBJ7LS4RLETQSOGBK/action/storage_attestation","attest_author":"https://pith.science/pith/XT2YWUJHUWBJ7LS4RLETQSOGBK/action/author_attestation","sign_citation":"https://pith.science/pith/XT2YWUJHUWBJ7LS4RLETQSOGBK/action/citation_signature","submit_replication":"https://pith.science/pith/XT2YWUJHUWBJ7LS4RLETQSOGBK/action/replication_record"}},"created_at":"2026-07-05T04:23:48.609961+00:00","updated_at":"2026-07-05T04:23:48.609961+00:00"}