{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:O4OJYYCAPGPBBLDXP4NYF65FMH","short_pith_number":"pith:O4OJYYCA","schema_version":"1.0","canonical_sha256":"771c9c6040799e10ac777f1b82fba561febb1c72c0fcc89dce336d149e23ca55","source":{"kind":"arxiv","id":"2010.11982","version":1},"attestation_state":"computed","paper":{"title":"The Turking Test: Can Language Models Understand Instructions?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Avia Efrat, Omer Levy","submitted_at":"2020-10-22T18:44:16Z","abstract_excerpt":"Supervised machine learning provides the learner with a set of input-output examples of the target task. Humans, however, can also learn to perform new tasks from instructions in natural language. Can machines learn to understand instructions as well? We present the Turking Test, which examines a model's ability to follow natural language instructions of varying complexity. These range from simple tasks, like retrieving the nth word of a sentence, to ones that require creativity, such as generating examples for SNLI and SQuAD in place of human intelligence workers (\"turkers\"). Despite our leni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.11982","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-10-22T18:44:16Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d3d8c387f079e35f6b0e6bd0c690ac045c0e416eb014215cdeedb2f2e507e243","abstract_canon_sha256":"907e9c0fe785f6834defbe744a8437fb6a201a524a0d04323ce21c98ae5fcd30"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:45:32.331258Z","signature_b64":"JWb0jzbLD0RASiq0TjT68TlYOq2iEB/U3XVE45f2KA+lqfLZA0h2sOnngZWACJU/QtUqI2u0odn2U+kaWge3AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"771c9c6040799e10ac777f1b82fba561febb1c72c0fcc89dce336d149e23ca55","last_reissued_at":"2026-07-05T01:45:32.330833Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:45:32.330833Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Turking Test: Can Language Models Understand Instructions?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Avia Efrat, Omer Levy","submitted_at":"2020-10-22T18:44:16Z","abstract_excerpt":"Supervised machine learning provides the learner with a set of input-output examples of the target task. Humans, however, can also learn to perform new tasks from instructions in natural language. Can machines learn to understand instructions as well? We present the Turking Test, which examines a model's ability to follow natural language instructions of varying complexity. These range from simple tasks, like retrieving the nth word of a sentence, to ones that require creativity, such as generating examples for SNLI and SQuAD in place of human intelligence workers (\"turkers\"). Despite our leni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.11982","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.11982/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.11982","created_at":"2026-07-05T01:45:32.330892+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.11982v1","created_at":"2026-07-05T01:45:32.330892+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.11982","created_at":"2026-07-05T01:45:32.330892+00:00"},{"alias_kind":"pith_short_12","alias_value":"O4OJYYCAPGPB","created_at":"2026-07-05T01:45:32.330892+00:00"},{"alias_kind":"pith_short_16","alias_value":"O4OJYYCAPGPBBLDX","created_at":"2026-07-05T01:45:32.330892+00:00"},{"alias_kind":"pith_short_8","alias_value":"O4OJYYCA","created_at":"2026-07-05T01:45:32.330892+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2104.08773","citing_title":"Cross-Task Generalization via Natural Language Crowdsourcing Instructions","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2406.10162","citing_title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","ref_index":152,"is_internal_anchor":false},{"citing_arxiv_id":"2305.11206","citing_title":"LIMA: Less Is More for Alignment","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08104","citing_title":"AgentCrypt: Advancing Privacy and (Secure) Computation in AI Agent Collaboration","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2202.12837","citing_title":"Rethinking the Role of Demonstrations: What Makes In-Context Learning Work?","ref_index":204,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O4OJYYCAPGPBBLDXP4NYF65FMH","json":"https://pith.science/pith/O4OJYYCAPGPBBLDXP4NYF65FMH.json","graph_json":"https://pith.science/api/pith-number/O4OJYYCAPGPBBLDXP4NYF65FMH/graph.json","events_json":"https://pith.science/api/pith-number/O4OJYYCAPGPBBLDXP4NYF65FMH/events.json","paper":"https://pith.science/paper/O4OJYYCA"},"agent_actions":{"view_html":"https://pith.science/pith/O4OJYYCAPGPBBLDXP4NYF65FMH","download_json":"https://pith.science/pith/O4OJYYCAPGPBBLDXP4NYF65FMH.json","view_paper":"https://pith.science/paper/O4OJYYCA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.11982&json=true","fetch_graph":"https://pith.science/api/pith-number/O4OJYYCAPGPBBLDXP4NYF65FMH/graph.json","fetch_events":"https://pith.science/api/pith-number/O4OJYYCAPGPBBLDXP4NYF65FMH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O4OJYYCAPGPBBLDXP4NYF65FMH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O4OJYYCAPGPBBLDXP4NYF65FMH/action/storage_attestation","attest_author":"https://pith.science/pith/O4OJYYCAPGPBBLDXP4NYF65FMH/action/author_attestation","sign_citation":"https://pith.science/pith/O4OJYYCAPGPBBLDXP4NYF65FMH/action/citation_signature","submit_replication":"https://pith.science/pith/O4OJYYCAPGPBBLDXP4NYF65FMH/action/replication_record"}},"created_at":"2026-07-05T01:45:32.330892+00:00","updated_at":"2026-07-05T01:45:32.330892+00:00"}