{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:BG3KMTVQ46YRA526T7UDECACT5","short_pith_number":"pith:BG3KMTVQ","schema_version":"1.0","canonical_sha256":"09b6a64eb0e7b110775e9fe83208029f4b34a7081bea5ee493bb576741c4e881","source":{"kind":"arxiv","id":"1904.09286","version":2},"attestation_state":"computed","paper":{"title":"Unifying Question Answering, Text Classification, and Regression via Span Extraction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bryan McCann, Caiming Xiong, Nitish Shirish Keskar, Richard Socher","submitted_at":"2019-04-19T17:58:29Z","abstract_excerpt":"Even as pre-trained language encoders such as BERT are shared across many tasks, the output layers of question answering, text classification, and regression models are significantly different. Span decoders are frequently used for question answering, fixed-class, classification layers for text classification, and similarity-scoring layers for regression tasks, We show that this distinction is not necessary and that all three can be unified as span extraction. A unified, span-extraction approach leads to superior or comparable performance in supplementary supervised pre-trained, low-data, and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1904.09286","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-04-19T17:58:29Z","cross_cats_sorted":[],"title_canon_sha256":"4d4a6a9ca0c3eafa880958dca2faeb240c7e0e65699ca963d6e7daa3c125f58c","abstract_canon_sha256":"016980ad7e999a192b71b66faf0094f09d855c155a088448efeec0018baf3651"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:06:07.563560Z","signature_b64":"aDbonxPah4bOK2UXfkjRjhVPYc+4qRgbYfRPITBdn++GlPshtrHJ5GwfxHzFIUHdcKY/kHCx9A3GWeSx7V6VBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"09b6a64eb0e7b110775e9fe83208029f4b34a7081bea5ee493bb576741c4e881","last_reissued_at":"2026-07-05T00:06:07.563048Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:06:07.563048Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unifying Question Answering, Text Classification, and Regression via Span Extraction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bryan McCann, Caiming Xiong, Nitish Shirish Keskar, Richard Socher","submitted_at":"2019-04-19T17:58:29Z","abstract_excerpt":"Even as pre-trained language encoders such as BERT are shared across many tasks, the output layers of question answering, text classification, and regression models are significantly different. Span decoders are frequently used for question answering, fixed-class, classification layers for text classification, and similarity-scoring layers for regression tasks, We show that this distinction is not necessary and that all three can be unified as span extraction. A unified, span-extraction approach leads to superior or comparable performance in supplementary supervised pre-trained, low-data, and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1904.09286","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1904.09286/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.09286","created_at":"2026-07-05T00:06:07.563108+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.09286v2","created_at":"2026-07-05T00:06:07.563108+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.09286","created_at":"2026-07-05T00:06:07.563108+00:00"},{"alias_kind":"pith_short_12","alias_value":"BG3KMTVQ46YR","created_at":"2026-07-05T00:06:07.563108+00:00"},{"alias_kind":"pith_short_16","alias_value":"BG3KMTVQ46YRA526","created_at":"2026-07-05T00:06:07.563108+00:00"},{"alias_kind":"pith_short_8","alias_value":"BG3KMTVQ","created_at":"2026-07-05T00:06:07.563108+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2301.13688","citing_title":"The Flan Collection: Designing Data and Methods for Effective Instruction Tuning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"1909.05858","citing_title":"CTRL: A Conditional Transformer Language Model for Controllable Generation","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BG3KMTVQ46YRA526T7UDECACT5","json":"https://pith.science/pith/BG3KMTVQ46YRA526T7UDECACT5.json","graph_json":"https://pith.science/api/pith-number/BG3KMTVQ46YRA526T7UDECACT5/graph.json","events_json":"https://pith.science/api/pith-number/BG3KMTVQ46YRA526T7UDECACT5/events.json","paper":"https://pith.science/paper/BG3KMTVQ"},"agent_actions":{"view_html":"https://pith.science/pith/BG3KMTVQ46YRA526T7UDECACT5","download_json":"https://pith.science/pith/BG3KMTVQ46YRA526T7UDECACT5.json","view_paper":"https://pith.science/paper/BG3KMTVQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.09286&json=true","fetch_graph":"https://pith.science/api/pith-number/BG3KMTVQ46YRA526T7UDECACT5/graph.json","fetch_events":"https://pith.science/api/pith-number/BG3KMTVQ46YRA526T7UDECACT5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BG3KMTVQ46YRA526T7UDECACT5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BG3KMTVQ46YRA526T7UDECACT5/action/storage_attestation","attest_author":"https://pith.science/pith/BG3KMTVQ46YRA526T7UDECACT5/action/author_attestation","sign_citation":"https://pith.science/pith/BG3KMTVQ46YRA526T7UDECACT5/action/citation_signature","submit_replication":"https://pith.science/pith/BG3KMTVQ46YRA526T7UDECACT5/action/replication_record"}},"created_at":"2026-07-05T00:06:07.563108+00:00","updated_at":"2026-07-05T00:06:07.563108+00:00"}