{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:4NEFIFBKDSRGPVKLMET4NFFTUL","short_pith_number":"pith:4NEFIFBK","schema_version":"1.0","canonical_sha256":"e34854142a1ca267d54b6127c694b3a2e0004a4b34a7b952fe58198455c6f587","source":{"kind":"arxiv","id":"1909.04849","version":1},"attestation_state":"computed","paper":{"title":"A Discrete Hard EM Approach for Weakly Supervised Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Danqi Chen, Hannaneh Hajishirzi, Luke Zettlemoyer, Sewon Min","submitted_at":"2019-09-11T04:47:36Z","abstract_excerpt":"Many question answering (QA) tasks only provide weak supervision for how the answer should be computed. For example, TriviaQA answers are entities that can be mentioned multiple times in supporting documents, while DROP answers can be computed by deriving many different equations from numbers in the reference text. In this paper, we show it is possible to convert such tasks into discrete latent variable learning problems with a precomputed, task-specific set of possible \"solutions\" (e.g. different mentions or equations) that contains one correct option. We then develop a hard EM learning schem"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.04849","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-09-11T04:47:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1311035738b08d7cbb08609f547deb98a81ef1efefa6c05fdd72bc251cde4d02","abstract_canon_sha256":"1e937a31eef2d4c9f881a5736106062c34afde32ef9ec919a59ccb7f5bc2e199"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:04:00.964232Z","signature_b64":"DCcR5VkVU3VMikymg2vQBMNeW7uGxsmztBM2dUON4BQ0FfaCQjqfR3Nl28IJ2mbeXwH5hiq4HDtQynBH2j20Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e34854142a1ca267d54b6127c694b3a2e0004a4b34a7b952fe58198455c6f587","last_reissued_at":"2026-07-05T00:04:00.963782Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:04:00.963782Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Discrete Hard EM Approach for Weakly Supervised Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Danqi Chen, Hannaneh Hajishirzi, Luke Zettlemoyer, Sewon Min","submitted_at":"2019-09-11T04:47:36Z","abstract_excerpt":"Many question answering (QA) tasks only provide weak supervision for how the answer should be computed. For example, TriviaQA answers are entities that can be mentioned multiple times in supporting documents, while DROP answers can be computed by deriving many different equations from numbers in the reference text. In this paper, we show it is possible to convert such tasks into discrete latent variable learning problems with a precomputed, task-specific set of possible \"solutions\" (e.g. different mentions or equations) that contains one correct option. We then develop a hard EM learning schem"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.04849","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.04849/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.04849","created_at":"2026-07-05T00:04:00.963838+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.04849v1","created_at":"2026-07-05T00:04:00.963838+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.04849","created_at":"2026-07-05T00:04:00.963838+00:00"},{"alias_kind":"pith_short_12","alias_value":"4NEFIFBKDSRG","created_at":"2026-07-05T00:04:00.963838+00:00"},{"alias_kind":"pith_short_16","alias_value":"4NEFIFBKDSRGPVKL","created_at":"2026-07-05T00:04:00.963838+00:00"},{"alias_kind":"pith_short_8","alias_value":"4NEFIFBK","created_at":"2026-07-05T00:04:00.963838+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2002.08909","citing_title":"REALM: Retrieval-Augmented Language Model Pre-Training","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2002.08910","citing_title":"How Much Knowledge Can You Pack Into the Parameters of a Language Model?","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4NEFIFBKDSRGPVKLMET4NFFTUL","json":"https://pith.science/pith/4NEFIFBKDSRGPVKLMET4NFFTUL.json","graph_json":"https://pith.science/api/pith-number/4NEFIFBKDSRGPVKLMET4NFFTUL/graph.json","events_json":"https://pith.science/api/pith-number/4NEFIFBKDSRGPVKLMET4NFFTUL/events.json","paper":"https://pith.science/paper/4NEFIFBK"},"agent_actions":{"view_html":"https://pith.science/pith/4NEFIFBKDSRGPVKLMET4NFFTUL","download_json":"https://pith.science/pith/4NEFIFBKDSRGPVKLMET4NFFTUL.json","view_paper":"https://pith.science/paper/4NEFIFBK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.04849&json=true","fetch_graph":"https://pith.science/api/pith-number/4NEFIFBKDSRGPVKLMET4NFFTUL/graph.json","fetch_events":"https://pith.science/api/pith-number/4NEFIFBKDSRGPVKLMET4NFFTUL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4NEFIFBKDSRGPVKLMET4NFFTUL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4NEFIFBKDSRGPVKLMET4NFFTUL/action/storage_attestation","attest_author":"https://pith.science/pith/4NEFIFBKDSRGPVKLMET4NFFTUL/action/author_attestation","sign_citation":"https://pith.science/pith/4NEFIFBKDSRGPVKLMET4NFFTUL/action/citation_signature","submit_replication":"https://pith.science/pith/4NEFIFBKDSRGPVKLMET4NFFTUL/action/replication_record"}},"created_at":"2026-07-05T00:04:00.963838+00:00","updated_at":"2026-07-05T00:04:00.963838+00:00"}