{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:TYX2EZZPLENNK3WYOZLONHJIRN","short_pith_number":"pith:TYX2EZZP","schema_version":"1.0","canonical_sha256":"9e2fa2672f591ad56ed87656e69d288b61c866701e62335d9057e6b18d868e34","source":{"kind":"arxiv","id":"2106.05346","version":2},"attestation_state":"computed","paper":{"title":"End-to-End Training of Multi-Document Reader and Retriever for Open-Domain Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Chris Dyer, Dani Yogatama, Devendra Singh Sachan, Siva Reddy, William Hamilton","submitted_at":"2021-06-09T19:25:37Z","abstract_excerpt":"We present an end-to-end differentiable training method for retrieval-augmented open-domain question answering systems that combine information from multiple retrieved documents when generating answers. We model retrieval decisions as latent variables over sets of relevant documents. Since marginalizing over sets of retrieved documents is computationally hard, we approximate this using an expectation-maximization algorithm. We iteratively estimate the value of our latent variable (the set of relevant documents for a given question) and then use this estimate to update the retriever and reader "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.05346","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-06-09T19:25:37Z","cross_cats_sorted":["cs.AI","cs.IR"],"title_canon_sha256":"ed846b00b61ed99a35c852470a493dc75e874d643b044afbb3f35a02f3dd72fc","abstract_canon_sha256":"ddc594dce2ecdde9b9d059ddadeb4860a76abbca3347afc0a13780725d874d62"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:37:43.271216Z","signature_b64":"aAHJaDnHycQLV9ob6iCD+5T8/3EEzWuYP1GjajcI8+4ifTt+bUJWUBmO1SsJi/Q5M8f2ri52YmBktmAa218vDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e2fa2672f591ad56ed87656e69d288b61c866701e62335d9057e6b18d868e34","last_reissued_at":"2026-07-05T03:37:43.270669Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:37:43.270669Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"End-to-End Training of Multi-Document Reader and Retriever for Open-Domain Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Chris Dyer, Dani Yogatama, Devendra Singh Sachan, Siva Reddy, William Hamilton","submitted_at":"2021-06-09T19:25:37Z","abstract_excerpt":"We present an end-to-end differentiable training method for retrieval-augmented open-domain question answering systems that combine information from multiple retrieved documents when generating answers. We model retrieval decisions as latent variables over sets of relevant documents. Since marginalizing over sets of retrieved documents is computationally hard, we approximate this using an expectation-maximization algorithm. We iteratively estimate the value of our latent variable (the set of relevant documents for a given question) and then use this estimate to update the retriever and reader "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.05346","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.05346/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.05346","created_at":"2026-07-05T03:37:43.270738+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.05346v2","created_at":"2026-07-05T03:37:43.270738+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.05346","created_at":"2026-07-05T03:37:43.270738+00:00"},{"alias_kind":"pith_short_12","alias_value":"TYX2EZZPLENN","created_at":"2026-07-05T03:37:43.270738+00:00"},{"alias_kind":"pith_short_16","alias_value":"TYX2EZZPLENNK3WY","created_at":"2026-07-05T03:37:43.270738+00:00"},{"alias_kind":"pith_short_8","alias_value":"TYX2EZZP","created_at":"2026-07-05T03:37:43.270738+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2112.04426","citing_title":"Improving language models by retrieving from trillions of tokens","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2208.03299","citing_title":"Atlas: Few-shot Learning with Retrieval Augmented Language Models","ref_index":241,"is_internal_anchor":false},{"citing_arxiv_id":"2208.03299","citing_title":"Atlas: Few-shot Learning with Retrieval Augmented Language Models","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TYX2EZZPLENNK3WYOZLONHJIRN","json":"https://pith.science/pith/TYX2EZZPLENNK3WYOZLONHJIRN.json","graph_json":"https://pith.science/api/pith-number/TYX2EZZPLENNK3WYOZLONHJIRN/graph.json","events_json":"https://pith.science/api/pith-number/TYX2EZZPLENNK3WYOZLONHJIRN/events.json","paper":"https://pith.science/paper/TYX2EZZP"},"agent_actions":{"view_html":"https://pith.science/pith/TYX2EZZPLENNK3WYOZLONHJIRN","download_json":"https://pith.science/pith/TYX2EZZPLENNK3WYOZLONHJIRN.json","view_paper":"https://pith.science/paper/TYX2EZZP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.05346&json=true","fetch_graph":"https://pith.science/api/pith-number/TYX2EZZPLENNK3WYOZLONHJIRN/graph.json","fetch_events":"https://pith.science/api/pith-number/TYX2EZZPLENNK3WYOZLONHJIRN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TYX2EZZPLENNK3WYOZLONHJIRN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TYX2EZZPLENNK3WYOZLONHJIRN/action/storage_attestation","attest_author":"https://pith.science/pith/TYX2EZZPLENNK3WYOZLONHJIRN/action/author_attestation","sign_citation":"https://pith.science/pith/TYX2EZZPLENNK3WYOZLONHJIRN/action/citation_signature","submit_replication":"https://pith.science/pith/TYX2EZZPLENNK3WYOZLONHJIRN/action/replication_record"}},"created_at":"2026-07-05T03:37:43.270738+00:00","updated_at":"2026-07-05T03:37:43.270738+00:00"}