{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OLVI6XL5I6DY52LLUYP3XSBDW6","short_pith_number":"pith:OLVI6XL5","schema_version":"1.0","canonical_sha256":"72ea8f5d7d47878ee96ba61fbbc823b7b247b44abf45dd961678dfa6e1a25b5d","source":{"kind":"arxiv","id":"2410.02525","version":4},"attestation_state":"computed","paper":{"title":"Contextual Document Embeddings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alexander M. Rush, John X. Morris","submitted_at":"2024-10-03T14:33:34Z","abstract_excerpt":"Dense document embeddings are central to neural retrieval. The dominant paradigm is to train and construct embeddings by running encoders directly on individual documents. In this work, we argue that these embeddings, while effective, are implicitly out-of-context for targeted use cases of retrieval, and that a contextualized document embedding should take into account both the document and neighboring documents in context - analogous to contextualized word embeddings. We propose two complementary methods for contextualized document embeddings: first, an alternative contrastive learning object"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02525","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-03T14:33:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5e562e9749d4c84b256997a84d492bbb906665832bbe1f1fff2682d537444751","abstract_canon_sha256":"1a294529f25be026c316c2bf2feedc69db9f55f82de14136d610a4adc6da9bfc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:32:48.518918Z","signature_b64":"iE4gVPRVjgJk6eh+jNMHJ8eNvDFRW1+oXmTl6i6k5d2hQoVIVRLXRI6vhLf6GrA2y9F2/3pKorIuEYL8a1NeBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"72ea8f5d7d47878ee96ba61fbbc823b7b247b44abf45dd961678dfa6e1a25b5d","last_reissued_at":"2026-07-05T09:32:48.518410Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:32:48.518410Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Contextual Document Embeddings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alexander M. Rush, John X. Morris","submitted_at":"2024-10-03T14:33:34Z","abstract_excerpt":"Dense document embeddings are central to neural retrieval. The dominant paradigm is to train and construct embeddings by running encoders directly on individual documents. In this work, we argue that these embeddings, while effective, are implicitly out-of-context for targeted use cases of retrieval, and that a contextualized document embedding should take into account both the document and neighboring documents in context - analogous to contextualized word embeddings. We propose two complementary methods for contextualized document embeddings: first, an alternative contrastive learning object"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02525","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02525/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02525","created_at":"2026-07-05T09:32:48.518469+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02525v4","created_at":"2026-07-05T09:32:48.518469+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02525","created_at":"2026-07-05T09:32:48.518469+00:00"},{"alias_kind":"pith_short_12","alias_value":"OLVI6XL5I6DY","created_at":"2026-07-05T09:32:48.518469+00:00"},{"alias_kind":"pith_short_16","alias_value":"OLVI6XL5I6DY52LL","created_at":"2026-07-05T09:32:48.518469+00:00"},{"alias_kind":"pith_short_8","alias_value":"OLVI6XL5","created_at":"2026-07-05T09:32:48.518469+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02293","citing_title":"AI as a Tool for Simulation-Based Experiments in Literary Studies","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OLVI6XL5I6DY52LLUYP3XSBDW6","json":"https://pith.science/pith/OLVI6XL5I6DY52LLUYP3XSBDW6.json","graph_json":"https://pith.science/api/pith-number/OLVI6XL5I6DY52LLUYP3XSBDW6/graph.json","events_json":"https://pith.science/api/pith-number/OLVI6XL5I6DY52LLUYP3XSBDW6/events.json","paper":"https://pith.science/paper/OLVI6XL5"},"agent_actions":{"view_html":"https://pith.science/pith/OLVI6XL5I6DY52LLUYP3XSBDW6","download_json":"https://pith.science/pith/OLVI6XL5I6DY52LLUYP3XSBDW6.json","view_paper":"https://pith.science/paper/OLVI6XL5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02525&json=true","fetch_graph":"https://pith.science/api/pith-number/OLVI6XL5I6DY52LLUYP3XSBDW6/graph.json","fetch_events":"https://pith.science/api/pith-number/OLVI6XL5I6DY52LLUYP3XSBDW6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OLVI6XL5I6DY52LLUYP3XSBDW6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OLVI6XL5I6DY52LLUYP3XSBDW6/action/storage_attestation","attest_author":"https://pith.science/pith/OLVI6XL5I6DY52LLUYP3XSBDW6/action/author_attestation","sign_citation":"https://pith.science/pith/OLVI6XL5I6DY52LLUYP3XSBDW6/action/citation_signature","submit_replication":"https://pith.science/pith/OLVI6XL5I6DY52LLUYP3XSBDW6/action/replication_record"}},"created_at":"2026-07-05T09:32:48.518469+00:00","updated_at":"2026-07-05T09:32:48.518469+00:00"}