{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JGH5G6LCDBGQ356TV7MVFQMMCZ","short_pith_number":"pith:JGH5G6LC","schema_version":"1.0","canonical_sha256":"498fd37962184d0df7d3afd952c18c166bad1bc296ff9f0719af4a43ab9ed6c1","source":{"kind":"arxiv","id":"2408.03402","version":1},"attestation_state":"computed","paper":{"title":"ULLME: A Unified Framework for Large Language Model Embeddings with Generation-Augmented Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Franck Dernoncourt, Hieu Man, Nghia Trung Ngo, Thien Huu Nguyen","submitted_at":"2024-08-06T18:53:54Z","abstract_excerpt":"Large Language Models (LLMs) excel in various natural language processing tasks, but leveraging them for dense passage embedding remains challenging. This is due to their causal attention mechanism and the misalignment between their pre-training objectives and the text ranking tasks. Despite some recent efforts to address these issues, existing frameworks for LLM-based text embeddings have been limited by their support for only a limited range of LLM architectures and fine-tuning strategies, limiting their practical application and versatility. In this work, we introduce the Unified framework "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.03402","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-06T18:53:54Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"e1d95fe3d9ca522bdca4ba704c40d3a647e39990edf78f6caca8371a0671799f","abstract_canon_sha256":"d7bf0b4f2d030ffe2d35dbb02d5ac4d5a53dbf99cd3cb36dbdc16dc66cdb3435"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:52:58.125459Z","signature_b64":"vOLNiG/ixgGN4XfyNOfSiQZ9wq91erESpeiJUElmgLZ+c6c+qZxLDI7i17IjdrxjfIBPmnzZRpzvqOsj9+7OBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"498fd37962184d0df7d3afd952c18c166bad1bc296ff9f0719af4a43ab9ed6c1","last_reissued_at":"2026-07-05T08:52:58.125074Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:52:58.125074Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ULLME: A Unified Framework for Large Language Model Embeddings with Generation-Augmented Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Franck Dernoncourt, Hieu Man, Nghia Trung Ngo, Thien Huu Nguyen","submitted_at":"2024-08-06T18:53:54Z","abstract_excerpt":"Large Language Models (LLMs) excel in various natural language processing tasks, but leveraging them for dense passage embedding remains challenging. This is due to their causal attention mechanism and the misalignment between their pre-training objectives and the text ranking tasks. Despite some recent efforts to address these issues, existing frameworks for LLM-based text embeddings have been limited by their support for only a limited range of LLM architectures and fine-tuning strategies, limiting their practical application and versatility. In this work, we introduce the Unified framework "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.03402","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.03402/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.03402","created_at":"2026-07-05T08:52:58.125129+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.03402v1","created_at":"2026-07-05T08:52:58.125129+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.03402","created_at":"2026-07-05T08:52:58.125129+00:00"},{"alias_kind":"pith_short_12","alias_value":"JGH5G6LCDBGQ","created_at":"2026-07-05T08:52:58.125129+00:00"},{"alias_kind":"pith_short_16","alias_value":"JGH5G6LCDBGQ356T","created_at":"2026-07-05T08:52:58.125129+00:00"},{"alias_kind":"pith_short_8","alias_value":"JGH5G6LC","created_at":"2026-07-05T08:52:58.125129+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.08354","citing_title":"Position: Text Embeddings Should Capture Implicit Semantics, Not Just Surface Meaning","ref_index":60,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JGH5G6LCDBGQ356TV7MVFQMMCZ","json":"https://pith.science/pith/JGH5G6LCDBGQ356TV7MVFQMMCZ.json","graph_json":"https://pith.science/api/pith-number/JGH5G6LCDBGQ356TV7MVFQMMCZ/graph.json","events_json":"https://pith.science/api/pith-number/JGH5G6LCDBGQ356TV7MVFQMMCZ/events.json","paper":"https://pith.science/paper/JGH5G6LC"},"agent_actions":{"view_html":"https://pith.science/pith/JGH5G6LCDBGQ356TV7MVFQMMCZ","download_json":"https://pith.science/pith/JGH5G6LCDBGQ356TV7MVFQMMCZ.json","view_paper":"https://pith.science/paper/JGH5G6LC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.03402&json=true","fetch_graph":"https://pith.science/api/pith-number/JGH5G6LCDBGQ356TV7MVFQMMCZ/graph.json","fetch_events":"https://pith.science/api/pith-number/JGH5G6LCDBGQ356TV7MVFQMMCZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JGH5G6LCDBGQ356TV7MVFQMMCZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JGH5G6LCDBGQ356TV7MVFQMMCZ/action/storage_attestation","attest_author":"https://pith.science/pith/JGH5G6LCDBGQ356TV7MVFQMMCZ/action/author_attestation","sign_citation":"https://pith.science/pith/JGH5G6LCDBGQ356TV7MVFQMMCZ/action/citation_signature","submit_replication":"https://pith.science/pith/JGH5G6LCDBGQ356TV7MVFQMMCZ/action/replication_record"}},"created_at":"2026-07-05T08:52:58.125129+00:00","updated_at":"2026-07-05T08:52:58.125129+00:00"}