{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ALNHBDBD7QT3BH3EQBWUVB3YCR","short_pith_number":"pith:ALNHBDBD","schema_version":"1.0","canonical_sha256":"02da708c23fc27b09f64806d4a8778144d1955deb403522dc29a0a142ad8de15","source":{"kind":"arxiv","id":"2502.16766","version":2},"attestation_state":"computed","paper":{"title":"ATEB: Evaluating and Improving Advanced NLP Tasks for Text Embedding Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arman Cohan, Chris Tar, Daniel Cer, Frank Palma Gomez, Gustavo Hernandez Abrego, Hansi Zeng, Simeng Han, Tu Vu, Zefei Li","submitted_at":"2025-02-24T01:08:15Z","abstract_excerpt":"Traditional text embedding benchmarks primarily evaluate embedding models' capabilities to capture semantic similarity. However, more advanced NLP tasks require a deeper understanding of text, such as safety and factuality. These tasks demand an ability to comprehend and process complex information, often involving the handling of sensitive content, or the verification of factual statements against reliable sources. We introduce a new benchmark designed to assess and highlight the limitations of embedding models trained on existing information retrieval data mixtures on advanced capabilities, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.16766","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-24T01:08:15Z","cross_cats_sorted":[],"title_canon_sha256":"33c89a7f72192e982b193d379ccf7001d4a587327354fe020491d85cbaa8c080","abstract_canon_sha256":"ce4fc8cd410162290bce911fad8bc79a1f4822d2b1471002e3eb6980f82ab1a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:29.838868Z","signature_b64":"3ZRyYH6asbtIHObBOn2JFXiK9lQwfcSqrNJdrCbdshsp+JCVknxtorND0UgFnNtE58lrxYdCeEGwKmUFeN+cDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"02da708c23fc27b09f64806d4a8778144d1955deb403522dc29a0a142ad8de15","last_reissued_at":"2026-07-05T10:23:29.837916Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:29.837916Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ATEB: Evaluating and Improving Advanced NLP Tasks for Text Embedding Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arman Cohan, Chris Tar, Daniel Cer, Frank Palma Gomez, Gustavo Hernandez Abrego, Hansi Zeng, Simeng Han, Tu Vu, Zefei Li","submitted_at":"2025-02-24T01:08:15Z","abstract_excerpt":"Traditional text embedding benchmarks primarily evaluate embedding models' capabilities to capture semantic similarity. However, more advanced NLP tasks require a deeper understanding of text, such as safety and factuality. These tasks demand an ability to comprehend and process complex information, often involving the handling of sensitive content, or the verification of factual statements against reliable sources. We introduce a new benchmark designed to assess and highlight the limitations of embedding models trained on existing information retrieval data mixtures on advanced capabilities, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16766","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.16766/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.16766","created_at":"2026-07-05T10:23:29.838024+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.16766v2","created_at":"2026-07-05T10:23:29.838024+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16766","created_at":"2026-07-05T10:23:29.838024+00:00"},{"alias_kind":"pith_short_12","alias_value":"ALNHBDBD7QT3","created_at":"2026-07-05T10:23:29.838024+00:00"},{"alias_kind":"pith_short_16","alias_value":"ALNHBDBD7QT3BH3E","created_at":"2026-07-05T10:23:29.838024+00:00"},{"alias_kind":"pith_short_8","alias_value":"ALNHBDBD","created_at":"2026-07-05T10:23:29.838024+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.04638","citing_title":"Advancing AI Research Assistants with Expert-Involved Learning","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ALNHBDBD7QT3BH3EQBWUVB3YCR","json":"https://pith.science/pith/ALNHBDBD7QT3BH3EQBWUVB3YCR.json","graph_json":"https://pith.science/api/pith-number/ALNHBDBD7QT3BH3EQBWUVB3YCR/graph.json","events_json":"https://pith.science/api/pith-number/ALNHBDBD7QT3BH3EQBWUVB3YCR/events.json","paper":"https://pith.science/paper/ALNHBDBD"},"agent_actions":{"view_html":"https://pith.science/pith/ALNHBDBD7QT3BH3EQBWUVB3YCR","download_json":"https://pith.science/pith/ALNHBDBD7QT3BH3EQBWUVB3YCR.json","view_paper":"https://pith.science/paper/ALNHBDBD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.16766&json=true","fetch_graph":"https://pith.science/api/pith-number/ALNHBDBD7QT3BH3EQBWUVB3YCR/graph.json","fetch_events":"https://pith.science/api/pith-number/ALNHBDBD7QT3BH3EQBWUVB3YCR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ALNHBDBD7QT3BH3EQBWUVB3YCR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ALNHBDBD7QT3BH3EQBWUVB3YCR/action/storage_attestation","attest_author":"https://pith.science/pith/ALNHBDBD7QT3BH3EQBWUVB3YCR/action/author_attestation","sign_citation":"https://pith.science/pith/ALNHBDBD7QT3BH3EQBWUVB3YCR/action/citation_signature","submit_replication":"https://pith.science/pith/ALNHBDBD7QT3BH3EQBWUVB3YCR/action/replication_record"}},"created_at":"2026-07-05T10:23:29.838024+00:00","updated_at":"2026-07-05T10:23:29.838024+00:00"}