{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HKPTG2KQUQFRCTRHSAYJPKLACB","short_pith_number":"pith:HKPTG2KQ","schema_version":"1.0","canonical_sha256":"3a9f336950a40b114e27903097a9601071c3cefe7c20a4e34191e8c21b1fa18e","source":{"kind":"arxiv","id":"2409.07737","version":1},"attestation_state":"computed","paper":{"title":"Ruri: Japanese General Text Embeddings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hayato Tsukagoshi, Ryohei Sasano","submitted_at":"2024-09-12T04:06:31Z","abstract_excerpt":"We report the development of Ruri, a series of Japanese general text embedding models. While the development of general-purpose text embedding models in English and multilingual contexts has been active in recent years, model development in Japanese remains insufficient. The primary reasons for this are the lack of datasets and the absence of necessary expertise. In this report, we provide a detailed account of the development process of Ruri. Specifically, we discuss the training of embedding models using synthesized datasets generated by LLMs, the construction of the reranker for dataset fil"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.07737","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-12T04:06:31Z","cross_cats_sorted":[],"title_canon_sha256":"5eb295f6bdd90bde974a98a6648718d28e5fa896c984a3b2d2be88f43290792e","abstract_canon_sha256":"c960d72ec92ca550aa672af843bdcb8db48fafd5d0240c194807e8fda45aeb7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:06:14.345609Z","signature_b64":"G/Hpxss1QW6g/RSgzbxNN5svA/XgOV/yqLg7WkbtrBr/EEZYIyKKF/IayF0rkItdYShcEoQtAGENQgZAD5WyDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a9f336950a40b114e27903097a9601071c3cefe7c20a4e34191e8c21b1fa18e","last_reissued_at":"2026-07-05T09:06:14.345133Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:06:14.345133Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ruri: Japanese General Text Embeddings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hayato Tsukagoshi, Ryohei Sasano","submitted_at":"2024-09-12T04:06:31Z","abstract_excerpt":"We report the development of Ruri, a series of Japanese general text embedding models. While the development of general-purpose text embedding models in English and multilingual contexts has been active in recent years, model development in Japanese remains insufficient. The primary reasons for this are the lack of datasets and the absence of necessary expertise. In this report, we provide a detailed account of the development process of Ruri. Specifically, we discuss the training of embedding models using synthesized datasets generated by LLMs, the construction of the reranker for dataset fil"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.07737","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.07737/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.07737","created_at":"2026-07-05T09:06:14.345189+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.07737v1","created_at":"2026-07-05T09:06:14.345189+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.07737","created_at":"2026-07-05T09:06:14.345189+00:00"},{"alias_kind":"pith_short_12","alias_value":"HKPTG2KQUQFR","created_at":"2026-07-05T09:06:14.345189+00:00"},{"alias_kind":"pith_short_16","alias_value":"HKPTG2KQUQFRCTRH","created_at":"2026-07-05T09:06:14.345189+00:00"},{"alias_kind":"pith_short_8","alias_value":"HKPTG2KQ","created_at":"2026-07-05T09:06:14.345189+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22778","citing_title":"HAKARI-Bench: A Lightweight Benchmark for Comparing Retrieval Architectures and Efficiency Settings under Unified Conditions","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12114","citing_title":"Detecting Sensitive Personal Information in Japanese Pre-Training Corpora for Large Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15882","citing_title":"JFinTEB: Japanese Financial Text Embedding Benchmark","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HKPTG2KQUQFRCTRHSAYJPKLACB","json":"https://pith.science/pith/HKPTG2KQUQFRCTRHSAYJPKLACB.json","graph_json":"https://pith.science/api/pith-number/HKPTG2KQUQFRCTRHSAYJPKLACB/graph.json","events_json":"https://pith.science/api/pith-number/HKPTG2KQUQFRCTRHSAYJPKLACB/events.json","paper":"https://pith.science/paper/HKPTG2KQ"},"agent_actions":{"view_html":"https://pith.science/pith/HKPTG2KQUQFRCTRHSAYJPKLACB","download_json":"https://pith.science/pith/HKPTG2KQUQFRCTRHSAYJPKLACB.json","view_paper":"https://pith.science/paper/HKPTG2KQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.07737&json=true","fetch_graph":"https://pith.science/api/pith-number/HKPTG2KQUQFRCTRHSAYJPKLACB/graph.json","fetch_events":"https://pith.science/api/pith-number/HKPTG2KQUQFRCTRHSAYJPKLACB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HKPTG2KQUQFRCTRHSAYJPKLACB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HKPTG2KQUQFRCTRHSAYJPKLACB/action/storage_attestation","attest_author":"https://pith.science/pith/HKPTG2KQUQFRCTRHSAYJPKLACB/action/author_attestation","sign_citation":"https://pith.science/pith/HKPTG2KQUQFRCTRHSAYJPKLACB/action/citation_signature","submit_replication":"https://pith.science/pith/HKPTG2KQUQFRCTRHSAYJPKLACB/action/replication_record"}},"created_at":"2026-07-05T09:06:14.345189+00:00","updated_at":"2026-07-05T09:06:14.345189+00:00"}