{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7RQFJ37QR2GXOX7CFPTOVP3FLC","short_pith_number":"pith:7RQFJ37Q","schema_version":"1.0","canonical_sha256":"fc6054eff08e8d775fe22be6eabf6558b9ee3bafb7d4e5cff391c61cff9b1f71","source":{"kind":"arxiv","id":"2305.03144","version":1},"attestation_state":"computed","paper":{"title":"Influence of various text embeddings on clustering performance in NLP","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.IR"],"primary_cat":"cs.LG","authors_text":"Rohan Saha","submitted_at":"2023-05-04T20:53:19Z","abstract_excerpt":"With the advent of e-commerce platforms, reviews are crucial for customers to assess the credibility of a product. The star ratings do not always match the review text written by the customer. For example, a three star rating (out of five) may be incongruous with the review text, which may be more suitable for a five star review. A clustering approach can be used to relabel the correct star ratings by grouping the text reviews into individual groups. In this work, we explore the task of choosing different text embeddings to represent these reviews and also explore the impact the embedding choi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.03144","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-05-04T20:53:19Z","cross_cats_sorted":["cs.CL","cs.IR"],"title_canon_sha256":"41bf207ba1e32044e997b9a61a7953d88f613a3fd5459a5bd4a6fcade1f3ed19","abstract_canon_sha256":"8dfd27b85a9f82249299691a0c82039f90b96e9e4bafac53c49da1a276a933e6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:07:11.425656Z","signature_b64":"CnzV7ggvauzL8PuRxEhoV1wQuHptDpjzicMnGwudiOrRgI0+eelpbYGziIDsRGU7LrlJ3NfakyskXFobiYmrDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fc6054eff08e8d775fe22be6eabf6558b9ee3bafb7d4e5cff391c61cff9b1f71","last_reissued_at":"2026-07-05T06:07:11.425305Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:07:11.425305Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Influence of various text embeddings on clustering performance in NLP","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.IR"],"primary_cat":"cs.LG","authors_text":"Rohan Saha","submitted_at":"2023-05-04T20:53:19Z","abstract_excerpt":"With the advent of e-commerce platforms, reviews are crucial for customers to assess the credibility of a product. The star ratings do not always match the review text written by the customer. For example, a three star rating (out of five) may be incongruous with the review text, which may be more suitable for a five star review. A clustering approach can be used to relabel the correct star ratings by grouping the text reviews into individual groups. In this work, we explore the task of choosing different text embeddings to represent these reviews and also explore the impact the embedding choi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.03144","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.03144/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.03144","created_at":"2026-07-05T06:07:11.425375+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.03144v1","created_at":"2026-07-05T06:07:11.425375+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.03144","created_at":"2026-07-05T06:07:11.425375+00:00"},{"alias_kind":"pith_short_12","alias_value":"7RQFJ37QR2GX","created_at":"2026-07-05T06:07:11.425375+00:00"},{"alias_kind":"pith_short_16","alias_value":"7RQFJ37QR2GXOX7C","created_at":"2026-07-05T06:07:11.425375+00:00"},{"alias_kind":"pith_short_8","alias_value":"7RQFJ37Q","created_at":"2026-07-05T06:07:11.425375+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.12269","citing_title":"A Cascaded Unsupervised-Supervised NLP Pipeline for Detecting Accusatory Language in Public Procurement","ref_index":35,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7RQFJ37QR2GXOX7CFPTOVP3FLC","json":"https://pith.science/pith/7RQFJ37QR2GXOX7CFPTOVP3FLC.json","graph_json":"https://pith.science/api/pith-number/7RQFJ37QR2GXOX7CFPTOVP3FLC/graph.json","events_json":"https://pith.science/api/pith-number/7RQFJ37QR2GXOX7CFPTOVP3FLC/events.json","paper":"https://pith.science/paper/7RQFJ37Q"},"agent_actions":{"view_html":"https://pith.science/pith/7RQFJ37QR2GXOX7CFPTOVP3FLC","download_json":"https://pith.science/pith/7RQFJ37QR2GXOX7CFPTOVP3FLC.json","view_paper":"https://pith.science/paper/7RQFJ37Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.03144&json=true","fetch_graph":"https://pith.science/api/pith-number/7RQFJ37QR2GXOX7CFPTOVP3FLC/graph.json","fetch_events":"https://pith.science/api/pith-number/7RQFJ37QR2GXOX7CFPTOVP3FLC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7RQFJ37QR2GXOX7CFPTOVP3FLC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7RQFJ37QR2GXOX7CFPTOVP3FLC/action/storage_attestation","attest_author":"https://pith.science/pith/7RQFJ37QR2GXOX7CFPTOVP3FLC/action/author_attestation","sign_citation":"https://pith.science/pith/7RQFJ37QR2GXOX7CFPTOVP3FLC/action/citation_signature","submit_replication":"https://pith.science/pith/7RQFJ37QR2GXOX7CFPTOVP3FLC/action/replication_record"}},"created_at":"2026-07-05T06:07:11.425375+00:00","updated_at":"2026-07-05T06:07:11.425375+00:00"}