{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:3FFRNQNCIRUUIWWC7O7KRHQXXN","short_pith_number":"pith:3FFRNQNC","schema_version":"1.0","canonical_sha256":"d94b16c1a24469445ac2fbbea89e17bb417ce712fcb8168c7f8dbd8ac300441f","source":{"kind":"arxiv","id":"1910.00637","version":2},"attestation_state":"computed","paper":{"title":"Essentia: Mining Domain-Specific Paraphrases with Word-Alignment Graphs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Behzad Golshan, Chen Chen, Danni Ma, Wang-Chiew Tan","submitted_at":"2019-10-01T19:51:57Z","abstract_excerpt":"Paraphrases are important linguistic resources for a wide variety of NLP applications. Many techniques for automatic paraphrase mining from general corpora have been proposed. While these techniques are successful at discovering generic paraphrases, they often fail to identify domain-specific paraphrases (e.g., {staff, concierge} in the hospitality domain). This is because current techniques are often based on statistical methods, while domain-specific corpora are too small to fit statistical methods. In this paper, we present an unsupervised graph-based technique to mine paraphrases from a sm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.00637","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-10-01T19:51:57Z","cross_cats_sorted":[],"title_canon_sha256":"98941dc06713be609aea9e02cb5a363ec2ba81939619b73ed1068da2e705a084","abstract_canon_sha256":"d47973874bf08632659c0f06e537ee68e0156239dba5d3d621a3b9a39173330c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:10:04.947494Z","signature_b64":"wQ0pSkV3Bvg8eQRdx62l4TdctSQUr83YND8aNjePlB2dWgORZC9JA+Bf2l4tUW8AqbyAKo8M2yGJei5ISVmVBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d94b16c1a24469445ac2fbbea89e17bb417ce712fcb8168c7f8dbd8ac300441f","last_reissued_at":"2026-07-05T00:10:04.947098Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:10:04.947098Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Essentia: Mining Domain-Specific Paraphrases with Word-Alignment Graphs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Behzad Golshan, Chen Chen, Danni Ma, Wang-Chiew Tan","submitted_at":"2019-10-01T19:51:57Z","abstract_excerpt":"Paraphrases are important linguistic resources for a wide variety of NLP applications. Many techniques for automatic paraphrase mining from general corpora have been proposed. While these techniques are successful at discovering generic paraphrases, they often fail to identify domain-specific paraphrases (e.g., {staff, concierge} in the hospitality domain). This is because current techniques are often based on statistical methods, while domain-specific corpora are too small to fit statistical methods. In this paper, we present an unsupervised graph-based technique to mine paraphrases from a sm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.00637","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.00637/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.00637","created_at":"2026-07-05T00:10:04.947170+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.00637v2","created_at":"2026-07-05T00:10:04.947170+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.00637","created_at":"2026-07-05T00:10:04.947170+00:00"},{"alias_kind":"pith_short_12","alias_value":"3FFRNQNCIRUU","created_at":"2026-07-05T00:10:04.947170+00:00"},{"alias_kind":"pith_short_16","alias_value":"3FFRNQNCIRUUIWWC","created_at":"2026-07-05T00:10:04.947170+00:00"},{"alias_kind":"pith_short_8","alias_value":"3FFRNQNC","created_at":"2026-07-05T00:10:04.947170+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3FFRNQNCIRUUIWWC7O7KRHQXXN","json":"https://pith.science/pith/3FFRNQNCIRUUIWWC7O7KRHQXXN.json","graph_json":"https://pith.science/api/pith-number/3FFRNQNCIRUUIWWC7O7KRHQXXN/graph.json","events_json":"https://pith.science/api/pith-number/3FFRNQNCIRUUIWWC7O7KRHQXXN/events.json","paper":"https://pith.science/paper/3FFRNQNC"},"agent_actions":{"view_html":"https://pith.science/pith/3FFRNQNCIRUUIWWC7O7KRHQXXN","download_json":"https://pith.science/pith/3FFRNQNCIRUUIWWC7O7KRHQXXN.json","view_paper":"https://pith.science/paper/3FFRNQNC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.00637&json=true","fetch_graph":"https://pith.science/api/pith-number/3FFRNQNCIRUUIWWC7O7KRHQXXN/graph.json","fetch_events":"https://pith.science/api/pith-number/3FFRNQNCIRUUIWWC7O7KRHQXXN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3FFRNQNCIRUUIWWC7O7KRHQXXN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3FFRNQNCIRUUIWWC7O7KRHQXXN/action/storage_attestation","attest_author":"https://pith.science/pith/3FFRNQNCIRUUIWWC7O7KRHQXXN/action/author_attestation","sign_citation":"https://pith.science/pith/3FFRNQNCIRUUIWWC7O7KRHQXXN/action/citation_signature","submit_replication":"https://pith.science/pith/3FFRNQNCIRUUIWWC7O7KRHQXXN/action/replication_record"}},"created_at":"2026-07-05T00:10:04.947170+00:00","updated_at":"2026-07-05T00:10:04.947170+00:00"}