{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:FMQIMKBQUPHOREFCK4V56ERPMS","short_pith_number":"pith:FMQIMKBQ","schema_version":"1.0","canonical_sha256":"2b20862830a3cee890a2572bdf122f648e2350c678e6897ef6b69a4c59274026","source":{"kind":"arxiv","id":"2008.10813","version":1},"attestation_state":"computed","paper":{"title":"Conceptualized Representation Learning for Chinese Biomedical Text Mining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Feng Gao, Kangping Yin, Liang Dong, Nengwei Hua, Ningyu Zhang, Qianghuai Jia","submitted_at":"2020-08-25T04:41:35Z","abstract_excerpt":"Biomedical text mining is becoming increasingly important as the number of biomedical documents and web data rapidly grows. Recently, word representation models such as BERT has gained popularity among researchers. However, it is difficult to estimate their performance on datasets containing biomedical texts as the word distributions of general and biomedical corpora are quite different. Moreover, the medical domain has long-tail concepts and terminologies that are difficult to be learned via language models. For the Chinese biomedical text, it is more difficult due to its complex structure an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2008.10813","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-08-25T04:41:35Z","cross_cats_sorted":["cs.AI","cs.IR","cs.LG"],"title_canon_sha256":"74aa56e8e8c88f8823c0d465d6c22fb60404c0f8161d545e1bd18a4886b08671","abstract_canon_sha256":"d032ae505cbbba46646ce6dc4af2ffe61a0f90138cba0d519c2237fc93a36a34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:35:40.463460Z","signature_b64":"8WgLwKCelV/S1xBzwCEX924Bect4Dn16oD/XHVD4KOct5T3EisTBxEyjPjeDKAHwq2ow9rrzT5f2/uZzYV7iCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b20862830a3cee890a2572bdf122f648e2350c678e6897ef6b69a4c59274026","last_reissued_at":"2026-07-05T05:35:40.462936Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:35:40.462936Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Conceptualized Representation Learning for Chinese Biomedical Text Mining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Feng Gao, Kangping Yin, Liang Dong, Nengwei Hua, Ningyu Zhang, Qianghuai Jia","submitted_at":"2020-08-25T04:41:35Z","abstract_excerpt":"Biomedical text mining is becoming increasingly important as the number of biomedical documents and web data rapidly grows. Recently, word representation models such as BERT has gained popularity among researchers. However, it is difficult to estimate their performance on datasets containing biomedical texts as the word distributions of general and biomedical corpora are quite different. Moreover, the medical domain has long-tail concepts and terminologies that are difficult to be learned via language models. For the Chinese biomedical text, it is more difficult due to its complex structure an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2008.10813","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2008.10813/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2008.10813","created_at":"2026-07-05T05:35:40.462990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2008.10813v1","created_at":"2026-07-05T05:35:40.462990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2008.10813","created_at":"2026-07-05T05:35:40.462990+00:00"},{"alias_kind":"pith_short_12","alias_value":"FMQIMKBQUPHO","created_at":"2026-07-05T05:35:40.462990+00:00"},{"alias_kind":"pith_short_16","alias_value":"FMQIMKBQUPHOREFC","created_at":"2026-07-05T05:35:40.462990+00:00"},{"alias_kind":"pith_short_8","alias_value":"FMQIMKBQ","created_at":"2026-07-05T05:35:40.462990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.03387","citing_title":"LIMO: Less is More for Reasoning","ref_index":279,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FMQIMKBQUPHOREFCK4V56ERPMS","json":"https://pith.science/pith/FMQIMKBQUPHOREFCK4V56ERPMS.json","graph_json":"https://pith.science/api/pith-number/FMQIMKBQUPHOREFCK4V56ERPMS/graph.json","events_json":"https://pith.science/api/pith-number/FMQIMKBQUPHOREFCK4V56ERPMS/events.json","paper":"https://pith.science/paper/FMQIMKBQ"},"agent_actions":{"view_html":"https://pith.science/pith/FMQIMKBQUPHOREFCK4V56ERPMS","download_json":"https://pith.science/pith/FMQIMKBQUPHOREFCK4V56ERPMS.json","view_paper":"https://pith.science/paper/FMQIMKBQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2008.10813&json=true","fetch_graph":"https://pith.science/api/pith-number/FMQIMKBQUPHOREFCK4V56ERPMS/graph.json","fetch_events":"https://pith.science/api/pith-number/FMQIMKBQUPHOREFCK4V56ERPMS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FMQIMKBQUPHOREFCK4V56ERPMS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FMQIMKBQUPHOREFCK4V56ERPMS/action/storage_attestation","attest_author":"https://pith.science/pith/FMQIMKBQUPHOREFCK4V56ERPMS/action/author_attestation","sign_citation":"https://pith.science/pith/FMQIMKBQUPHOREFCK4V56ERPMS/action/citation_signature","submit_replication":"https://pith.science/pith/FMQIMKBQUPHOREFCK4V56ERPMS/action/replication_record"}},"created_at":"2026-07-05T05:35:40.462990+00:00","updated_at":"2026-07-05T05:35:40.462990+00:00"}