{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:FFHLKE6FYRKU52SQWQJBTE2DIV","short_pith_number":"pith:FFHLKE6F","schema_version":"1.0","canonical_sha256":"294eb513c5c4554eea50b41219934345469e912be2940eb38fc4de9380365931","source":{"kind":"arxiv","id":"2204.04779","version":2},"attestation_state":"computed","paper":{"title":"MedDistant19: Towards an Accurate Benchmark for Broad-Coverage Biomedical Relation Extraction","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"David Chang, G\\\"unter Neumann, Pasquale Minervini, Pontus Stenetorp, Saadullah Amin","submitted_at":"2022-04-10T22:07:25Z","abstract_excerpt":"Relation extraction in the biomedical domain is challenging due to the lack of labeled data and high annotation costs, needing domain experts. Distant supervision is commonly used to tackle the scarcity of annotated data by automatically pairing knowledge graph relationships with raw texts. Such a pipeline is prone to noise and has added challenges to scale for covering a large number of biomedical concepts. We investigated existing broad-coverage distantly supervised biomedical relation extraction benchmarks and found a significant overlap between training and test relationships ranging from "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.04779","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2022-04-10T22:07:25Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a80efc427b2bc2abbb03f00b4d4760a535909261bc843719cd95c8c158223dc2","abstract_canon_sha256":"fa7bb6fe05956f70ed54bc61c392174c5b7919ea68a646c535b9fe3a881fe1da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:57:07.357030Z","signature_b64":"jbLQaNGK9e81GuNbyCvCqCKLUaXseK3WDZR1ctpiHEQDY7rtHQQdscmpf1/L2zrPQXBAQE1FO55VMjCAEon+CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"294eb513c5c4554eea50b41219934345469e912be2940eb38fc4de9380365931","last_reissued_at":"2026-07-05T04:57:07.356520Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:57:07.356520Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MedDistant19: Towards an Accurate Benchmark for Broad-Coverage Biomedical Relation Extraction","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"David Chang, G\\\"unter Neumann, Pasquale Minervini, Pontus Stenetorp, Saadullah Amin","submitted_at":"2022-04-10T22:07:25Z","abstract_excerpt":"Relation extraction in the biomedical domain is challenging due to the lack of labeled data and high annotation costs, needing domain experts. Distant supervision is commonly used to tackle the scarcity of annotated data by automatically pairing knowledge graph relationships with raw texts. Such a pipeline is prone to noise and has added challenges to scale for covering a large number of biomedical concepts. We investigated existing broad-coverage distantly supervised biomedical relation extraction benchmarks and found a significant overlap between training and test relationships ranging from "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.04779","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.04779/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.04779","created_at":"2026-07-05T04:57:07.356588+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.04779v2","created_at":"2026-07-05T04:57:07.356588+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.04779","created_at":"2026-07-05T04:57:07.356588+00:00"},{"alias_kind":"pith_short_12","alias_value":"FFHLKE6FYRKU","created_at":"2026-07-05T04:57:07.356588+00:00"},{"alias_kind":"pith_short_16","alias_value":"FFHLKE6FYRKU52SQ","created_at":"2026-07-05T04:57:07.356588+00:00"},{"alias_kind":"pith_short_8","alias_value":"FFHLKE6F","created_at":"2026-07-05T04:57:07.356588+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FFHLKE6FYRKU52SQWQJBTE2DIV","json":"https://pith.science/pith/FFHLKE6FYRKU52SQWQJBTE2DIV.json","graph_json":"https://pith.science/api/pith-number/FFHLKE6FYRKU52SQWQJBTE2DIV/graph.json","events_json":"https://pith.science/api/pith-number/FFHLKE6FYRKU52SQWQJBTE2DIV/events.json","paper":"https://pith.science/paper/FFHLKE6F"},"agent_actions":{"view_html":"https://pith.science/pith/FFHLKE6FYRKU52SQWQJBTE2DIV","download_json":"https://pith.science/pith/FFHLKE6FYRKU52SQWQJBTE2DIV.json","view_paper":"https://pith.science/paper/FFHLKE6F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.04779&json=true","fetch_graph":"https://pith.science/api/pith-number/FFHLKE6FYRKU52SQWQJBTE2DIV/graph.json","fetch_events":"https://pith.science/api/pith-number/FFHLKE6FYRKU52SQWQJBTE2DIV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FFHLKE6FYRKU52SQWQJBTE2DIV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FFHLKE6FYRKU52SQWQJBTE2DIV/action/storage_attestation","attest_author":"https://pith.science/pith/FFHLKE6FYRKU52SQWQJBTE2DIV/action/author_attestation","sign_citation":"https://pith.science/pith/FFHLKE6FYRKU52SQWQJBTE2DIV/action/citation_signature","submit_replication":"https://pith.science/pith/FFHLKE6FYRKU52SQWQJBTE2DIV/action/replication_record"}},"created_at":"2026-07-05T04:57:07.356588+00:00","updated_at":"2026-07-05T04:57:07.356588+00:00"}