{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UWGYM62TEXVRWWH6KT2CJX6N3H","short_pith_number":"pith:UWGYM62T","schema_version":"1.0","canonical_sha256":"a58d867b5325eb1b58fe54f424dfcdd9c4158211c706f42d7bb590722d31712b","source":{"kind":"arxiv","id":"2308.16744","version":1},"attestation_state":"computed","paper":{"title":"MS-BioGraphs: Sequence Similarity Graph Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AR","cs.CE","cs.DM","cs.PF"],"primary_cat":"cs.DC","authors_text":"Hans Vandierendonck, Mohsen Koohi Esfahani, Paolo Boldi, Peter Kilpatrick, Sebastiano Vigna","submitted_at":"2023-08-31T14:04:28Z","abstract_excerpt":"Progress in High-Performance Computing in general, and High-Performance Graph Processing in particular, is highly dependent on the availability of publicly-accessible, relevant, and realistic data sets.\n  To ensure continuation of this progress, we (i) investigate and optimize the process of generating large sequence similarity graphs as an HPC challenge and (ii) demonstrate this process in creating MS-BioGraphs, a new family of publicly available real-world edge-weighted graph datasets with up to $2.5$ trillion edges, that is, $6.6$ times greater than the largest graph published recently. The"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.16744","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2023-08-31T14:04:28Z","cross_cats_sorted":["cs.AR","cs.CE","cs.DM","cs.PF"],"title_canon_sha256":"b63df35ab56d9b381a13e3c22db4bc53e6c9c18e8842321c564815f645263888","abstract_canon_sha256":"9ecc10c78c70493858770e4f2b48d24b491cd1055eaeb66999b02f1c0e0eec93"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:46:38.639825Z","signature_b64":"CNxaynPLxT9xVtga8vDqTZAMgc1X+r6dCxJnMT29GN+8I2by/ax0uLrXjLDGsHUgmKtyv7OAwncUr4/f5fBEAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a58d867b5325eb1b58fe54f424dfcdd9c4158211c706f42d7bb590722d31712b","last_reissued_at":"2026-07-05T06:46:38.639285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:46:38.639285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MS-BioGraphs: Sequence Similarity Graph Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AR","cs.CE","cs.DM","cs.PF"],"primary_cat":"cs.DC","authors_text":"Hans Vandierendonck, Mohsen Koohi Esfahani, Paolo Boldi, Peter Kilpatrick, Sebastiano Vigna","submitted_at":"2023-08-31T14:04:28Z","abstract_excerpt":"Progress in High-Performance Computing in general, and High-Performance Graph Processing in particular, is highly dependent on the availability of publicly-accessible, relevant, and realistic data sets.\n  To ensure continuation of this progress, we (i) investigate and optimize the process of generating large sequence similarity graphs as an HPC challenge and (ii) demonstrate this process in creating MS-BioGraphs, a new family of publicly available real-world edge-weighted graph datasets with up to $2.5$ trillion edges, that is, $6.6$ times greater than the largest graph published recently. The"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.16744","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.16744/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.16744","created_at":"2026-07-05T06:46:38.639361+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.16744v1","created_at":"2026-07-05T06:46:38.639361+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.16744","created_at":"2026-07-05T06:46:38.639361+00:00"},{"alias_kind":"pith_short_12","alias_value":"UWGYM62TEXVR","created_at":"2026-07-05T06:46:38.639361+00:00"},{"alias_kind":"pith_short_16","alias_value":"UWGYM62TEXVRWWH6","created_at":"2026-07-05T06:46:38.639361+00:00"},{"alias_kind":"pith_short_8","alias_value":"UWGYM62T","created_at":"2026-07-05T06:46:38.639361+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.00716","citing_title":"Accelerating Loading WebGraphs in ParaGrapher","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UWGYM62TEXVRWWH6KT2CJX6N3H","json":"https://pith.science/pith/UWGYM62TEXVRWWH6KT2CJX6N3H.json","graph_json":"https://pith.science/api/pith-number/UWGYM62TEXVRWWH6KT2CJX6N3H/graph.json","events_json":"https://pith.science/api/pith-number/UWGYM62TEXVRWWH6KT2CJX6N3H/events.json","paper":"https://pith.science/paper/UWGYM62T"},"agent_actions":{"view_html":"https://pith.science/pith/UWGYM62TEXVRWWH6KT2CJX6N3H","download_json":"https://pith.science/pith/UWGYM62TEXVRWWH6KT2CJX6N3H.json","view_paper":"https://pith.science/paper/UWGYM62T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.16744&json=true","fetch_graph":"https://pith.science/api/pith-number/UWGYM62TEXVRWWH6KT2CJX6N3H/graph.json","fetch_events":"https://pith.science/api/pith-number/UWGYM62TEXVRWWH6KT2CJX6N3H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UWGYM62TEXVRWWH6KT2CJX6N3H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UWGYM62TEXVRWWH6KT2CJX6N3H/action/storage_attestation","attest_author":"https://pith.science/pith/UWGYM62TEXVRWWH6KT2CJX6N3H/action/author_attestation","sign_citation":"https://pith.science/pith/UWGYM62TEXVRWWH6KT2CJX6N3H/action/citation_signature","submit_replication":"https://pith.science/pith/UWGYM62TEXVRWWH6KT2CJX6N3H/action/replication_record"}},"created_at":"2026-07-05T06:46:38.639361+00:00","updated_at":"2026-07-05T06:46:38.639361+00:00"}