{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:L7N5YNZPBMXHUIOKEFVAJEY5XG","short_pith_number":"pith:L7N5YNZP","schema_version":"1.0","canonical_sha256":"5fdbdc372f0b2e7a21ca216a04931db98b7754115fbf139e4d68214cba95d53c","source":{"kind":"arxiv","id":"2411.13245","version":2},"attestation_state":"computed","paper":{"title":"[Experiments \\& Analysis] Hash-Based vs. Sort-Based Group-By-Aggregate: A Focused Empirical Study [Extended Version]","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Gaurav Vaghasiya, Shiva Jahangiri","submitted_at":"2024-11-20T12:03:50Z","abstract_excerpt":"Group-by-aggregate (GBA) queries are integral to data analysis, allowing users to group data by specific attributes and apply aggregate functions such as sum, average, and count. Database Management Systems (DBMSs) typically execute GBA queries using either sort- or hash-based methods, each with unique advantages and trade-offs. Sort-based approaches are efficient for large datasets but become computationally expensive due to record comparisons, especially in cases with a small number of groups. In contrast, hash-based approaches offer faster performance in general but require significant memo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.13245","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DB","submitted_at":"2024-11-20T12:03:50Z","cross_cats_sorted":[],"title_canon_sha256":"6847503881ddd3c4e4e5a5f34d55fa95171bdcbecce5113b1ecc8320c5829897","abstract_canon_sha256":"fa44153f502423962f0f8b0d47ca2743e13eb2dcad7e22df5dbbf0c610366d7c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:42:33.007508Z","signature_b64":"1jo51mGp2QbedWtV0w20UZsqp2Zg04Li2PZbw0p1ZTx3bfFOFvWIAS8JxfzdVXSTxK3RCc+kKPc7iSnblXHYCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5fdbdc372f0b2e7a21ca216a04931db98b7754115fbf139e4d68214cba95d53c","last_reissued_at":"2026-07-05T09:42:33.006967Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:42:33.006967Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"[Experiments \\& Analysis] Hash-Based vs. Sort-Based Group-By-Aggregate: A Focused Empirical Study [Extended Version]","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DB","authors_text":"Gaurav Vaghasiya, Shiva Jahangiri","submitted_at":"2024-11-20T12:03:50Z","abstract_excerpt":"Group-by-aggregate (GBA) queries are integral to data analysis, allowing users to group data by specific attributes and apply aggregate functions such as sum, average, and count. Database Management Systems (DBMSs) typically execute GBA queries using either sort- or hash-based methods, each with unique advantages and trade-offs. Sort-based approaches are efficient for large datasets but become computationally expensive due to record comparisons, especially in cases with a small number of groups. In contrast, hash-based approaches offer faster performance in general but require significant memo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.13245","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.13245/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.13245","created_at":"2026-07-05T09:42:33.007029+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.13245v2","created_at":"2026-07-05T09:42:33.007029+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.13245","created_at":"2026-07-05T09:42:33.007029+00:00"},{"alias_kind":"pith_short_12","alias_value":"L7N5YNZPBMXH","created_at":"2026-07-05T09:42:33.007029+00:00"},{"alias_kind":"pith_short_16","alias_value":"L7N5YNZPBMXHUIOK","created_at":"2026-07-05T09:42:33.007029+00:00"},{"alias_kind":"pith_short_8","alias_value":"L7N5YNZP","created_at":"2026-07-05T09:42:33.007029+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.04153","citing_title":"Global Hash Tables Strike Back! An Analysis of Parallel GROUP BY Aggregation","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L7N5YNZPBMXHUIOKEFVAJEY5XG","json":"https://pith.science/pith/L7N5YNZPBMXHUIOKEFVAJEY5XG.json","graph_json":"https://pith.science/api/pith-number/L7N5YNZPBMXHUIOKEFVAJEY5XG/graph.json","events_json":"https://pith.science/api/pith-number/L7N5YNZPBMXHUIOKEFVAJEY5XG/events.json","paper":"https://pith.science/paper/L7N5YNZP"},"agent_actions":{"view_html":"https://pith.science/pith/L7N5YNZPBMXHUIOKEFVAJEY5XG","download_json":"https://pith.science/pith/L7N5YNZPBMXHUIOKEFVAJEY5XG.json","view_paper":"https://pith.science/paper/L7N5YNZP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.13245&json=true","fetch_graph":"https://pith.science/api/pith-number/L7N5YNZPBMXHUIOKEFVAJEY5XG/graph.json","fetch_events":"https://pith.science/api/pith-number/L7N5YNZPBMXHUIOKEFVAJEY5XG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L7N5YNZPBMXHUIOKEFVAJEY5XG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L7N5YNZPBMXHUIOKEFVAJEY5XG/action/storage_attestation","attest_author":"https://pith.science/pith/L7N5YNZPBMXHUIOKEFVAJEY5XG/action/author_attestation","sign_citation":"https://pith.science/pith/L7N5YNZPBMXHUIOKEFVAJEY5XG/action/citation_signature","submit_replication":"https://pith.science/pith/L7N5YNZPBMXHUIOKEFVAJEY5XG/action/replication_record"}},"created_at":"2026-07-05T09:42:33.007029+00:00","updated_at":"2026-07-05T09:42:33.007029+00:00"}