{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YTRU5AI6J7NQJQW4C23U5547JW","short_pith_number":"pith:YTRU5AI6","schema_version":"1.0","canonical_sha256":"c4e34e811e4fdb04c2dc16b74ef79f4daad4725a01185635a471b493037dfd79","source":{"kind":"arxiv","id":"2402.09766","version":2},"attestation_state":"computed","paper":{"title":"From Variability to Stability: Advancing RecSys Benchmarking Practices","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.IR","authors_text":"Alexey Vasilev, Alexey Zaytsev, Andrey Savchenko, Anna Volodkevich, Artyom Sosedka, Natalia Semenova, Nikita Belousov, Valeriy Shevchenko, Vladimir Zholobov","submitted_at":"2024-02-15T07:35:52Z","abstract_excerpt":"In the rapidly evolving domain of Recommender Systems (RecSys), new algorithms frequently claim state-of-the-art performance based on evaluations over a limited set of arbitrarily selected datasets. However, this approach may fail to holistically reflect their effectiveness due to the significant impact of dataset characteristics on algorithm performance. Addressing this deficiency, this paper introduces a novel benchmarking methodology to facilitate a fair and robust comparison of RecSys algorithms, thereby advancing evaluation practices. By utilizing a diverse set of $30$ open datasets, incl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.09766","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2024-02-15T07:35:52Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"cd3d261a4fed61dc78f5acf570cefeceeca5c9f7a597c8868e6650cf0c37d834","abstract_canon_sha256":"68b2dc8446890ff3bd8ecb8a95f66b82b973cbb2993d3b47204f9b86b52ea173"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:59:26.002242Z","signature_b64":"fxn1OJFMVUJ4WLobtPGcwaJjp1l+1vJ+p9L11qUSLhl26LpzVW1jPR7h1Y7dzE/DPnRUJOHUmEZsMPA/ioDnAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4e34e811e4fdb04c2dc16b74ef79f4daad4725a01185635a471b493037dfd79","last_reissued_at":"2026-07-05T08:59:26.001716Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:59:26.001716Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Variability to Stability: Advancing RecSys Benchmarking Practices","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.IR","authors_text":"Alexey Vasilev, Alexey Zaytsev, Andrey Savchenko, Anna Volodkevich, Artyom Sosedka, Natalia Semenova, Nikita Belousov, Valeriy Shevchenko, Vladimir Zholobov","submitted_at":"2024-02-15T07:35:52Z","abstract_excerpt":"In the rapidly evolving domain of Recommender Systems (RecSys), new algorithms frequently claim state-of-the-art performance based on evaluations over a limited set of arbitrarily selected datasets. However, this approach may fail to holistically reflect their effectiveness due to the significant impact of dataset characteristics on algorithm performance. Addressing this deficiency, this paper introduces a novel benchmarking methodology to facilitate a fair and robust comparison of RecSys algorithms, thereby advancing evaluation practices. By utilizing a diverse set of $30$ open datasets, incl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.09766","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.09766/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.09766","created_at":"2026-07-05T08:59:26.001787+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.09766v2","created_at":"2026-07-05T08:59:26.001787+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.09766","created_at":"2026-07-05T08:59:26.001787+00:00"},{"alias_kind":"pith_short_12","alias_value":"YTRU5AI6J7NQ","created_at":"2026-07-05T08:59:26.001787+00:00"},{"alias_kind":"pith_short_16","alias_value":"YTRU5AI6J7NQJQW4","created_at":"2026-07-05T08:59:26.001787+00:00"},{"alias_kind":"pith_short_8","alias_value":"YTRU5AI6","created_at":"2026-07-05T08:59:26.001787+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.14302","citing_title":"SAFERec: Self-Attention and Frequency Enriched Model for Next Basket Recommendation","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YTRU5AI6J7NQJQW4C23U5547JW","json":"https://pith.science/pith/YTRU5AI6J7NQJQW4C23U5547JW.json","graph_json":"https://pith.science/api/pith-number/YTRU5AI6J7NQJQW4C23U5547JW/graph.json","events_json":"https://pith.science/api/pith-number/YTRU5AI6J7NQJQW4C23U5547JW/events.json","paper":"https://pith.science/paper/YTRU5AI6"},"agent_actions":{"view_html":"https://pith.science/pith/YTRU5AI6J7NQJQW4C23U5547JW","download_json":"https://pith.science/pith/YTRU5AI6J7NQJQW4C23U5547JW.json","view_paper":"https://pith.science/paper/YTRU5AI6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.09766&json=true","fetch_graph":"https://pith.science/api/pith-number/YTRU5AI6J7NQJQW4C23U5547JW/graph.json","fetch_events":"https://pith.science/api/pith-number/YTRU5AI6J7NQJQW4C23U5547JW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YTRU5AI6J7NQJQW4C23U5547JW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YTRU5AI6J7NQJQW4C23U5547JW/action/storage_attestation","attest_author":"https://pith.science/pith/YTRU5AI6J7NQJQW4C23U5547JW/action/author_attestation","sign_citation":"https://pith.science/pith/YTRU5AI6J7NQJQW4C23U5547JW/action/citation_signature","submit_replication":"https://pith.science/pith/YTRU5AI6J7NQJQW4C23U5547JW/action/replication_record"}},"created_at":"2026-07-05T08:59:26.001787+00:00","updated_at":"2026-07-05T08:59:26.001787+00:00"}