{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OCVSUFPTAFJ2ILGTXW4RHSZDZG","short_pith_number":"pith:OCVSUFPT","schema_version":"1.0","canonical_sha256":"70ab2a15f30153a42cd3bdb913cb23c9bd2058b636f05ab9fca6924eaeac6348","source":{"kind":"arxiv","id":"2412.11068","version":1},"attestation_state":"computed","paper":{"title":"RecSys Arena: Pair-wise Recommender System Evaluation with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.IR","authors_text":"Chuhan Wu, Qinglin Jia, Shuai Wang, Zan Wang, Zhaocheng Du, Zhenhua Dong, Zhuo Wu","submitted_at":"2024-12-15T05:57:36Z","abstract_excerpt":"Evaluating the quality of recommender systems is critical for algorithm design and optimization. Most evaluation methods are computed based on offline metrics for quick algorithm evolution, since online experiments are usually risky and time-consuming. However, offline evaluation usually cannot fully reflect users' preference for the outcome of different recommendation algorithms, and the results may not be consistent with online A/B test. Moreover, many offline metrics such as AUC do not offer sufficient information for comparing the subtle differences between two competitive recommender syst"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.11068","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2024-12-15T05:57:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0e0cda64f0cf53ebeb4560f809f9010c1d6c48cfbf5d22297a495bdd061cf787","abstract_canon_sha256":"b2f2ff2ec4f75098a80f962be08b7778563c7e6a214bd796b16c0bcebd7a89f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:26.817207Z","signature_b64":"rcinw48LTWfIp++7Me53Cd+a4I1ovnHV0amYPTKsH+WaHJepC7h2rvH07qTHq5Fa4vENgiy7QrNfEZInVcvMBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"70ab2a15f30153a42cd3bdb913cb23c9bd2058b636f05ab9fca6924eaeac6348","last_reissued_at":"2026-07-05T09:49:26.816772Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:26.816772Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RecSys Arena: Pair-wise Recommender System Evaluation with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.IR","authors_text":"Chuhan Wu, Qinglin Jia, Shuai Wang, Zan Wang, Zhaocheng Du, Zhenhua Dong, Zhuo Wu","submitted_at":"2024-12-15T05:57:36Z","abstract_excerpt":"Evaluating the quality of recommender systems is critical for algorithm design and optimization. Most evaluation methods are computed based on offline metrics for quick algorithm evolution, since online experiments are usually risky and time-consuming. However, offline evaluation usually cannot fully reflect users' preference for the outcome of different recommendation algorithms, and the results may not be consistent with online A/B test. Moreover, many offline metrics such as AUC do not offer sufficient information for comparing the subtle differences between two competitive recommender syst"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.11068","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.11068/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.11068","created_at":"2026-07-05T09:49:26.816825+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.11068v1","created_at":"2026-07-05T09:49:26.816825+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.11068","created_at":"2026-07-05T09:49:26.816825+00:00"},{"alias_kind":"pith_short_12","alias_value":"OCVSUFPTAFJ2","created_at":"2026-07-05T09:49:26.816825+00:00"},{"alias_kind":"pith_short_16","alias_value":"OCVSUFPTAFJ2ILGT","created_at":"2026-07-05T09:49:26.816825+00:00"},{"alias_kind":"pith_short_8","alias_value":"OCVSUFPT","created_at":"2026-07-05T09:49:26.816825+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22961","citing_title":"LLM-as-a-Judge for Reliable and Explainable Offline Evaluation in Top-K Recommendation","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OCVSUFPTAFJ2ILGTXW4RHSZDZG","json":"https://pith.science/pith/OCVSUFPTAFJ2ILGTXW4RHSZDZG.json","graph_json":"https://pith.science/api/pith-number/OCVSUFPTAFJ2ILGTXW4RHSZDZG/graph.json","events_json":"https://pith.science/api/pith-number/OCVSUFPTAFJ2ILGTXW4RHSZDZG/events.json","paper":"https://pith.science/paper/OCVSUFPT"},"agent_actions":{"view_html":"https://pith.science/pith/OCVSUFPTAFJ2ILGTXW4RHSZDZG","download_json":"https://pith.science/pith/OCVSUFPTAFJ2ILGTXW4RHSZDZG.json","view_paper":"https://pith.science/paper/OCVSUFPT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.11068&json=true","fetch_graph":"https://pith.science/api/pith-number/OCVSUFPTAFJ2ILGTXW4RHSZDZG/graph.json","fetch_events":"https://pith.science/api/pith-number/OCVSUFPTAFJ2ILGTXW4RHSZDZG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OCVSUFPTAFJ2ILGTXW4RHSZDZG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OCVSUFPTAFJ2ILGTXW4RHSZDZG/action/storage_attestation","attest_author":"https://pith.science/pith/OCVSUFPTAFJ2ILGTXW4RHSZDZG/action/author_attestation","sign_citation":"https://pith.science/pith/OCVSUFPTAFJ2ILGTXW4RHSZDZG/action/citation_signature","submit_replication":"https://pith.science/pith/OCVSUFPTAFJ2ILGTXW4RHSZDZG/action/replication_record"}},"created_at":"2026-07-05T09:49:26.816825+00:00","updated_at":"2026-07-05T09:49:26.816825+00:00"}