{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HLURFTNVMPXVPUHVLI7QV3232H","short_pith_number":"pith:HLURFTNV","schema_version":"1.0","canonical_sha256":"3ae912cdb563ef57d0f55a3f0aef5bd1f1f53dad3872d27eb30ecb2100a43ca6","source":{"kind":"arxiv","id":"2502.10985","version":1},"attestation_state":"computed","paper":{"title":"Is Elo Rating Reliable? A Study Under Model Misspecification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ME","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chi Jin, Shange Tang, Yuanhao Wang","submitted_at":"2025-02-16T04:07:33Z","abstract_excerpt":"Elo rating, widely used for skill assessment across diverse domains ranging from competitive games to large language models, is often understood as an incremental update algorithm for estimating a stationary Bradley-Terry (BT) model. However, our empirical analysis of practical matching datasets reveals two surprising findings: (1) Most games deviate significantly from the assumptions of the BT model and stationarity, raising questions on the reliability of Elo. (2) Despite these deviations, Elo frequently outperforms more complex rating systems, such as mElo and pairwise models, which are spe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.10985","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-16T04:07:33Z","cross_cats_sorted":["cs.AI","stat.ME","stat.ML"],"title_canon_sha256":"d7af3541565486f85e9a38120f88553cda91d3bed32af978185b51c2401991ae","abstract_canon_sha256":"6e7db8104e5f0b0c3316e01fe5ac18f72ed41c0a7f78a2fd1b6c17a872b0c52d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:02.091191Z","signature_b64":"JlO8SQAFL5a8AmyjPyf1lKCyVRdI9PmZppiS60AP51Cw6agf+qb2+cM6zGnkduNTnv7eXpaR7u7GPvPF+ljeDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ae912cdb563ef57d0f55a3f0aef5bd1f1f53dad3872d27eb30ecb2100a43ca6","last_reissued_at":"2026-07-05T10:15:02.090720Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:02.090720Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is Elo Rating Reliable? A Study Under Model Misspecification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ME","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chi Jin, Shange Tang, Yuanhao Wang","submitted_at":"2025-02-16T04:07:33Z","abstract_excerpt":"Elo rating, widely used for skill assessment across diverse domains ranging from competitive games to large language models, is often understood as an incremental update algorithm for estimating a stationary Bradley-Terry (BT) model. However, our empirical analysis of practical matching datasets reveals two surprising findings: (1) Most games deviate significantly from the assumptions of the BT model and stationarity, raising questions on the reliability of Elo. (2) Despite these deviations, Elo frequently outperforms more complex rating systems, such as mElo and pairwise models, which are spe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.10985","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.10985/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.10985","created_at":"2026-07-05T10:15:02.090779+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.10985v1","created_at":"2026-07-05T10:15:02.090779+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.10985","created_at":"2026-07-05T10:15:02.090779+00:00"},{"alias_kind":"pith_short_12","alias_value":"HLURFTNVMPXV","created_at":"2026-07-05T10:15:02.090779+00:00"},{"alias_kind":"pith_short_16","alias_value":"HLURFTNVMPXVPUHV","created_at":"2026-07-05T10:15:02.090779+00:00"},{"alias_kind":"pith_short_8","alias_value":"HLURFTNV","created_at":"2026-07-05T10:15:02.090779+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.03637","citing_title":"RewardAnything: Generalizable Principle-Following Reward Models","ref_index":99,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HLURFTNVMPXVPUHVLI7QV3232H","json":"https://pith.science/pith/HLURFTNVMPXVPUHVLI7QV3232H.json","graph_json":"https://pith.science/api/pith-number/HLURFTNVMPXVPUHVLI7QV3232H/graph.json","events_json":"https://pith.science/api/pith-number/HLURFTNVMPXVPUHVLI7QV3232H/events.json","paper":"https://pith.science/paper/HLURFTNV"},"agent_actions":{"view_html":"https://pith.science/pith/HLURFTNVMPXVPUHVLI7QV3232H","download_json":"https://pith.science/pith/HLURFTNVMPXVPUHVLI7QV3232H.json","view_paper":"https://pith.science/paper/HLURFTNV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.10985&json=true","fetch_graph":"https://pith.science/api/pith-number/HLURFTNVMPXVPUHVLI7QV3232H/graph.json","fetch_events":"https://pith.science/api/pith-number/HLURFTNVMPXVPUHVLI7QV3232H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HLURFTNVMPXVPUHVLI7QV3232H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HLURFTNVMPXVPUHVLI7QV3232H/action/storage_attestation","attest_author":"https://pith.science/pith/HLURFTNVMPXVPUHVLI7QV3232H/action/author_attestation","sign_citation":"https://pith.science/pith/HLURFTNVMPXVPUHVLI7QV3232H/action/citation_signature","submit_replication":"https://pith.science/pith/HLURFTNVMPXVPUHVLI7QV3232H/action/replication_record"}},"created_at":"2026-07-05T10:15:02.090779+00:00","updated_at":"2026-07-05T10:15:02.090779+00:00"}