{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZO3RRKG7WFFTOOIZGDYKJCFMUM","short_pith_number":"pith:ZO3RRKG7","schema_version":"1.0","canonical_sha256":"cbb718a8dfb14b37391930f0a488aca303b531216368184e8f1bd09de8aefe7f","source":{"kind":"arxiv","id":"2312.03121","version":4},"attestation_state":"computed","paper":{"title":"Evaluating Agents using Social Choice Theory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.GT","cs.MA"],"primary_cat":"cs.AI","authors_text":"Anna Koop, Avishkar Bhoopchand, Brian Tanner, Kate Larson, Luke Marris, Marc Lanctot, Thomas Anthony, Yoram Bachrach, Zun Li","submitted_at":"2023-12-05T20:40:37Z","abstract_excerpt":"We argue that many general evaluation problems can be viewed through the lens of voting theory. Each task is interpreted as a separate voter, which requires only ordinal rankings or pairwise comparisons of agents to produce an overall evaluation. By viewing the aggregator as a social welfare function, we are able to leverage centuries of research in social choice theory to derive principled evaluation frameworks with axiomatic foundations. These evaluations are interpretable and flexible, while avoiding many of the problems currently facing cross-task evaluation. We apply this Voting-as-Evalua"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.03121","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-12-05T20:40:37Z","cross_cats_sorted":["cs.GT","cs.MA"],"title_canon_sha256":"1a53074d155be443683a0a54b7fc8170ae3fe62f0527e1d081513c56f246fcc8","abstract_canon_sha256":"b88f841f5dd8b652d3cdc8619a9935f6894def8c5843bad67ce68604af57ffeb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:29.239060Z","signature_b64":"EB+1UqRY0dSc4P69ixpbX3L9N85JVy9LuQvyeB3T3tTXig1l+QutQ9o2WtICoaagmrc2sZfzu4srh2UZrmE+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cbb718a8dfb14b37391930f0a488aca303b531216368184e8f1bd09de8aefe7f","last_reissued_at":"2026-07-05T11:28:29.238582Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:29.238582Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Agents using Social Choice Theory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.GT","cs.MA"],"primary_cat":"cs.AI","authors_text":"Anna Koop, Avishkar Bhoopchand, Brian Tanner, Kate Larson, Luke Marris, Marc Lanctot, Thomas Anthony, Yoram Bachrach, Zun Li","submitted_at":"2023-12-05T20:40:37Z","abstract_excerpt":"We argue that many general evaluation problems can be viewed through the lens of voting theory. Each task is interpreted as a separate voter, which requires only ordinal rankings or pairwise comparisons of agents to produce an overall evaluation. By viewing the aggregator as a social welfare function, we are able to leverage centuries of research in social choice theory to derive principled evaluation frameworks with axiomatic foundations. These evaluations are interpretable and flexible, while avoiding many of the problems currently facing cross-task evaluation. We apply this Voting-as-Evalua"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.03121","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.03121/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.03121","created_at":"2026-07-05T11:28:29.238639+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.03121v4","created_at":"2026-07-05T11:28:29.238639+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.03121","created_at":"2026-07-05T11:28:29.238639+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZO3RRKG7WFFT","created_at":"2026-07-05T11:28:29.238639+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZO3RRKG7WFFTOOIZ","created_at":"2026-07-05T11:28:29.238639+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZO3RRKG7","created_at":"2026-07-05T11:28:29.238639+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21550","citing_title":"AI Alignment From Social Choice Perspectives","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23628","citing_title":"How Hard is it to Rig a Benchmark? A Social Choice Analysis of Leaderboard Robustness","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21769","citing_title":"Who Defines \"Best\"? Towards Interactive, User-Defined Evaluation of LLM Leaderboards","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07996","citing_title":"Nash without Numbers: A Social Choice Approach to Mixed Equilibria in Context-Ordinal Games","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZO3RRKG7WFFTOOIZGDYKJCFMUM","json":"https://pith.science/pith/ZO3RRKG7WFFTOOIZGDYKJCFMUM.json","graph_json":"https://pith.science/api/pith-number/ZO3RRKG7WFFTOOIZGDYKJCFMUM/graph.json","events_json":"https://pith.science/api/pith-number/ZO3RRKG7WFFTOOIZGDYKJCFMUM/events.json","paper":"https://pith.science/paper/ZO3RRKG7"},"agent_actions":{"view_html":"https://pith.science/pith/ZO3RRKG7WFFTOOIZGDYKJCFMUM","download_json":"https://pith.science/pith/ZO3RRKG7WFFTOOIZGDYKJCFMUM.json","view_paper":"https://pith.science/paper/ZO3RRKG7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.03121&json=true","fetch_graph":"https://pith.science/api/pith-number/ZO3RRKG7WFFTOOIZGDYKJCFMUM/graph.json","fetch_events":"https://pith.science/api/pith-number/ZO3RRKG7WFFTOOIZGDYKJCFMUM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZO3RRKG7WFFTOOIZGDYKJCFMUM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZO3RRKG7WFFTOOIZGDYKJCFMUM/action/storage_attestation","attest_author":"https://pith.science/pith/ZO3RRKG7WFFTOOIZGDYKJCFMUM/action/author_attestation","sign_citation":"https://pith.science/pith/ZO3RRKG7WFFTOOIZGDYKJCFMUM/action/citation_signature","submit_replication":"https://pith.science/pith/ZO3RRKG7WFFTOOIZGDYKJCFMUM/action/replication_record"}},"created_at":"2026-07-05T11:28:29.238639+00:00","updated_at":"2026-07-05T11:28:29.238639+00:00"}