{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OZM5KMKUWINMFSI23NMS5BQGUF","short_pith_number":"pith:OZM5KMKU","schema_version":"1.0","canonical_sha256":"7659d53154b21ac2c91adb592e8606a14889813dace3c0b07ec030bc23afac32","source":{"kind":"arxiv","id":"2401.01735","version":1},"attestation_state":"computed","paper":{"title":"Economics Arena for Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.GT","authors_text":"Dianbo Sui, Haochuan Wang, Haoran Bu, Shangmin Guo, Siting Lu, Yi Ren, Yuming Shang","submitted_at":"2024-01-03T13:18:24Z","abstract_excerpt":"Large language models (LLMs) have been extensively used as the backbones for general-purpose agents, and some economics literature suggest that LLMs are capable of playing various types of economics games. Following these works, to overcome the limitation of evaluating LLMs using static benchmarks, we propose to explore competitive games as an evaluation for LLMs to incorporate multi-players and dynamicise the environment. By varying the game history revealed to LLMs-based players, we find that most of LLMs are rational in that they play strategies that can increase their payoffs, but not as r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.01735","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.GT","submitted_at":"2024-01-03T13:18:24Z","cross_cats_sorted":[],"title_canon_sha256":"fd2bdd6a5f41e7ffd554cb576322f9b8a0bba4fb5bf9529614ff17448a633191","abstract_canon_sha256":"c1ecf12848f012e313503554e6d0332e14c55dec8250095a57838d456fb8ecc3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:29:56.487781Z","signature_b64":"SNWs5CGoyAQqWlxaSnjBmPO40BnrIPpBs4fzgQrUVx0Z6DdPTrNSj2VU4vgtsdLeB/Ib7DxaEerHvbP6Hb3MDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7659d53154b21ac2c91adb592e8606a14889813dace3c0b07ec030bc23afac32","last_reissued_at":"2026-07-05T07:29:56.487446Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:29:56.487446Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Economics Arena for Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.GT","authors_text":"Dianbo Sui, Haochuan Wang, Haoran Bu, Shangmin Guo, Siting Lu, Yi Ren, Yuming Shang","submitted_at":"2024-01-03T13:18:24Z","abstract_excerpt":"Large language models (LLMs) have been extensively used as the backbones for general-purpose agents, and some economics literature suggest that LLMs are capable of playing various types of economics games. Following these works, to overcome the limitation of evaluating LLMs using static benchmarks, we propose to explore competitive games as an evaluation for LLMs to incorporate multi-players and dynamicise the environment. By varying the game history revealed to LLMs-based players, we find that most of LLMs are rational in that they play strategies that can increase their payoffs, but not as r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.01735","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.01735/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.01735","created_at":"2026-07-05T07:29:56.487499+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.01735v1","created_at":"2026-07-05T07:29:56.487499+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.01735","created_at":"2026-07-05T07:29:56.487499+00:00"},{"alias_kind":"pith_short_12","alias_value":"OZM5KMKUWINM","created_at":"2026-07-05T07:29:56.487499+00:00"},{"alias_kind":"pith_short_16","alias_value":"OZM5KMKUWINMFSI2","created_at":"2026-07-05T07:29:56.487499+00:00"},{"alias_kind":"pith_short_8","alias_value":"OZM5KMKU","created_at":"2026-07-05T07:29:56.487499+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24162","citing_title":"BehaviorBench: Benchmarking Foundation Models for Behavioral Science Tasks","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OZM5KMKUWINMFSI23NMS5BQGUF","json":"https://pith.science/pith/OZM5KMKUWINMFSI23NMS5BQGUF.json","graph_json":"https://pith.science/api/pith-number/OZM5KMKUWINMFSI23NMS5BQGUF/graph.json","events_json":"https://pith.science/api/pith-number/OZM5KMKUWINMFSI23NMS5BQGUF/events.json","paper":"https://pith.science/paper/OZM5KMKU"},"agent_actions":{"view_html":"https://pith.science/pith/OZM5KMKUWINMFSI23NMS5BQGUF","download_json":"https://pith.science/pith/OZM5KMKUWINMFSI23NMS5BQGUF.json","view_paper":"https://pith.science/paper/OZM5KMKU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.01735&json=true","fetch_graph":"https://pith.science/api/pith-number/OZM5KMKUWINMFSI23NMS5BQGUF/graph.json","fetch_events":"https://pith.science/api/pith-number/OZM5KMKUWINMFSI23NMS5BQGUF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OZM5KMKUWINMFSI23NMS5BQGUF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OZM5KMKUWINMFSI23NMS5BQGUF/action/storage_attestation","attest_author":"https://pith.science/pith/OZM5KMKUWINMFSI23NMS5BQGUF/action/author_attestation","sign_citation":"https://pith.science/pith/OZM5KMKUWINMFSI23NMS5BQGUF/action/citation_signature","submit_replication":"https://pith.science/pith/OZM5KMKUWINMFSI23NMS5BQGUF/action/replication_record"}},"created_at":"2026-07-05T07:29:56.487499+00:00","updated_at":"2026-07-05T07:29:56.487499+00:00"}