{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DALASW7SPVCB7SY3VCNKW26YWW","short_pith_number":"pith:DALASW7S","schema_version":"1.0","canonical_sha256":"1816095bf27d441fcb1ba89aab6bd8b59af31751091fe9fbcc7601e93efa2f02","source":{"kind":"arxiv","id":"2507.13302","version":1},"attestation_state":"computed","paper":{"title":"The Generative Energy Arena (GEA): Incorporating Energy Awareness in Large Language Model (LLM) Human Evaluations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Carlos Arriaga, Eneko Sendin, Gonzalo Mart\\'inez, Javier Conde, Pedro Reviriego","submitted_at":"2025-07-17T17:11:14Z","abstract_excerpt":"The evaluation of large language models is a complex task, in which several approaches have been proposed. The most common is the use of automated benchmarks in which LLMs have to answer multiple-choice questions of different topics. However, this method has certain limitations, being the most concerning, the poor correlation with the humans. An alternative approach, is to have humans evaluate the LLMs. This poses scalability issues as there is a large and growing number of models to evaluate making it impractical (and costly) to run traditional studies based on recruiting a number of evaluato"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.13302","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-07-17T17:11:14Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"7e7e73b5df5ad68a377a057d592784109e69d95b033c83c4745db33ec7fe5228","abstract_canon_sha256":"f8054117edbd3384b84cb4b6658e0123d88437477a05611dc3f52d70d0ecae52"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:38:58.754392Z","signature_b64":"s/F1LdEdk+AYXya0Q2dcdCDt/cNkU2ca/el+oE6tj/zGLDTVxJ9W0dEVdv2l0cqmWPZBNchqZMEDD5+kPDJhCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1816095bf27d441fcb1ba89aab6bd8b59af31751091fe9fbcc7601e93efa2f02","last_reissued_at":"2026-07-05T11:38:58.753927Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:38:58.753927Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Generative Energy Arena (GEA): Incorporating Energy Awareness in Large Language Model (LLM) Human Evaluations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Carlos Arriaga, Eneko Sendin, Gonzalo Mart\\'inez, Javier Conde, Pedro Reviriego","submitted_at":"2025-07-17T17:11:14Z","abstract_excerpt":"The evaluation of large language models is a complex task, in which several approaches have been proposed. The most common is the use of automated benchmarks in which LLMs have to answer multiple-choice questions of different topics. However, this method has certain limitations, being the most concerning, the poor correlation with the humans. An alternative approach, is to have humans evaluate the LLMs. This poses scalability issues as there is a large and growing number of models to evaluate making it impractical (and costly) to run traditional studies based on recruiting a number of evaluato"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.13302","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.13302/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.13302","created_at":"2026-07-05T11:38:58.753982+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.13302v1","created_at":"2026-07-05T11:38:58.753982+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.13302","created_at":"2026-07-05T11:38:58.753982+00:00"},{"alias_kind":"pith_short_12","alias_value":"DALASW7SPVCB","created_at":"2026-07-05T11:38:58.753982+00:00"},{"alias_kind":"pith_short_16","alias_value":"DALASW7SPVCB7SY3","created_at":"2026-07-05T11:38:58.753982+00:00"},{"alias_kind":"pith_short_8","alias_value":"DALASW7S","created_at":"2026-07-05T11:38:58.753982+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DALASW7SPVCB7SY3VCNKW26YWW","json":"https://pith.science/pith/DALASW7SPVCB7SY3VCNKW26YWW.json","graph_json":"https://pith.science/api/pith-number/DALASW7SPVCB7SY3VCNKW26YWW/graph.json","events_json":"https://pith.science/api/pith-number/DALASW7SPVCB7SY3VCNKW26YWW/events.json","paper":"https://pith.science/paper/DALASW7S"},"agent_actions":{"view_html":"https://pith.science/pith/DALASW7SPVCB7SY3VCNKW26YWW","download_json":"https://pith.science/pith/DALASW7SPVCB7SY3VCNKW26YWW.json","view_paper":"https://pith.science/paper/DALASW7S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.13302&json=true","fetch_graph":"https://pith.science/api/pith-number/DALASW7SPVCB7SY3VCNKW26YWW/graph.json","fetch_events":"https://pith.science/api/pith-number/DALASW7SPVCB7SY3VCNKW26YWW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DALASW7SPVCB7SY3VCNKW26YWW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DALASW7SPVCB7SY3VCNKW26YWW/action/storage_attestation","attest_author":"https://pith.science/pith/DALASW7SPVCB7SY3VCNKW26YWW/action/author_attestation","sign_citation":"https://pith.science/pith/DALASW7SPVCB7SY3VCNKW26YWW/action/citation_signature","submit_replication":"https://pith.science/pith/DALASW7SPVCB7SY3VCNKW26YWW/action/replication_record"}},"created_at":"2026-07-05T11:38:58.753982+00:00","updated_at":"2026-07-05T11:38:58.753982+00:00"}