{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OMZ77NWPBYIBVMPJRVHHKHUWUT","short_pith_number":"pith:OMZ77NWP","schema_version":"1.0","canonical_sha256":"7333ffb6cf0e101ab1e98d4e751e96a4e51c30960b2228c6fe3dfb9b98276909","source":{"kind":"arxiv","id":"2306.03100","version":4},"attestation_state":"computed","paper":{"title":"Rethinking Model Evaluation as Narrowing the Socio-Technical Gap","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.HC","authors_text":"Q. Vera Liao, Ziang Xiao","submitted_at":"2023-06-01T00:01:43Z","abstract_excerpt":"The recent development of generative large language models (LLMs) poses new challenges for model evaluation that the research community and industry have been grappling with. While the versatile capabilities of these models ignite much excitement, they also inevitably make a leap toward homogenization: powering a wide range of applications with a single, often referred to as ``general-purpose'', model. In this position paper, we argue that model evaluation practices must take on a critical task to cope with the challenges and responsibilities brought by this homogenization: providing valid ass"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.03100","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.HC","submitted_at":"2023-06-01T00:01:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b48e998b1a06fa6200339a904113fe4a9bfb26f8726774ed84510c47f824c55c","abstract_canon_sha256":"be9607f24553a026c5f60ea69f3394c25c730450a7e1b23252b768f619d29e71"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:07:38.069559Z","signature_b64":"mTmmcS3EWK64iz0s9+TJigLSNAD4pW/k37m3QvlKwvEQjh/RemTnVCnWtZrcb2yn5y0QnE0oBrpAeoMgIL41CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7333ffb6cf0e101ab1e98d4e751e96a4e51c30960b2228c6fe3dfb9b98276909","last_reissued_at":"2026-07-05T10:07:38.069051Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:07:38.069051Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rethinking Model Evaluation as Narrowing the Socio-Technical Gap","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.HC","authors_text":"Q. Vera Liao, Ziang Xiao","submitted_at":"2023-06-01T00:01:43Z","abstract_excerpt":"The recent development of generative large language models (LLMs) poses new challenges for model evaluation that the research community and industry have been grappling with. While the versatile capabilities of these models ignite much excitement, they also inevitably make a leap toward homogenization: powering a wide range of applications with a single, often referred to as ``general-purpose'', model. In this position paper, we argue that model evaluation practices must take on a critical task to cope with the challenges and responsibilities brought by this homogenization: providing valid ass"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.03100","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.03100/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.03100","created_at":"2026-07-05T10:07:38.069113+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.03100v4","created_at":"2026-07-05T10:07:38.069113+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.03100","created_at":"2026-07-05T10:07:38.069113+00:00"},{"alias_kind":"pith_short_12","alias_value":"OMZ77NWPBYIB","created_at":"2026-07-05T10:07:38.069113+00:00"},{"alias_kind":"pith_short_16","alias_value":"OMZ77NWPBYIBVMPJ","created_at":"2026-07-05T10:07:38.069113+00:00"},{"alias_kind":"pith_short_8","alias_value":"OMZ77NWP","created_at":"2026-07-05T10:07:38.069113+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15205","citing_title":"Does Theory of Mind Improvement Really Benefit Human-AI Interactions? Empirical Findings from Interactive Evaluations","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2511.05501","citing_title":"Towards Real-World Validity in Generative AI Benchmarks: Understanding and Designing Domain-Centered Evaluations for Journalism Practitioners","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20761","citing_title":"TS-Arena -- A Live Forecast Pre-Registration Platform","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16304","citing_title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16403","citing_title":"Computational Hermeneutics: Evaluating generative AI as a cultural technology","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23842","citing_title":"Reheat Nachos for Dinner? Evaluating AI Support for Cross-Cultural Communication of Neologisms","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OMZ77NWPBYIBVMPJRVHHKHUWUT","json":"https://pith.science/pith/OMZ77NWPBYIBVMPJRVHHKHUWUT.json","graph_json":"https://pith.science/api/pith-number/OMZ77NWPBYIBVMPJRVHHKHUWUT/graph.json","events_json":"https://pith.science/api/pith-number/OMZ77NWPBYIBVMPJRVHHKHUWUT/events.json","paper":"https://pith.science/paper/OMZ77NWP"},"agent_actions":{"view_html":"https://pith.science/pith/OMZ77NWPBYIBVMPJRVHHKHUWUT","download_json":"https://pith.science/pith/OMZ77NWPBYIBVMPJRVHHKHUWUT.json","view_paper":"https://pith.science/paper/OMZ77NWP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.03100&json=true","fetch_graph":"https://pith.science/api/pith-number/OMZ77NWPBYIBVMPJRVHHKHUWUT/graph.json","fetch_events":"https://pith.science/api/pith-number/OMZ77NWPBYIBVMPJRVHHKHUWUT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OMZ77NWPBYIBVMPJRVHHKHUWUT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OMZ77NWPBYIBVMPJRVHHKHUWUT/action/storage_attestation","attest_author":"https://pith.science/pith/OMZ77NWPBYIBVMPJRVHHKHUWUT/action/author_attestation","sign_citation":"https://pith.science/pith/OMZ77NWPBYIBVMPJRVHHKHUWUT/action/citation_signature","submit_replication":"https://pith.science/pith/OMZ77NWPBYIBVMPJRVHHKHUWUT/action/replication_record"}},"created_at":"2026-07-05T10:07:38.069113+00:00","updated_at":"2026-07-05T10:07:38.069113+00:00"}