{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:5LGD5HNBSYJ2KV6MYRYNBK7NZE","short_pith_number":"pith:5LGD5HNB","schema_version":"1.0","canonical_sha256":"eacc3e9da19613a557ccc470d0abedc92ca7312f01a4e907de0c7638a30be0dc","source":{"kind":"arxiv","id":"2608.10503","version":1},"attestation_state":"computed","paper":{"title":"Every Token Counts: Exact Likert-Scale Distributions for Measuring LLM Attitudes and Biases","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Davood Wadi, Matthew Philp, Mohsen Ghodrat","submitted_at":"2026-08-11T05:20:06Z","abstract_excerpt":"As Large Language Models (LLMs) are increasingly deployed as autonomous agents, accurately evaluating their latent values and biases is critical. The NLP community typically evaluates models using large, unstructured benchmarks. While effective for general capabilities, these datasets fundamentally conflate causal mechanisms: even when an aggregate bias is detected, unstructured evaluations cannot disentangle whether it stems from baseline traits, contextual confounders, or complex interactions. To address this, we introduce an analytically exact framework for the controlled behavioral evaluat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.10503","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-08-11T05:20:06Z","cross_cats_sorted":[],"title_canon_sha256":"25d9e2cb978c8409c4e830cb96b4f25a816f0337aaa5ef81dd95c2b6592240f1","abstract_canon_sha256":"cf69a24793b1d0a4a4e3d41b50f7c1be253f7b6c9c5503d5e791f439d9a771fd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-12T01:22:37.362003Z","signature_b64":"vAJWppBtxY84K3qd9A5jx8U+msY8fVQDcRY+YBvY3twd+T30UnZx2Z1twaXK7aX3Ie4sHiO1C4zHY4aYUmtVDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eacc3e9da19613a557ccc470d0abedc92ca7312f01a4e907de0c7638a30be0dc","last_reissued_at":"2026-08-12T01:22:37.360298Z","signature_status":"signed_v1","first_computed_at":"2026-08-12T01:22:37.360298Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Every Token Counts: Exact Likert-Scale Distributions for Measuring LLM Attitudes and Biases","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Davood Wadi, Matthew Philp, Mohsen Ghodrat","submitted_at":"2026-08-11T05:20:06Z","abstract_excerpt":"As Large Language Models (LLMs) are increasingly deployed as autonomous agents, accurately evaluating their latent values and biases is critical. The NLP community typically evaluates models using large, unstructured benchmarks. While effective for general capabilities, these datasets fundamentally conflate causal mechanisms: even when an aggregate bias is detected, unstructured evaluations cannot disentangle whether it stems from baseline traits, contextual confounders, or complex interactions. To address this, we introduce an analytically exact framework for the controlled behavioral evaluat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.10503","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.10503/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.10503","created_at":"2026-08-12T01:22:37.364224+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.10503v1","created_at":"2026-08-12T01:22:37.364224+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.10503","created_at":"2026-08-12T01:22:37.364224+00:00"},{"alias_kind":"pith_short_12","alias_value":"5LGD5HNBSYJ2","created_at":"2026-08-12T01:22:37.364224+00:00"},{"alias_kind":"pith_short_16","alias_value":"5LGD5HNBSYJ2KV6M","created_at":"2026-08-12T01:22:37.364224+00:00"},{"alias_kind":"pith_short_8","alias_value":"5LGD5HNB","created_at":"2026-08-12T01:22:37.364224+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5LGD5HNBSYJ2KV6MYRYNBK7NZE","json":"https://pith.science/pith/5LGD5HNBSYJ2KV6MYRYNBK7NZE.json","graph_json":"https://pith.science/api/pith-number/5LGD5HNBSYJ2KV6MYRYNBK7NZE/graph.json","events_json":"https://pith.science/api/pith-number/5LGD5HNBSYJ2KV6MYRYNBK7NZE/events.json","paper":"https://pith.science/paper/5LGD5HNB"},"agent_actions":{"view_html":"https://pith.science/pith/5LGD5HNBSYJ2KV6MYRYNBK7NZE","download_json":"https://pith.science/pith/5LGD5HNBSYJ2KV6MYRYNBK7NZE.json","view_paper":"https://pith.science/paper/5LGD5HNB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.10503&json=true","fetch_graph":"https://pith.science/api/pith-number/5LGD5HNBSYJ2KV6MYRYNBK7NZE/graph.json","fetch_events":"https://pith.science/api/pith-number/5LGD5HNBSYJ2KV6MYRYNBK7NZE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5LGD5HNBSYJ2KV6MYRYNBK7NZE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5LGD5HNBSYJ2KV6MYRYNBK7NZE/action/storage_attestation","attest_author":"https://pith.science/pith/5LGD5HNBSYJ2KV6MYRYNBK7NZE/action/author_attestation","sign_citation":"https://pith.science/pith/5LGD5HNBSYJ2KV6MYRYNBK7NZE/action/citation_signature","submit_replication":"https://pith.science/pith/5LGD5HNBSYJ2KV6MYRYNBK7NZE/action/replication_record"}},"created_at":"2026-08-12T01:22:37.364224+00:00","updated_at":"2026-08-12T01:22:37.364224+00:00"}