{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:4QMW5NXLBSJCRIWOVEL57B45V7","short_pith_number":"pith:4QMW5NXL","schema_version":"1.0","canonical_sha256":"e4196eb6eb0c9228a2cea917df879daff903754060b4f0d81ccca59bd13c4904","source":{"kind":"arxiv","id":"2607.20485","version":1},"attestation_state":"computed","paper":{"title":"Expectation Alignment of Language Models for Real-World User Expectations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Bin Liang, Kam-Fai Wong, Miaomiao Li, Shudong Liu, Yang Wang, Zhiwei Zhang","submitted_at":"2026-06-02T13:24:47Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable performance on standard benchmarks, yet it remains largely unexplored whether they truly meet user expectations. Existing evaluation approaches, relying on model heuristics, expert rubrics, or user simulation, fail to capture the diversity and subtlety of real human expectations, causing models to appear competent while misaligning with what users actually seek. We present the first systematic study of user expectations in real-world LLM interactions, proposing a principled procedure to extract semantically rich expectations and introdu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.20485","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-06-02T13:24:47Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"0485e060e1518354b9da792cfad874b827f0c02ba7b6540812f235c32da60fb3","abstract_canon_sha256":"793789c4b7727cf39798143a2466568d33c0ec396217cd1672bdca674306272d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-24T00:23:18.221999Z","signature_b64":"33gavBrPVfSaMqJPkLt8ACgvwlG/CWr1RWRTukRDNw5qgZmbx8HaYXthMoQmlntP3iarA+TgALf1FVkUyjwiDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e4196eb6eb0c9228a2cea917df879daff903754060b4f0d81ccca59bd13c4904","last_reissued_at":"2026-07-24T00:23:18.221136Z","signature_status":"signed_v1","first_computed_at":"2026-07-24T00:23:18.221136Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Expectation Alignment of Language Models for Real-World User Expectations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Bin Liang, Kam-Fai Wong, Miaomiao Li, Shudong Liu, Yang Wang, Zhiwei Zhang","submitted_at":"2026-06-02T13:24:47Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable performance on standard benchmarks, yet it remains largely unexplored whether they truly meet user expectations. Existing evaluation approaches, relying on model heuristics, expert rubrics, or user simulation, fail to capture the diversity and subtlety of real human expectations, causing models to appear competent while misaligning with what users actually seek. We present the first systematic study of user expectations in real-world LLM interactions, proposing a principled procedure to extract semantically rich expectations and introdu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.20485","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.20485/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.20485","created_at":"2026-07-24T00:23:18.221583+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.20485v1","created_at":"2026-07-24T00:23:18.221583+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.20485","created_at":"2026-07-24T00:23:18.221583+00:00"},{"alias_kind":"pith_short_12","alias_value":"4QMW5NXLBSJC","created_at":"2026-07-24T00:23:18.221583+00:00"},{"alias_kind":"pith_short_16","alias_value":"4QMW5NXLBSJCRIWO","created_at":"2026-07-24T00:23:18.221583+00:00"},{"alias_kind":"pith_short_8","alias_value":"4QMW5NXL","created_at":"2026-07-24T00:23:18.221583+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4QMW5NXLBSJCRIWOVEL57B45V7","json":"https://pith.science/pith/4QMW5NXLBSJCRIWOVEL57B45V7.json","graph_json":"https://pith.science/api/pith-number/4QMW5NXLBSJCRIWOVEL57B45V7/graph.json","events_json":"https://pith.science/api/pith-number/4QMW5NXLBSJCRIWOVEL57B45V7/events.json","paper":"https://pith.science/paper/4QMW5NXL"},"agent_actions":{"view_html":"https://pith.science/pith/4QMW5NXLBSJCRIWOVEL57B45V7","download_json":"https://pith.science/pith/4QMW5NXLBSJCRIWOVEL57B45V7.json","view_paper":"https://pith.science/paper/4QMW5NXL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.20485&json=true","fetch_graph":"https://pith.science/api/pith-number/4QMW5NXLBSJCRIWOVEL57B45V7/graph.json","fetch_events":"https://pith.science/api/pith-number/4QMW5NXLBSJCRIWOVEL57B45V7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4QMW5NXLBSJCRIWOVEL57B45V7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4QMW5NXLBSJCRIWOVEL57B45V7/action/storage_attestation","attest_author":"https://pith.science/pith/4QMW5NXLBSJCRIWOVEL57B45V7/action/author_attestation","sign_citation":"https://pith.science/pith/4QMW5NXLBSJCRIWOVEL57B45V7/action/citation_signature","submit_replication":"https://pith.science/pith/4QMW5NXLBSJCRIWOVEL57B45V7/action/replication_record"}},"created_at":"2026-07-24T00:23:18.221583+00:00","updated_at":"2026-07-24T00:23:18.221583+00:00"}