{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7MSW4WLAMDEMUSILHVWYNTIFBO","short_pith_number":"pith:7MSW4WLA","schema_version":"1.0","canonical_sha256":"fb256e596060c8ca490b3d6d86cd050b8e7a20f247f63c41208a4927b20d0dd4","source":{"kind":"arxiv","id":"2509.02855","version":1},"attestation_state":"computed","paper":{"title":"IDEAlign: Comparing Large Language Models to Human Experts in Open-ended Interpretive Annotations","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.CL","authors_text":"Dorottya Demszky, Hyunji Nam, James Malamut, Lucia Langlois, Mei Tan","submitted_at":"2025-09-02T21:58:58Z","abstract_excerpt":"Large language models (LLMs) are increasingly applied to open-ended, interpretive annotation tasks, such as thematic analysis by researchers or generating feedback on student work by teachers. These tasks involve free-text annotations requiring expert-level judgments grounded in specific objectives (e.g., research questions or instructional goals). Evaluating whether LLM-generated annotations align with those generated by expert humans is challenging to do at scale, and currently, no validated, scalable measure of similarity in ideas exists. In this paper, we (i) introduce the scalable evaluat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.02855","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-02T21:58:58Z","cross_cats_sorted":["cs.CY"],"title_canon_sha256":"4b11149731f1526e336e184cec79927b129ec4e3a98f437be4c8980b2a320ede","abstract_canon_sha256":"25b70e9b4f0cc61582b950a7f74165a07a80be6411d4fdb8e198013c2555787a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:03:58.529372Z","signature_b64":"CLfXw4TIgHRJQL3CV0Qutcg8jBxR35I1NAbIMj8Sh7NCn3tRdgwKrDO9HtfOKjfM5+N7BPnwuXXIaE9pmIvDCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb256e596060c8ca490b3d6d86cd050b8e7a20f247f63c41208a4927b20d0dd4","last_reissued_at":"2026-07-05T12:03:58.528693Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:03:58.528693Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"IDEAlign: Comparing Large Language Models to Human Experts in Open-ended Interpretive Annotations","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.CL","authors_text":"Dorottya Demszky, Hyunji Nam, James Malamut, Lucia Langlois, Mei Tan","submitted_at":"2025-09-02T21:58:58Z","abstract_excerpt":"Large language models (LLMs) are increasingly applied to open-ended, interpretive annotation tasks, such as thematic analysis by researchers or generating feedback on student work by teachers. These tasks involve free-text annotations requiring expert-level judgments grounded in specific objectives (e.g., research questions or instructional goals). Evaluating whether LLM-generated annotations align with those generated by expert humans is challenging to do at scale, and currently, no validated, scalable measure of similarity in ideas exists. In this paper, we (i) introduce the scalable evaluat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.02855","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.02855/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.02855","created_at":"2026-07-05T12:03:58.528780+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.02855v1","created_at":"2026-07-05T12:03:58.528780+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.02855","created_at":"2026-07-05T12:03:58.528780+00:00"},{"alias_kind":"pith_short_12","alias_value":"7MSW4WLAMDEM","created_at":"2026-07-05T12:03:58.528780+00:00"},{"alias_kind":"pith_short_16","alias_value":"7MSW4WLAMDEMUSIL","created_at":"2026-07-05T12:03:58.528780+00:00"},{"alias_kind":"pith_short_8","alias_value":"7MSW4WLA","created_at":"2026-07-05T12:03:58.528780+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7MSW4WLAMDEMUSILHVWYNTIFBO","json":"https://pith.science/pith/7MSW4WLAMDEMUSILHVWYNTIFBO.json","graph_json":"https://pith.science/api/pith-number/7MSW4WLAMDEMUSILHVWYNTIFBO/graph.json","events_json":"https://pith.science/api/pith-number/7MSW4WLAMDEMUSILHVWYNTIFBO/events.json","paper":"https://pith.science/paper/7MSW4WLA"},"agent_actions":{"view_html":"https://pith.science/pith/7MSW4WLAMDEMUSILHVWYNTIFBO","download_json":"https://pith.science/pith/7MSW4WLAMDEMUSILHVWYNTIFBO.json","view_paper":"https://pith.science/paper/7MSW4WLA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.02855&json=true","fetch_graph":"https://pith.science/api/pith-number/7MSW4WLAMDEMUSILHVWYNTIFBO/graph.json","fetch_events":"https://pith.science/api/pith-number/7MSW4WLAMDEMUSILHVWYNTIFBO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7MSW4WLAMDEMUSILHVWYNTIFBO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7MSW4WLAMDEMUSILHVWYNTIFBO/action/storage_attestation","attest_author":"https://pith.science/pith/7MSW4WLAMDEMUSILHVWYNTIFBO/action/author_attestation","sign_citation":"https://pith.science/pith/7MSW4WLAMDEMUSILHVWYNTIFBO/action/citation_signature","submit_replication":"https://pith.science/pith/7MSW4WLAMDEMUSILHVWYNTIFBO/action/replication_record"}},"created_at":"2026-07-05T12:03:58.528780+00:00","updated_at":"2026-07-05T12:03:58.528780+00:00"}