{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2I3JERYW45R4HRYXQAKTHGYWXJ","short_pith_number":"pith:2I3JERYW","schema_version":"1.0","canonical_sha256":"d236924716e763c3c7178015339b16ba59843d8de7bcec75b38d2f99817ed87b","source":{"kind":"arxiv","id":"2506.10378","version":1},"attestation_state":"computed","paper":{"title":"Discovering Hierarchical Latent Capabilities of Language Models via Causal Representation Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Hanlin Zhang, Jikai Jin, Sham Kakade, Vasilis Syrgkanis","submitted_at":"2025-06-12T06:07:42Z","abstract_excerpt":"Faithful evaluation of language model capabilities is crucial for deriving actionable insights that can inform model development. However, rigorous causal evaluations in this domain face significant methodological challenges, including complex confounding effects and prohibitive computational costs associated with extensive retraining. To tackle these challenges, we propose a causal representation learning framework wherein observed benchmark performance is modeled as a linear transformation of a few latent capability factors. Crucially, these latent factors are identified as causally interrel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.10378","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-12T06:07:42Z","cross_cats_sorted":["cs.AI","cs.CL","stat.ML"],"title_canon_sha256":"8248ac30504be4f4dc83a59d0fbca0397300bccdd5167d35526ce4cb9c07fd08","abstract_canon_sha256":"ce800f931f9fe9d56310931755b62de55b26a799d80ca569896c16a0becd30e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:22.298906Z","signature_b64":"QVxYRw5fSyCrjPW4rqr9vJ9u90sZhVAlIkfa2ZtfvsnfwX0+VzV9BNrxFRXhM3u7FioidniuRUdVCzaJCPjaAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d236924716e763c3c7178015339b16ba59843d8de7bcec75b38d2f99817ed87b","last_reissued_at":"2026-07-05T11:20:22.298308Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:22.298308Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Discovering Hierarchical Latent Capabilities of Language Models via Causal Representation Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Hanlin Zhang, Jikai Jin, Sham Kakade, Vasilis Syrgkanis","submitted_at":"2025-06-12T06:07:42Z","abstract_excerpt":"Faithful evaluation of language model capabilities is crucial for deriving actionable insights that can inform model development. However, rigorous causal evaluations in this domain face significant methodological challenges, including complex confounding effects and prohibitive computational costs associated with extensive retraining. To tackle these challenges, we propose a causal representation learning framework wherein observed benchmark performance is modeled as a linear transformation of a few latent capability factors. Crucially, these latent factors are identified as causally interrel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.10378","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.10378/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.10378","created_at":"2026-07-05T11:20:22.298396+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.10378v1","created_at":"2026-07-05T11:20:22.298396+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.10378","created_at":"2026-07-05T11:20:22.298396+00:00"},{"alias_kind":"pith_short_12","alias_value":"2I3JERYW45R4","created_at":"2026-07-05T11:20:22.298396+00:00"},{"alias_kind":"pith_short_16","alias_value":"2I3JERYW45R4HRYX","created_at":"2026-07-05T11:20:22.298396+00:00"},{"alias_kind":"pith_short_8","alias_value":"2I3JERYW","created_at":"2026-07-05T11:20:22.298396+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28179","citing_title":"SuperValid: Capability-Aligned OOD Validation for Generalizable Downstream Scaling","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2I3JERYW45R4HRYXQAKTHGYWXJ","json":"https://pith.science/pith/2I3JERYW45R4HRYXQAKTHGYWXJ.json","graph_json":"https://pith.science/api/pith-number/2I3JERYW45R4HRYXQAKTHGYWXJ/graph.json","events_json":"https://pith.science/api/pith-number/2I3JERYW45R4HRYXQAKTHGYWXJ/events.json","paper":"https://pith.science/paper/2I3JERYW"},"agent_actions":{"view_html":"https://pith.science/pith/2I3JERYW45R4HRYXQAKTHGYWXJ","download_json":"https://pith.science/pith/2I3JERYW45R4HRYXQAKTHGYWXJ.json","view_paper":"https://pith.science/paper/2I3JERYW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.10378&json=true","fetch_graph":"https://pith.science/api/pith-number/2I3JERYW45R4HRYXQAKTHGYWXJ/graph.json","fetch_events":"https://pith.science/api/pith-number/2I3JERYW45R4HRYXQAKTHGYWXJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2I3JERYW45R4HRYXQAKTHGYWXJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2I3JERYW45R4HRYXQAKTHGYWXJ/action/storage_attestation","attest_author":"https://pith.science/pith/2I3JERYW45R4HRYXQAKTHGYWXJ/action/author_attestation","sign_citation":"https://pith.science/pith/2I3JERYW45R4HRYXQAKTHGYWXJ/action/citation_signature","submit_replication":"https://pith.science/pith/2I3JERYW45R4HRYXQAKTHGYWXJ/action/replication_record"}},"created_at":"2026-07-05T11:20:22.298396+00:00","updated_at":"2026-07-05T11:20:22.298396+00:00"}