{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UJLI3D6MYI57NY2DM7JSNRPR36","short_pith_number":"pith:UJLI3D6M","schema_version":"1.0","canonical_sha256":"a2568d8fccc23bf6e34367d326c5f1dfbc528d8e6f893918e99ca177f18a156d","source":{"kind":"arxiv","id":"2312.10091","version":1},"attestation_state":"computed","paper":{"title":"Look Before You Leap: A Universal Emergent Decomposition of Retrieval Tasks in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.IR","authors_text":"Alexandre Variengien, Eric Winsor","submitted_at":"2023-12-13T18:36:43Z","abstract_excerpt":"When solving challenging problems, language models (LMs) are able to identify relevant information from long and complicated contexts. To study how LMs solve retrieval tasks in diverse situations, we introduce ORION, a collection of structured retrieval tasks spanning six domains, from text understanding to coding. Each task in ORION can be represented abstractly by a request (e.g. a question) that retrieves an attribute (e.g. the character name) from a context (e.g. a story). We apply causal analysis on 18 open-source language models with sizes ranging from 125 million to 70 billion parameter"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.10091","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.IR","submitted_at":"2023-12-13T18:36:43Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"0efaa2627ed973618efdacad4bf709678c87ac8dc8fd2c28516223eab24c8451","abstract_canon_sha256":"e789f2bef247a84e5451a7c8cafea840a0ed327d0c679587d979ced80a9c405e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:25:00.722904Z","signature_b64":"DEUuZyDMeh5P6A37m2MH/Fpe418QOGOQv3tKT2VPCBi8HkPS9h41E5GH/9XkulYF3Z9x4GFOw/16HdU+rt05DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2568d8fccc23bf6e34367d326c5f1dfbc528d8e6f893918e99ca177f18a156d","last_reissued_at":"2026-07-05T07:25:00.722432Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:25:00.722432Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Look Before You Leap: A Universal Emergent Decomposition of Retrieval Tasks in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.IR","authors_text":"Alexandre Variengien, Eric Winsor","submitted_at":"2023-12-13T18:36:43Z","abstract_excerpt":"When solving challenging problems, language models (LMs) are able to identify relevant information from long and complicated contexts. To study how LMs solve retrieval tasks in diverse situations, we introduce ORION, a collection of structured retrieval tasks spanning six domains, from text understanding to coding. Each task in ORION can be represented abstractly by a request (e.g. a question) that retrieves an attribute (e.g. the character name) from a context (e.g. a story). We apply causal analysis on 18 open-source language models with sizes ranging from 125 million to 70 billion parameter"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.10091","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.10091/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.10091","created_at":"2026-07-05T07:25:00.722489+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.10091v1","created_at":"2026-07-05T07:25:00.722489+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.10091","created_at":"2026-07-05T07:25:00.722489+00:00"},{"alias_kind":"pith_short_12","alias_value":"UJLI3D6MYI57","created_at":"2026-07-05T07:25:00.722489+00:00"},{"alias_kind":"pith_short_16","alias_value":"UJLI3D6MYI57NY2D","created_at":"2026-07-05T07:25:00.722489+00:00"},{"alias_kind":"pith_short_8","alias_value":"UJLI3D6M","created_at":"2026-07-05T07:25:00.722489+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07560","citing_title":"Function-Vector Heads Are Two Populations: Writers and Cancellers in In-Context Learning","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UJLI3D6MYI57NY2DM7JSNRPR36","json":"https://pith.science/pith/UJLI3D6MYI57NY2DM7JSNRPR36.json","graph_json":"https://pith.science/api/pith-number/UJLI3D6MYI57NY2DM7JSNRPR36/graph.json","events_json":"https://pith.science/api/pith-number/UJLI3D6MYI57NY2DM7JSNRPR36/events.json","paper":"https://pith.science/paper/UJLI3D6M"},"agent_actions":{"view_html":"https://pith.science/pith/UJLI3D6MYI57NY2DM7JSNRPR36","download_json":"https://pith.science/pith/UJLI3D6MYI57NY2DM7JSNRPR36.json","view_paper":"https://pith.science/paper/UJLI3D6M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.10091&json=true","fetch_graph":"https://pith.science/api/pith-number/UJLI3D6MYI57NY2DM7JSNRPR36/graph.json","fetch_events":"https://pith.science/api/pith-number/UJLI3D6MYI57NY2DM7JSNRPR36/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UJLI3D6MYI57NY2DM7JSNRPR36/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UJLI3D6MYI57NY2DM7JSNRPR36/action/storage_attestation","attest_author":"https://pith.science/pith/UJLI3D6MYI57NY2DM7JSNRPR36/action/author_attestation","sign_citation":"https://pith.science/pith/UJLI3D6MYI57NY2DM7JSNRPR36/action/citation_signature","submit_replication":"https://pith.science/pith/UJLI3D6MYI57NY2DM7JSNRPR36/action/replication_record"}},"created_at":"2026-07-05T07:25:00.722489+00:00","updated_at":"2026-07-05T07:25:00.722489+00:00"}