{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:IPACBRHJEEE3XVPLWFFYRGITD6","short_pith_number":"pith:IPACBRHJ","schema_version":"1.0","canonical_sha256":"43c020c4e92109bbd5ebb14b8899131f85add5e1fe9de11515ee5cdc05c1dd33","source":{"kind":"arxiv","id":"2202.08370","version":4},"attestation_state":"computed","paper":{"title":"CAREER: A Foundation Model for Labor Sequence Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["econ.EM"],"primary_cat":"cs.LG","authors_text":"Ayush Kanodia, David M. Blei, Emil Palikot, Keyon Vafa, Susan Athey, Tianyu Du","submitted_at":"2022-02-16T23:23:50Z","abstract_excerpt":"Labor economists regularly analyze employment data by fitting predictive models to small, carefully constructed longitudinal survey datasets. Although machine learning methods offer promise for such problems, these survey datasets are too small to take advantage of them. In recent years large datasets of online resumes have also become available, providing data about the career trajectories of millions of individuals. However, standard econometric models cannot take advantage of their scale or incorporate them into the analysis of survey data. To this end we develop CAREER, a foundation model "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.08370","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-16T23:23:50Z","cross_cats_sorted":["econ.EM"],"title_canon_sha256":"5ab18b270e48978657653e895566f9a5ff8ab05a1bc9bff9efb85e56e603987a","abstract_canon_sha256":"08ce0a5d5df60d9a724c46ea4845aad1632e11b2e34c31c8567a142894b7f538"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:50:36.989024Z","signature_b64":"ErZ+VHSKkMVlnVMgNStuFNso2qam+SFazHCn1Ywi8elKiFoTIiQthSls5beKBQn/pfglX/VttUTm9mPAWcwKAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43c020c4e92109bbd5ebb14b8899131f85add5e1fe9de11515ee5cdc05c1dd33","last_reissued_at":"2026-07-05T07:50:36.988498Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:50:36.988498Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CAREER: A Foundation Model for Labor Sequence Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["econ.EM"],"primary_cat":"cs.LG","authors_text":"Ayush Kanodia, David M. Blei, Emil Palikot, Keyon Vafa, Susan Athey, Tianyu Du","submitted_at":"2022-02-16T23:23:50Z","abstract_excerpt":"Labor economists regularly analyze employment data by fitting predictive models to small, carefully constructed longitudinal survey datasets. Although machine learning methods offer promise for such problems, these survey datasets are too small to take advantage of them. In recent years large datasets of online resumes have also become available, providing data about the career trajectories of millions of individuals. However, standard econometric models cannot take advantage of their scale or incorporate them into the analysis of survey data. To this end we develop CAREER, a foundation model "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.08370","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.08370/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.08370","created_at":"2026-07-05T07:50:36.988557+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.08370v4","created_at":"2026-07-05T07:50:36.988557+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.08370","created_at":"2026-07-05T07:50:36.988557+00:00"},{"alias_kind":"pith_short_12","alias_value":"IPACBRHJEEE3","created_at":"2026-07-05T07:50:36.988557+00:00"},{"alias_kind":"pith_short_16","alias_value":"IPACBRHJEEE3XVPL","created_at":"2026-07-05T07:50:36.988557+00:00"},{"alias_kind":"pith_short_8","alias_value":"IPACBRHJ","created_at":"2026-07-05T07:50:36.988557+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11220","citing_title":"LifeSentence: Language models can encode human life course trajectories from longitudinal panel data","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IPACBRHJEEE3XVPLWFFYRGITD6","json":"https://pith.science/pith/IPACBRHJEEE3XVPLWFFYRGITD6.json","graph_json":"https://pith.science/api/pith-number/IPACBRHJEEE3XVPLWFFYRGITD6/graph.json","events_json":"https://pith.science/api/pith-number/IPACBRHJEEE3XVPLWFFYRGITD6/events.json","paper":"https://pith.science/paper/IPACBRHJ"},"agent_actions":{"view_html":"https://pith.science/pith/IPACBRHJEEE3XVPLWFFYRGITD6","download_json":"https://pith.science/pith/IPACBRHJEEE3XVPLWFFYRGITD6.json","view_paper":"https://pith.science/paper/IPACBRHJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.08370&json=true","fetch_graph":"https://pith.science/api/pith-number/IPACBRHJEEE3XVPLWFFYRGITD6/graph.json","fetch_events":"https://pith.science/api/pith-number/IPACBRHJEEE3XVPLWFFYRGITD6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IPACBRHJEEE3XVPLWFFYRGITD6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IPACBRHJEEE3XVPLWFFYRGITD6/action/storage_attestation","attest_author":"https://pith.science/pith/IPACBRHJEEE3XVPLWFFYRGITD6/action/author_attestation","sign_citation":"https://pith.science/pith/IPACBRHJEEE3XVPLWFFYRGITD6/action/citation_signature","submit_replication":"https://pith.science/pith/IPACBRHJEEE3XVPLWFFYRGITD6/action/replication_record"}},"created_at":"2026-07-05T07:50:36.988557+00:00","updated_at":"2026-07-05T07:50:36.988557+00:00"}