{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D4WQ5E2C2T3AOHX5KGASV66UV4","short_pith_number":"pith:D4WQ5E2C","schema_version":"1.0","canonical_sha256":"1f2d0e9342d4f6071efd51812afbd4af04feb39d12abe71a2e209fef01406e07","source":{"kind":"arxiv","id":"2408.09343","version":1},"attestation_state":"computed","paper":{"title":"Large-Scale Pretraining and Finetuning for Efficient Jet Classification in Particle Physics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["physics.data-an"],"primary_cat":"hep-ex","authors_text":"Farouk Mokhtar, Haoyang Li, Javier Duarte, Raghav Kansal, Zihan Zhao","submitted_at":"2024-08-18T03:31:21Z","abstract_excerpt":"This study introduces an innovative approach to analyzing unlabeled data in high-energy physics (HEP) through the application of self-supervised learning (SSL). Faced with the increasing computational cost of producing high-quality labeled simulation samples at the CERN LHC, we propose leveraging large volumes of unlabeled data to overcome the limitations of supervised learning methods, which heavily rely on detailed labeled simulations. By pretraining models on these vast, mostly untapped datasets, we aim to learn generic representations that can be finetuned with smaller quantities of labele"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.09343","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"hep-ex","submitted_at":"2024-08-18T03:31:21Z","cross_cats_sorted":["physics.data-an"],"title_canon_sha256":"dd060823c5e95e990ef03d109e31f70b52b893d66d564f05d88a7851d14a9c12","abstract_canon_sha256":"beabe059aa98cab7b4217ea52c7dd71af64b3c65b0ec48f393695a4f07ff774d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:56:31.270615Z","signature_b64":"zGUtHLljyYiph0m95t2JnhVT5DAGCaD9F3v7nE6xEwak/SRbJv/ehQlA/8sc5rBRiz7ioZKn/nZOI1b06yjaDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f2d0e9342d4f6071efd51812afbd4af04feb39d12abe71a2e209fef01406e07","last_reissued_at":"2026-07-05T08:56:31.270144Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:56:31.270144Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large-Scale Pretraining and Finetuning for Efficient Jet Classification in Particle Physics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["physics.data-an"],"primary_cat":"hep-ex","authors_text":"Farouk Mokhtar, Haoyang Li, Javier Duarte, Raghav Kansal, Zihan Zhao","submitted_at":"2024-08-18T03:31:21Z","abstract_excerpt":"This study introduces an innovative approach to analyzing unlabeled data in high-energy physics (HEP) through the application of self-supervised learning (SSL). Faced with the increasing computational cost of producing high-quality labeled simulation samples at the CERN LHC, we propose leveraging large volumes of unlabeled data to overcome the limitations of supervised learning methods, which heavily rely on detailed labeled simulations. By pretraining models on these vast, mostly untapped datasets, we aim to learn generic representations that can be finetuned with smaller quantities of labele"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.09343","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.09343/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.09343","created_at":"2026-07-05T08:56:31.270204+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.09343v1","created_at":"2026-07-05T08:56:31.270204+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.09343","created_at":"2026-07-05T08:56:31.270204+00:00"},{"alias_kind":"pith_short_12","alias_value":"D4WQ5E2C2T3A","created_at":"2026-07-05T08:56:31.270204+00:00"},{"alias_kind":"pith_short_16","alias_value":"D4WQ5E2C2T3AOHX5","created_at":"2026-07-05T08:56:31.270204+00:00"},{"alias_kind":"pith_short_8","alias_value":"D4WQ5E2C","created_at":"2026-07-05T08:56:31.270204+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.07724","citing_title":"Contrastive Learning for Robust Representations of Neutrino Data","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D4WQ5E2C2T3AOHX5KGASV66UV4","json":"https://pith.science/pith/D4WQ5E2C2T3AOHX5KGASV66UV4.json","graph_json":"https://pith.science/api/pith-number/D4WQ5E2C2T3AOHX5KGASV66UV4/graph.json","events_json":"https://pith.science/api/pith-number/D4WQ5E2C2T3AOHX5KGASV66UV4/events.json","paper":"https://pith.science/paper/D4WQ5E2C"},"agent_actions":{"view_html":"https://pith.science/pith/D4WQ5E2C2T3AOHX5KGASV66UV4","download_json":"https://pith.science/pith/D4WQ5E2C2T3AOHX5KGASV66UV4.json","view_paper":"https://pith.science/paper/D4WQ5E2C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.09343&json=true","fetch_graph":"https://pith.science/api/pith-number/D4WQ5E2C2T3AOHX5KGASV66UV4/graph.json","fetch_events":"https://pith.science/api/pith-number/D4WQ5E2C2T3AOHX5KGASV66UV4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D4WQ5E2C2T3AOHX5KGASV66UV4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D4WQ5E2C2T3AOHX5KGASV66UV4/action/storage_attestation","attest_author":"https://pith.science/pith/D4WQ5E2C2T3AOHX5KGASV66UV4/action/author_attestation","sign_citation":"https://pith.science/pith/D4WQ5E2C2T3AOHX5KGASV66UV4/action/citation_signature","submit_replication":"https://pith.science/pith/D4WQ5E2C2T3AOHX5KGASV66UV4/action/replication_record"}},"created_at":"2026-07-05T08:56:31.270204+00:00","updated_at":"2026-07-05T08:56:31.270204+00:00"}