{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AOUUILEXUKRB2XIKTC7FTOR6X5","short_pith_number":"pith:AOUUILEX","schema_version":"1.0","canonical_sha256":"03a9442c97a2a21d5d0a98be59ba3ebf68c90a54d6620bc11828402511460ff5","source":{"kind":"arxiv","id":"2411.01176","version":1},"attestation_state":"computed","paper":{"title":"CmdCaliper: A Semantic-Aware Command-Line Embedding Model and Dataset for Security Research","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cheng-Lin Yang, Che-Yu Lin, Chun-Ying Huang, Sian-Yao Huang","submitted_at":"2024-11-02T08:30:45Z","abstract_excerpt":"This research addresses command-line embedding in cybersecurity, a field obstructed by the lack of comprehensive datasets due to privacy and regulation concerns. We propose the first dataset of similar command lines, named CyPHER, for training and unbiased evaluation. The training set is generated using a set of large language models (LLMs) comprising 28,520 similar command-line pairs. Our testing dataset consists of 2,807 similar command-line pairs sourced from authentic command-line data.\n  In addition, we propose a command-line embedding model named CmdCaliper, enabling the computation of s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.01176","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-02T08:30:45Z","cross_cats_sorted":[],"title_canon_sha256":"df391afaad3eb97e7b520976360d40b2720b1c4da8f419251b9286fac63d1b2e","abstract_canon_sha256":"753a5796ebbc4a638e96d0d9da840a975cf3c465037da334696e18f2aa24d1c3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:30:26.140046Z","signature_b64":"LavL5tlTx+vgH06WGStKApgyRtLP9L8jrJgUKEZyfYNRlRCs420OF8unnnf40+E9ylHEqwMGa5ODBx/NmCG4Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03a9442c97a2a21d5d0a98be59ba3ebf68c90a54d6620bc11828402511460ff5","last_reissued_at":"2026-07-05T09:30:26.139536Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:30:26.139536Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CmdCaliper: A Semantic-Aware Command-Line Embedding Model and Dataset for Security Research","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cheng-Lin Yang, Che-Yu Lin, Chun-Ying Huang, Sian-Yao Huang","submitted_at":"2024-11-02T08:30:45Z","abstract_excerpt":"This research addresses command-line embedding in cybersecurity, a field obstructed by the lack of comprehensive datasets due to privacy and regulation concerns. We propose the first dataset of similar command lines, named CyPHER, for training and unbiased evaluation. The training set is generated using a set of large language models (LLMs) comprising 28,520 similar command-line pairs. Our testing dataset consists of 2,807 similar command-line pairs sourced from authentic command-line data.\n  In addition, we propose a command-line embedding model named CmdCaliper, enabling the computation of s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01176","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.01176/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.01176","created_at":"2026-07-05T09:30:26.139605+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.01176v1","created_at":"2026-07-05T09:30:26.139605+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01176","created_at":"2026-07-05T09:30:26.139605+00:00"},{"alias_kind":"pith_short_12","alias_value":"AOUUILEXUKRB","created_at":"2026-07-05T09:30:26.139605+00:00"},{"alias_kind":"pith_short_16","alias_value":"AOUUILEXUKRB2XIK","created_at":"2026-07-05T09:30:26.139605+00:00"},{"alias_kind":"pith_short_8","alias_value":"AOUUILEX","created_at":"2026-07-05T09:30:26.139605+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.04259","citing_title":"SCADE: Scalable Framework for Anomaly Detection in High-Performance System","ref_index":2024,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AOUUILEXUKRB2XIKTC7FTOR6X5","json":"https://pith.science/pith/AOUUILEXUKRB2XIKTC7FTOR6X5.json","graph_json":"https://pith.science/api/pith-number/AOUUILEXUKRB2XIKTC7FTOR6X5/graph.json","events_json":"https://pith.science/api/pith-number/AOUUILEXUKRB2XIKTC7FTOR6X5/events.json","paper":"https://pith.science/paper/AOUUILEX"},"agent_actions":{"view_html":"https://pith.science/pith/AOUUILEXUKRB2XIKTC7FTOR6X5","download_json":"https://pith.science/pith/AOUUILEXUKRB2XIKTC7FTOR6X5.json","view_paper":"https://pith.science/paper/AOUUILEX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.01176&json=true","fetch_graph":"https://pith.science/api/pith-number/AOUUILEXUKRB2XIKTC7FTOR6X5/graph.json","fetch_events":"https://pith.science/api/pith-number/AOUUILEXUKRB2XIKTC7FTOR6X5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AOUUILEXUKRB2XIKTC7FTOR6X5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AOUUILEXUKRB2XIKTC7FTOR6X5/action/storage_attestation","attest_author":"https://pith.science/pith/AOUUILEXUKRB2XIKTC7FTOR6X5/action/author_attestation","sign_citation":"https://pith.science/pith/AOUUILEXUKRB2XIKTC7FTOR6X5/action/citation_signature","submit_replication":"https://pith.science/pith/AOUUILEXUKRB2XIKTC7FTOR6X5/action/replication_record"}},"created_at":"2026-07-05T09:30:26.139605+00:00","updated_at":"2026-07-05T09:30:26.139605+00:00"}