{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RUIKS42UI223ZTV7C24RSMXLSB","short_pith_number":"pith:RUIKS42U","schema_version":"1.0","canonical_sha256":"8d10a9735446b5bccebf16b91932eb907caf58006148756fb6162803049a3abc","source":{"kind":"arxiv","id":"2405.06807","version":2},"attestation_state":"computed","paper":{"title":"Execution-Based Evaluation of Natural Language to Bash and PowerShell for Incident Remediation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CL","authors_text":"Brent Paulovicks, Ngoc Phuoc An Vo, Vadim Sheinin","submitted_at":"2024-05-10T20:45:34Z","abstract_excerpt":"Given recent advancements of Large Language Models (LLMs), code generation tasks attract immense attention for wide application in different domains. In an effort to evaluate and select a best model to automatically remediate system incidents discovered by Application Performance Monitoring (APM) platforms, it is crucial to verify if the generated code is syntactically and semantically correct, and whether it can be executed correctly as intended. However, current methods for evaluating the quality of code generated by LLMs heavily rely on surface form similarity metrics (e.g. BLEU, ROUGE, and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.06807","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-10T20:45:34Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"f1ddf483604234fa54edfcbab3dad992d44998a4e6356dbd400f1ba0031c6651","abstract_canon_sha256":"cbebb174459d8780e73a14fe6073a7634af65ebf2f2e221e1a3392703d86e953"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:50:02.430236Z","signature_b64":"CcQmw6TAHXTLoqLji6/PrYoGnaTTeEf8AZnCWV75VL6tSlz8MbdUWEJmGWA2n5ylUtqL8YiEH77YHh2GvCprCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8d10a9735446b5bccebf16b91932eb907caf58006148756fb6162803049a3abc","last_reissued_at":"2026-07-05T09:50:02.429777Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:50:02.429777Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Execution-Based Evaluation of Natural Language to Bash and PowerShell for Incident Remediation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CL","authors_text":"Brent Paulovicks, Ngoc Phuoc An Vo, Vadim Sheinin","submitted_at":"2024-05-10T20:45:34Z","abstract_excerpt":"Given recent advancements of Large Language Models (LLMs), code generation tasks attract immense attention for wide application in different domains. In an effort to evaluate and select a best model to automatically remediate system incidents discovered by Application Performance Monitoring (APM) platforms, it is crucial to verify if the generated code is syntactically and semantically correct, and whether it can be executed correctly as intended. However, current methods for evaluating the quality of code generated by LLMs heavily rely on surface form similarity metrics (e.g. BLEU, ROUGE, and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.06807","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.06807/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.06807","created_at":"2026-07-05T09:50:02.429835+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.06807v2","created_at":"2026-07-05T09:50:02.429835+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.06807","created_at":"2026-07-05T09:50:02.429835+00:00"},{"alias_kind":"pith_short_12","alias_value":"RUIKS42UI223","created_at":"2026-07-05T09:50:02.429835+00:00"},{"alias_kind":"pith_short_16","alias_value":"RUIKS42UI223ZTV7","created_at":"2026-07-05T09:50:02.429835+00:00"},{"alias_kind":"pith_short_8","alias_value":"RUIKS42U","created_at":"2026-07-05T09:50:02.429835+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.11237","citing_title":"LLM-as-a-Judge for Reference-less Automatic Code Validation and Refinement for Natural Language to Bash in IT Automation","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RUIKS42UI223ZTV7C24RSMXLSB","json":"https://pith.science/pith/RUIKS42UI223ZTV7C24RSMXLSB.json","graph_json":"https://pith.science/api/pith-number/RUIKS42UI223ZTV7C24RSMXLSB/graph.json","events_json":"https://pith.science/api/pith-number/RUIKS42UI223ZTV7C24RSMXLSB/events.json","paper":"https://pith.science/paper/RUIKS42U"},"agent_actions":{"view_html":"https://pith.science/pith/RUIKS42UI223ZTV7C24RSMXLSB","download_json":"https://pith.science/pith/RUIKS42UI223ZTV7C24RSMXLSB.json","view_paper":"https://pith.science/paper/RUIKS42U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.06807&json=true","fetch_graph":"https://pith.science/api/pith-number/RUIKS42UI223ZTV7C24RSMXLSB/graph.json","fetch_events":"https://pith.science/api/pith-number/RUIKS42UI223ZTV7C24RSMXLSB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RUIKS42UI223ZTV7C24RSMXLSB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RUIKS42UI223ZTV7C24RSMXLSB/action/storage_attestation","attest_author":"https://pith.science/pith/RUIKS42UI223ZTV7C24RSMXLSB/action/author_attestation","sign_citation":"https://pith.science/pith/RUIKS42UI223ZTV7C24RSMXLSB/action/citation_signature","submit_replication":"https://pith.science/pith/RUIKS42UI223ZTV7C24RSMXLSB/action/replication_record"}},"created_at":"2026-07-05T09:50:02.429835+00:00","updated_at":"2026-07-05T09:50:02.429835+00:00"}