{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QGMG5TIYV2FHBQZDCHEJ5J2ARL","short_pith_number":"pith:QGMG5TIY","schema_version":"1.0","canonical_sha256":"81986ecd18ae8a70c32311c89ea7408add7e1f4fcd9cbfae81684fde15ca5569","source":{"kind":"arxiv","id":"2506.17812","version":1},"attestation_state":"computed","paper":{"title":"Is Your Automated Software Engineer Trustworthy?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Meiyappan Nagappan, Noble Saji Mathews","submitted_at":"2025-06-21T20:56:20Z","abstract_excerpt":"Large Language Models (LLMs) are being increasingly used in software engineering tasks, with an increased focus on bug report resolution over the past year. However, most proposed systems fail to properly handle uncertain or incorrect inputs and outputs. Existing LLM-based tools and coding agents respond to every issue and generate a patch for every case, even when the input is vague or their own output is incorrect. There are no mechanisms in place to abstain when confidence is low. This leads to unreliable behaviour, such as hallucinated code changes or responses based on vague issue reports"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.17812","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SE","submitted_at":"2025-06-21T20:56:20Z","cross_cats_sorted":[],"title_canon_sha256":"f19191f0fd8e5fcd8e611e4430a9d1debfc20fd0d4a69d6ce5c5830c73a1c825","abstract_canon_sha256":"ce47e6a76b42cd9641bd42ac47b89f6192eb9dae324441e87546e59b2b592c10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:23.514112Z","signature_b64":"9V3sFUE79fiBsrscLonYKvrme7JssdMKh6JvBC77zWxLbhkWpVDz9FSPcSxn35OW30G1z7ApuF5twZB3F8F0Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"81986ecd18ae8a70c32311c89ea7408add7e1f4fcd9cbfae81684fde15ca5569","last_reissued_at":"2026-07-05T11:25:23.513637Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:23.513637Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is Your Automated Software Engineer Trustworthy?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Meiyappan Nagappan, Noble Saji Mathews","submitted_at":"2025-06-21T20:56:20Z","abstract_excerpt":"Large Language Models (LLMs) are being increasingly used in software engineering tasks, with an increased focus on bug report resolution over the past year. However, most proposed systems fail to properly handle uncertain or incorrect inputs and outputs. Existing LLM-based tools and coding agents respond to every issue and generate a patch for every case, even when the input is vague or their own output is incorrect. There are no mechanisms in place to abstain when confidence is low. This leads to unreliable behaviour, such as hallucinated code changes or responses based on vague issue reports"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.17812","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.17812/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.17812","created_at":"2026-07-05T11:25:23.513695+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.17812v1","created_at":"2026-07-05T11:25:23.513695+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.17812","created_at":"2026-07-05T11:25:23.513695+00:00"},{"alias_kind":"pith_short_12","alias_value":"QGMG5TIYV2FH","created_at":"2026-07-05T11:25:23.513695+00:00"},{"alias_kind":"pith_short_16","alias_value":"QGMG5TIYV2FHBQZD","created_at":"2026-07-05T11:25:23.513695+00:00"},{"alias_kind":"pith_short_8","alias_value":"QGMG5TIY","created_at":"2026-07-05T11:25:23.513695+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01311","citing_title":"The Partial Testimony of Logs: Evaluation of Language Model Generation under Confounded Model Choice","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16752","citing_title":"Don't Start What You Can't Finish: A Counterfactual Audit of Support-State Triage in LLM Agents","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QGMG5TIYV2FHBQZDCHEJ5J2ARL","json":"https://pith.science/pith/QGMG5TIYV2FHBQZDCHEJ5J2ARL.json","graph_json":"https://pith.science/api/pith-number/QGMG5TIYV2FHBQZDCHEJ5J2ARL/graph.json","events_json":"https://pith.science/api/pith-number/QGMG5TIYV2FHBQZDCHEJ5J2ARL/events.json","paper":"https://pith.science/paper/QGMG5TIY"},"agent_actions":{"view_html":"https://pith.science/pith/QGMG5TIYV2FHBQZDCHEJ5J2ARL","download_json":"https://pith.science/pith/QGMG5TIYV2FHBQZDCHEJ5J2ARL.json","view_paper":"https://pith.science/paper/QGMG5TIY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.17812&json=true","fetch_graph":"https://pith.science/api/pith-number/QGMG5TIYV2FHBQZDCHEJ5J2ARL/graph.json","fetch_events":"https://pith.science/api/pith-number/QGMG5TIYV2FHBQZDCHEJ5J2ARL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QGMG5TIYV2FHBQZDCHEJ5J2ARL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QGMG5TIYV2FHBQZDCHEJ5J2ARL/action/storage_attestation","attest_author":"https://pith.science/pith/QGMG5TIYV2FHBQZDCHEJ5J2ARL/action/author_attestation","sign_citation":"https://pith.science/pith/QGMG5TIYV2FHBQZDCHEJ5J2ARL/action/citation_signature","submit_replication":"https://pith.science/pith/QGMG5TIYV2FHBQZDCHEJ5J2ARL/action/replication_record"}},"created_at":"2026-07-05T11:25:23.513695+00:00","updated_at":"2026-07-05T11:25:23.513695+00:00"}