{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:Q673SXFMAXHU5HRT6EKSAGSILK","short_pith_number":"pith:Q673SXFM","schema_version":"1.0","canonical_sha256":"87bfb95cac05cf4e9e33f115201a485a8d1b03c4b41ace4d9577bcb8900520a5","source":{"kind":"arxiv","id":"2607.13820","version":1},"attestation_state":"computed","paper":{"title":"PROBE: Benchmarking Code Generation in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Jo\\~ao R. Campos, Marco Vieira, Rodrigo Pato Nogueira","submitted_at":"2026-07-15T13:28:18Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly being used in everyday software engineering tasks, particularly in automated code generation. Despite their widespread adoption, these models remain far from perfect, making systematic and fair evaluation essential to understand their strengths and limitations. In the context of code generation, existing benchmarks are limited: they often target a single programming language and rely primarily on unit test outcomes, while overlooking other critical dimensions such as the overall quality of the generated code and its closeness to a valid solution. T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.13820","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2026-07-15T13:28:18Z","cross_cats_sorted":[],"title_canon_sha256":"03ea9ecea1b2ed4cfcf95d0b84f75d248853b6f23f4e39d54189b250e23868ba","abstract_canon_sha256":"7f538b77cc93d783609f35dd6f255e5dad1ca91e1e4506e56d1312b30b6fd0b4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-16T01:23:08.095706Z","signature_b64":"TcgFu9IXlwNuTZ7hYgOqASHpCqidriFLVpyrJKaoZzvEePscfxfGj74Jl33o7cuoJutu9prvn472H0R0K2v7Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"87bfb95cac05cf4e9e33f115201a485a8d1b03c4b41ace4d9577bcb8900520a5","last_reissued_at":"2026-07-16T01:23:08.094900Z","signature_status":"signed_v1","first_computed_at":"2026-07-16T01:23:08.094900Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PROBE: Benchmarking Code Generation in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Jo\\~ao R. Campos, Marco Vieira, Rodrigo Pato Nogueira","submitted_at":"2026-07-15T13:28:18Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly being used in everyday software engineering tasks, particularly in automated code generation. Despite their widespread adoption, these models remain far from perfect, making systematic and fair evaluation essential to understand their strengths and limitations. In the context of code generation, existing benchmarks are limited: they often target a single programming language and rely primarily on unit test outcomes, while overlooking other critical dimensions such as the overall quality of the generated code and its closeness to a valid solution. T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.13820","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.13820/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.13820","created_at":"2026-07-16T01:23:08.095311+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.13820v1","created_at":"2026-07-16T01:23:08.095311+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.13820","created_at":"2026-07-16T01:23:08.095311+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q673SXFMAXHU","created_at":"2026-07-16T01:23:08.095311+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q673SXFMAXHU5HRT","created_at":"2026-07-16T01:23:08.095311+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q673SXFM","created_at":"2026-07-16T01:23:08.095311+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q673SXFMAXHU5HRT6EKSAGSILK","json":"https://pith.science/pith/Q673SXFMAXHU5HRT6EKSAGSILK.json","graph_json":"https://pith.science/api/pith-number/Q673SXFMAXHU5HRT6EKSAGSILK/graph.json","events_json":"https://pith.science/api/pith-number/Q673SXFMAXHU5HRT6EKSAGSILK/events.json","paper":"https://pith.science/paper/Q673SXFM"},"agent_actions":{"view_html":"https://pith.science/pith/Q673SXFMAXHU5HRT6EKSAGSILK","download_json":"https://pith.science/pith/Q673SXFMAXHU5HRT6EKSAGSILK.json","view_paper":"https://pith.science/paper/Q673SXFM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.13820&json=true","fetch_graph":"https://pith.science/api/pith-number/Q673SXFMAXHU5HRT6EKSAGSILK/graph.json","fetch_events":"https://pith.science/api/pith-number/Q673SXFMAXHU5HRT6EKSAGSILK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q673SXFMAXHU5HRT6EKSAGSILK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q673SXFMAXHU5HRT6EKSAGSILK/action/storage_attestation","attest_author":"https://pith.science/pith/Q673SXFMAXHU5HRT6EKSAGSILK/action/author_attestation","sign_citation":"https://pith.science/pith/Q673SXFMAXHU5HRT6EKSAGSILK/action/citation_signature","submit_replication":"https://pith.science/pith/Q673SXFMAXHU5HRT6EKSAGSILK/action/replication_record"}},"created_at":"2026-07-16T01:23:08.095311+00:00","updated_at":"2026-07-16T01:23:08.095311+00:00"}