{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5QTY2JFTGRDPJIPXCKO2MKVZBL","short_pith_number":"pith:5QTY2JFT","schema_version":"1.0","canonical_sha256":"ec278d24b33446f4a1f7129da62ab90ac46a216dc1a63e2c582169a6817a2b0e","source":{"kind":"arxiv","id":"2509.04810","version":1},"attestation_state":"computed","paper":{"title":"Code Review Without Borders: Evaluating Synthetic vs. Real Data for Review Recommendation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Alexander Apartsin, Dudi Ohayon, Romy Somkin, Yehudit Aperstein, Yogev Cohen","submitted_at":"2025-09-05T05:17:14Z","abstract_excerpt":"Automating the decision of whether a code change requires manual review is vital for maintaining software quality in modern development workflows. However, the emergence of new programming languages and frameworks creates a critical bottleneck: while large volumes of unlabelled code are readily available, there is an insufficient amount of labelled data to train supervised models for review classification. We address this challenge by leveraging Large Language Models (LLMs) to translate code changes from well-resourced languages into equivalent changes in underrepresented or emerging languages"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.04810","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-09-05T05:17:14Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"43f8b6f1ae295365baf206d243d80e3d14469bd97b8481fea3ce11242c1f1d63","abstract_canon_sha256":"810e1994043f91249c511df3666f67ed0832d8d780d28abc8ac5226378ed3c27"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:05:30.118541Z","signature_b64":"TX52Dv1XJM5FO3eo/LS4mXwoYZycJ10ISWzMtxZQ4EyKx0r7VSCqczUAKO3/ppLWIamorGktuf5SV2c1VyQKBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec278d24b33446f4a1f7129da62ab90ac46a216dc1a63e2c582169a6817a2b0e","last_reissued_at":"2026-07-05T12:05:30.117969Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:05:30.117969Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Code Review Without Borders: Evaluating Synthetic vs. Real Data for Review Recommendation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Alexander Apartsin, Dudi Ohayon, Romy Somkin, Yehudit Aperstein, Yogev Cohen","submitted_at":"2025-09-05T05:17:14Z","abstract_excerpt":"Automating the decision of whether a code change requires manual review is vital for maintaining software quality in modern development workflows. However, the emergence of new programming languages and frameworks creates a critical bottleneck: while large volumes of unlabelled code are readily available, there is an insufficient amount of labelled data to train supervised models for review classification. We address this challenge by leveraging Large Language Models (LLMs) to translate code changes from well-resourced languages into equivalent changes in underrepresented or emerging languages"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.04810","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.04810/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.04810","created_at":"2026-07-05T12:05:30.118068+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.04810v1","created_at":"2026-07-05T12:05:30.118068+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.04810","created_at":"2026-07-05T12:05:30.118068+00:00"},{"alias_kind":"pith_short_12","alias_value":"5QTY2JFTGRDP","created_at":"2026-07-05T12:05:30.118068+00:00"},{"alias_kind":"pith_short_16","alias_value":"5QTY2JFTGRDPJIPX","created_at":"2026-07-05T12:05:30.118068+00:00"},{"alias_kind":"pith_short_8","alias_value":"5QTY2JFT","created_at":"2026-07-05T12:05:30.118068+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.05993","citing_title":"Clinical Communication Processing with Models Trained on LLM-Generated Synthetic Data: A Structured Survey and Novel Application Case Studies","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5QTY2JFTGRDPJIPXCKO2MKVZBL","json":"https://pith.science/pith/5QTY2JFTGRDPJIPXCKO2MKVZBL.json","graph_json":"https://pith.science/api/pith-number/5QTY2JFTGRDPJIPXCKO2MKVZBL/graph.json","events_json":"https://pith.science/api/pith-number/5QTY2JFTGRDPJIPXCKO2MKVZBL/events.json","paper":"https://pith.science/paper/5QTY2JFT"},"agent_actions":{"view_html":"https://pith.science/pith/5QTY2JFTGRDPJIPXCKO2MKVZBL","download_json":"https://pith.science/pith/5QTY2JFTGRDPJIPXCKO2MKVZBL.json","view_paper":"https://pith.science/paper/5QTY2JFT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.04810&json=true","fetch_graph":"https://pith.science/api/pith-number/5QTY2JFTGRDPJIPXCKO2MKVZBL/graph.json","fetch_events":"https://pith.science/api/pith-number/5QTY2JFTGRDPJIPXCKO2MKVZBL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5QTY2JFTGRDPJIPXCKO2MKVZBL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5QTY2JFTGRDPJIPXCKO2MKVZBL/action/storage_attestation","attest_author":"https://pith.science/pith/5QTY2JFTGRDPJIPXCKO2MKVZBL/action/author_attestation","sign_citation":"https://pith.science/pith/5QTY2JFTGRDPJIPXCKO2MKVZBL/action/citation_signature","submit_replication":"https://pith.science/pith/5QTY2JFTGRDPJIPXCKO2MKVZBL/action/replication_record"}},"created_at":"2026-07-05T12:05:30.118068+00:00","updated_at":"2026-07-05T12:05:30.118068+00:00"}