{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5E7XJHCG3RUOUMLORCIHWJ2DIL","short_pith_number":"pith:5E7XJHCG","schema_version":"1.0","canonical_sha256":"e93f749c46dc68ea316e88907b274342f60332c3a16c9793f3296b30937851d7","source":{"kind":"arxiv","id":"2406.05053","version":2},"attestation_state":"computed","paper":{"title":"Hints-In-Browser: Benchmarking Language Models for Programming Feedback Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Adish Singla, Alkis Gotovos, Nachiket Kotalwar","submitted_at":"2024-06-07T16:22:51Z","abstract_excerpt":"Generative AI and large language models hold great promise in enhancing programming education by generating individualized feedback and hints for learners. Recent works have primarily focused on improving the quality of generated feedback to achieve human tutors' quality. While quality is an important performance criterion, it is not the only criterion to optimize for real-world educational deployments. In this paper, we benchmark language models for programming feedback generation across several performance criteria, including quality, cost, time, and data privacy. The key idea is to leverage"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05053","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-07T16:22:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b8bb60a555a224cea6b13f19f862ac3d9ff1017fa14acdd80858f380e12aa1c9","abstract_canon_sha256":"d6544f8098a2b20362a44306f6f42988d37c7e778ad0170c4cbf0843913245e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:04.345729Z","signature_b64":"tGIoaosFgSyDcaOoAyrK3NYhBqPGeRWc7vQ8iLD66ZHjMOux5XYUpoqdd5cfGyC8G3leBetu1h1T42s/lc5zAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e93f749c46dc68ea316e88907b274342f60332c3a16c9793f3296b30937851d7","last_reissued_at":"2026-07-05T10:26:04.345126Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:04.345126Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hints-In-Browser: Benchmarking Language Models for Programming Feedback Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Adish Singla, Alkis Gotovos, Nachiket Kotalwar","submitted_at":"2024-06-07T16:22:51Z","abstract_excerpt":"Generative AI and large language models hold great promise in enhancing programming education by generating individualized feedback and hints for learners. Recent works have primarily focused on improving the quality of generated feedback to achieve human tutors' quality. While quality is an important performance criterion, it is not the only criterion to optimize for real-world educational deployments. In this paper, we benchmark language models for programming feedback generation across several performance criteria, including quality, cost, time, and data privacy. The key idea is to leverage"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05053","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05053/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05053","created_at":"2026-07-05T10:26:04.345188+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05053v2","created_at":"2026-07-05T10:26:04.345188+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05053","created_at":"2026-07-05T10:26:04.345188+00:00"},{"alias_kind":"pith_short_12","alias_value":"5E7XJHCG3RUO","created_at":"2026-07-05T10:26:04.345188+00:00"},{"alias_kind":"pith_short_16","alias_value":"5E7XJHCG3RUOUMLO","created_at":"2026-07-05T10:26:04.345188+00:00"},{"alias_kind":"pith_short_8","alias_value":"5E7XJHCG","created_at":"2026-07-05T10:26:04.345188+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.18147","citing_title":"NoEsis: Differentially Private Knowledge Transfer in Modular LLM Adaptation","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5E7XJHCG3RUOUMLORCIHWJ2DIL","json":"https://pith.science/pith/5E7XJHCG3RUOUMLORCIHWJ2DIL.json","graph_json":"https://pith.science/api/pith-number/5E7XJHCG3RUOUMLORCIHWJ2DIL/graph.json","events_json":"https://pith.science/api/pith-number/5E7XJHCG3RUOUMLORCIHWJ2DIL/events.json","paper":"https://pith.science/paper/5E7XJHCG"},"agent_actions":{"view_html":"https://pith.science/pith/5E7XJHCG3RUOUMLORCIHWJ2DIL","download_json":"https://pith.science/pith/5E7XJHCG3RUOUMLORCIHWJ2DIL.json","view_paper":"https://pith.science/paper/5E7XJHCG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05053&json=true","fetch_graph":"https://pith.science/api/pith-number/5E7XJHCG3RUOUMLORCIHWJ2DIL/graph.json","fetch_events":"https://pith.science/api/pith-number/5E7XJHCG3RUOUMLORCIHWJ2DIL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5E7XJHCG3RUOUMLORCIHWJ2DIL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5E7XJHCG3RUOUMLORCIHWJ2DIL/action/storage_attestation","attest_author":"https://pith.science/pith/5E7XJHCG3RUOUMLORCIHWJ2DIL/action/author_attestation","sign_citation":"https://pith.science/pith/5E7XJHCG3RUOUMLORCIHWJ2DIL/action/citation_signature","submit_replication":"https://pith.science/pith/5E7XJHCG3RUOUMLORCIHWJ2DIL/action/replication_record"}},"created_at":"2026-07-05T10:26:04.345188+00:00","updated_at":"2026-07-05T10:26:04.345188+00:00"}