{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:27V2ZU7JIRGGKIN34GMLFCHXIX","short_pith_number":"pith:27V2ZU7J","schema_version":"1.0","canonical_sha256":"d7ebacd3e9444c6521bbe198b288f745d998711d6a960639f3355afbc05b3468","source":{"kind":"arxiv","id":"2312.14852","version":3},"attestation_state":"computed","paper":{"title":"TACO: Topics in Algorithmic COde generation dataset","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bo-Wen Zhang, Chen Lyu, Ge Li, Guang Liu, Jie Fu, Rongao Li, Tao Huang, Zhihong Sun, Zhi Jin","submitted_at":"2023-12-22T17:25:42Z","abstract_excerpt":"We introduce TACO, an open-source, large-scale code generation dataset, with a focus on the optics of algorithms, designed to provide a more challenging training dataset and evaluation benchmark in the field of code generation models. TACO includes competition-level programming questions that are more challenging, to enhance or evaluate problem understanding and reasoning abilities in real-world programming scenarios. There are 25433 and 1000 coding problems in training and test set, as well as up to 1.55 million diverse solution answers. Moreover, each TACO problem includes several fine-grain"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.14852","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-12-22T17:25:42Z","cross_cats_sorted":[],"title_canon_sha256":"08da05579e75d5b8967d22e3484bb48c317ca40ce13c3a68380a9949bc67c188","abstract_canon_sha256":"03cc182f4695b946b4bc4bfbe8fa9759f405e04ae9ae93ad4e6aff7d178cceb1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:28:13.344324Z","signature_b64":"JddRm4di7n77Gfel/858wuijYyJ8fg0RSdPyXwkh0kgQaUGRdZhfpGhldk16eB2dSrBkozjJJJYjdqBvHD6kBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d7ebacd3e9444c6521bbe198b288f745d998711d6a960639f3355afbc05b3468","last_reissued_at":"2026-07-05T07:28:13.343799Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:28:13.343799Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TACO: Topics in Algorithmic COde generation dataset","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bo-Wen Zhang, Chen Lyu, Ge Li, Guang Liu, Jie Fu, Rongao Li, Tao Huang, Zhihong Sun, Zhi Jin","submitted_at":"2023-12-22T17:25:42Z","abstract_excerpt":"We introduce TACO, an open-source, large-scale code generation dataset, with a focus on the optics of algorithms, designed to provide a more challenging training dataset and evaluation benchmark in the field of code generation models. TACO includes competition-level programming questions that are more challenging, to enhance or evaluate problem understanding and reasoning abilities in real-world programming scenarios. There are 25433 and 1000 coding problems in training and test set, as well as up to 1.55 million diverse solution answers. Moreover, each TACO problem includes several fine-grain"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.14852","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.14852/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.14852","created_at":"2026-07-05T07:28:13.343865+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.14852v3","created_at":"2026-07-05T07:28:13.343865+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.14852","created_at":"2026-07-05T07:28:13.343865+00:00"},{"alias_kind":"pith_short_12","alias_value":"27V2ZU7JIRGG","created_at":"2026-07-05T07:28:13.343865+00:00"},{"alias_kind":"pith_short_16","alias_value":"27V2ZU7JIRGGKIN3","created_at":"2026-07-05T07:28:13.343865+00:00"},{"alias_kind":"pith_short_8","alias_value":"27V2ZU7J","created_at":"2026-07-05T07:28:13.343865+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":31,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01490","citing_title":"Don't Let Gains FADE: Breaking Down Policy Gradient Weights in RL","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02390","citing_title":"DecompRL: Solving Harder Problems by Learning Modular Code Generation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12864","citing_title":"Beyond Problem Solving: UOJ-Bench for Evaluating Code Generation, Hacking, and Repair in Competitive Programming","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06712","citing_title":"Data-Efficient Autoregressive-to-Diffusion Language Models via On-Policy Distillation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03108","citing_title":"EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28751","citing_title":"Extrapolative Weight Averaging Reveals Correctness-Efficiency Frontiers in Code RL","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31058","citing_title":"Combinatorial Synthesis: Scaling Code RLVR via Atomic Decomposition and Recombination","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01476","citing_title":"OmniOPD: Logit-Free On-Policy Distillation via Speculative Verification","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11052","citing_title":"Attention Amnesia in Hybrid LLMs: When CoT Fine-Tuning Breaks Long-Range Recall, and How to Fix It","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17682","citing_title":"From Trainee to Trainer: LLM-Designed Training Environment for RL with Multi-Agent Reasoning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19988","citing_title":"Repository-Level Solidity Code Generation with Large Language Models: From Prompting to Fine-Tuning","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07711","citing_title":"SimCT: Recovering Lost Supervision for Cross-Tokenizer On-Policy Distillation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04970","citing_title":"Skill Neologisms: Towards Skill-based Continual Learning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17361","citing_title":"MasFACT: Continual Multi-Agent Topology Learning via Geometry-Aware Posterior Transfer","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2508.12851","citing_title":"Accelerating Edge Inference for Distributed MoE Models with Latency-Optimized Expert Placement","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18471","citing_title":"CodeRL+: Improving Code Generation via Reinforcement with Execution Semantics Alignment","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20856","citing_title":"NVIDIA Nemotron 3: Efficient and Open Intelligence","ref_index":175,"is_internal_anchor":false},{"citing_arxiv_id":"2505.22312","citing_title":"Skywork Open Reasoner 1 Technical Report","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2601.05242","citing_title":"GDPO: Group reward-Decoupled Normalization Policy Optimization for Multi-reward RL Optimization","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02721","citing_title":"GrandCode: Achieving Grandmaster Level in Competitive Programming via Agentic Reinforcement Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09188","citing_title":"DARE: Difficulty-Adaptive Reinforcement Learning with Co-Evolved Difficulty Estimation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01456","citing_title":"Process Reinforcement through Implicit Rewards","ref_index":153,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04970","citing_title":"Skill Neologisms: Towards Skill-based Continual Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10449","citing_title":"AdverMCTS: Combating Pseudo-Correctness in Code Generation via Adversarial Monte Carlo Tree Search","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2602.15763","citing_title":"GLM-5: from Vibe Coding to Agentic Engineering","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/27V2ZU7JIRGGKIN34GMLFCHXIX","json":"https://pith.science/pith/27V2ZU7JIRGGKIN34GMLFCHXIX.json","graph_json":"https://pith.science/api/pith-number/27V2ZU7JIRGGKIN34GMLFCHXIX/graph.json","events_json":"https://pith.science/api/pith-number/27V2ZU7JIRGGKIN34GMLFCHXIX/events.json","paper":"https://pith.science/paper/27V2ZU7J"},"agent_actions":{"view_html":"https://pith.science/pith/27V2ZU7JIRGGKIN34GMLFCHXIX","download_json":"https://pith.science/pith/27V2ZU7JIRGGKIN34GMLFCHXIX.json","view_paper":"https://pith.science/paper/27V2ZU7J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.14852&json=true","fetch_graph":"https://pith.science/api/pith-number/27V2ZU7JIRGGKIN34GMLFCHXIX/graph.json","fetch_events":"https://pith.science/api/pith-number/27V2ZU7JIRGGKIN34GMLFCHXIX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/27V2ZU7JIRGGKIN34GMLFCHXIX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/27V2ZU7JIRGGKIN34GMLFCHXIX/action/storage_attestation","attest_author":"https://pith.science/pith/27V2ZU7JIRGGKIN34GMLFCHXIX/action/author_attestation","sign_citation":"https://pith.science/pith/27V2ZU7JIRGGKIN34GMLFCHXIX/action/citation_signature","submit_replication":"https://pith.science/pith/27V2ZU7JIRGGKIN34GMLFCHXIX/action/replication_record"}},"created_at":"2026-07-05T07:28:13.343865+00:00","updated_at":"2026-07-05T07:28:13.343865+00:00"}