{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HREEYD5XMZZCLKUZX7P2U3453C","short_pith_number":"pith:HREEYD5X","schema_version":"1.0","canonical_sha256":"3c484c0fb7667225aa99bfdfaa6f9dd884debc5ff776f0937499018b27e63182","source":{"kind":"arxiv","id":"2506.01716","version":1},"attestation_state":"computed","paper":{"title":"Self-Challenging Language Model Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Jason Weston, Sainbayar Sukhbaatar, Sergey Levine, Xian Li, Yifei Zhou","submitted_at":"2025-06-02T14:23:33Z","abstract_excerpt":"Large language models are quickly becoming the foundation for intelligent agents that are capable of using tools. However, training such agents is challenging because it requires human creation and annotation of a diverse set of tasks, tools, and evaluation criteria. In this paper, we propose the Self-Challenging framework for training an agent on high-quality tasks that are generated by itself. The agent first plays the role of challenger and generates a task after interacting with the given tools. The tasks take the form of a novel general class of problems termed Code-as-Task, which are def"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.01716","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-02T14:23:33Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"17c29d89e9e65373e1cdb827990aa112bebd39e96390dc2bf2d25b597cf740c5","abstract_canon_sha256":"9c981448de726c7c4bbf0e506b85b26f62f1b60ff2ae37dcf12b950463506b9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:18.282421Z","signature_b64":"T41UBOcctBAv0BXYnQ86Iquwb1GDffEedv8+4rZX5s9oTT4taToXcdQpvFPmJizTPe6TAEATLaKWWrPLDDtrAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c484c0fb7667225aa99bfdfaa6f9dd884debc5ff776f0937499018b27e63182","last_reissued_at":"2026-07-05T11:14:18.281933Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:18.281933Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Challenging Language Model Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Jason Weston, Sainbayar Sukhbaatar, Sergey Levine, Xian Li, Yifei Zhou","submitted_at":"2025-06-02T14:23:33Z","abstract_excerpt":"Large language models are quickly becoming the foundation for intelligent agents that are capable of using tools. However, training such agents is challenging because it requires human creation and annotation of a diverse set of tasks, tools, and evaluation criteria. In this paper, we propose the Self-Challenging framework for training an agent on high-quality tasks that are generated by itself. The agent first plays the role of challenger and generates a task after interacting with the given tools. The tasks take the form of a novel general class of problems termed Code-as-Task, which are def"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.01716","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.01716/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.01716","created_at":"2026-07-05T11:14:18.281988+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.01716v1","created_at":"2026-07-05T11:14:18.281988+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.01716","created_at":"2026-07-05T11:14:18.281988+00:00"},{"alias_kind":"pith_short_12","alias_value":"HREEYD5XMZZC","created_at":"2026-07-05T11:14:18.281988+00:00"},{"alias_kind":"pith_short_16","alias_value":"HREEYD5XMZZCLKUZ","created_at":"2026-07-05T11:14:18.281988+00:00"},{"alias_kind":"pith_short_8","alias_value":"HREEYD5X","created_at":"2026-07-05T11:14:18.281988+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25996","citing_title":"Autodata: An agentic data scientist to create high quality synthetic data","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25996","citing_title":"Autodata: An agentic data scientist to create high quality synthetic data","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20913","citing_title":"PROTON: Prototype-Based Test-Time Online OOD Detection for Medical VLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12908","citing_title":"SENTINEL: Failure-Driven Reinforcement Learning for Training Tool-Using Language Model Agents","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01286","citing_title":"BenchEvolver: Frontier Task Synthesis via Solution-Centric Evolution","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27483","citing_title":"Internalizing the Future: A Unified Agentic Training Paradigm for World Model Planning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29115","citing_title":"unix-ctf: Procedural Environments for Unix-Competence Reinforcement Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09579","citing_title":"Help Without Being Asked: A Deployed Proactive Agent System for On-Call Support with Continuous Self-Improvement","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09959","citing_title":"G-Zero: Self-Play for Open-Ended Generation from Zero Data","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20051","citing_title":"Bootstrapping Post-training Signals for Open-ended Tasks via Rubric-based Self-play on Pre-training Text","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18131","citing_title":"Training LLM Agents for Spontaneous, Reward-Free Self-Evolution via World Knowledge Exploration","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HREEYD5XMZZCLKUZX7P2U3453C","json":"https://pith.science/pith/HREEYD5XMZZCLKUZX7P2U3453C.json","graph_json":"https://pith.science/api/pith-number/HREEYD5XMZZCLKUZX7P2U3453C/graph.json","events_json":"https://pith.science/api/pith-number/HREEYD5XMZZCLKUZX7P2U3453C/events.json","paper":"https://pith.science/paper/HREEYD5X"},"agent_actions":{"view_html":"https://pith.science/pith/HREEYD5XMZZCLKUZX7P2U3453C","download_json":"https://pith.science/pith/HREEYD5XMZZCLKUZX7P2U3453C.json","view_paper":"https://pith.science/paper/HREEYD5X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.01716&json=true","fetch_graph":"https://pith.science/api/pith-number/HREEYD5XMZZCLKUZX7P2U3453C/graph.json","fetch_events":"https://pith.science/api/pith-number/HREEYD5XMZZCLKUZX7P2U3453C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HREEYD5XMZZCLKUZX7P2U3453C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HREEYD5XMZZCLKUZX7P2U3453C/action/storage_attestation","attest_author":"https://pith.science/pith/HREEYD5XMZZCLKUZX7P2U3453C/action/author_attestation","sign_citation":"https://pith.science/pith/HREEYD5XMZZCLKUZX7P2U3453C/action/citation_signature","submit_replication":"https://pith.science/pith/HREEYD5XMZZCLKUZX7P2U3453C/action/replication_record"}},"created_at":"2026-07-05T11:14:18.281988+00:00","updated_at":"2026-07-05T11:14:18.281988+00:00"}