{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:S3BVIFCFG4NELXTFRMVXM7WTW5","short_pith_number":"pith:S3BVIFCF","schema_version":"1.0","canonical_sha256":"96c3541445371a45de658b2b767ed3b76497f731c7bdf55019425965eb53db51","source":{"kind":"arxiv","id":"2507.15061","version":1},"attestation_state":"computed","paper":{"title":"WebShaper: Agentically Data Synthesizing via Information-Seeking Formalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Baixuan Li, Fei Huang, Haiyang Shen, Jialong Wu, Jingren Zhou, Junkai Zhang, Kuan Li, Liwen Zhang, Pengjun Xie, Wenbiao Yin, Xinyu Wang, Yong Jiang, Zhengwei Tao","submitted_at":"2025-07-20T17:53:37Z","abstract_excerpt":"The advent of Large Language Model (LLM)-powered agents has revolutionized artificial intelligence by enabling solutions to complex, open-ended tasks through web-based information-seeking (IS) capabilities. The scarcity of high-quality training data has limited the development of IS agents. Existing approaches typically adopt an information-driven paradigm that first collects web data and then generates questions based on the retrieval. However, this may lead to inconsistency between information structure and reasoning structure, question and answer. To mitigate, we propose a formalization-dri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.15061","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-07-20T17:53:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e243d55a885a6b6f21e70c894fd3566916b300820f7463df6846eb2033bf0ec9","abstract_canon_sha256":"818472d091d4f36b24cdb5fb392babbd698ba3f73e2c6da50dc0df2b9e502408"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:17.912333Z","signature_b64":"LuAOtq2Qdriy4SndcnriepVi02fdivLIkzed8h4Q8cxUaR2fDKnfrDzk+2lnkqVJiXQdHPPJnDTFtpXFrjqMBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"96c3541445371a45de658b2b767ed3b76497f731c7bdf55019425965eb53db51","last_reissued_at":"2026-07-05T11:40:17.911839Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:17.911839Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WebShaper: Agentically Data Synthesizing via Information-Seeking Formalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Baixuan Li, Fei Huang, Haiyang Shen, Jialong Wu, Jingren Zhou, Junkai Zhang, Kuan Li, Liwen Zhang, Pengjun Xie, Wenbiao Yin, Xinyu Wang, Yong Jiang, Zhengwei Tao","submitted_at":"2025-07-20T17:53:37Z","abstract_excerpt":"The advent of Large Language Model (LLM)-powered agents has revolutionized artificial intelligence by enabling solutions to complex, open-ended tasks through web-based information-seeking (IS) capabilities. The scarcity of high-quality training data has limited the development of IS agents. Existing approaches typically adopt an information-driven paradigm that first collects web data and then generates questions based on the retrieval. However, this may lead to inconsistency between information structure and reasoning structure, question and answer. To mitigate, we propose a formalization-dri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.15061","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.15061/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.15061","created_at":"2026-07-05T11:40:17.911898+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.15061v1","created_at":"2026-07-05T11:40:17.911898+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.15061","created_at":"2026-07-05T11:40:17.911898+00:00"},{"alias_kind":"pith_short_12","alias_value":"S3BVIFCFG4NE","created_at":"2026-07-05T11:40:17.911898+00:00"},{"alias_kind":"pith_short_16","alias_value":"S3BVIFCFG4NELXTF","created_at":"2026-07-05T11:40:17.911898+00:00"},{"alias_kind":"pith_short_8","alias_value":"S3BVIFCF","created_at":"2026-07-05T11:40:17.911898+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24233","citing_title":"Latent Visual States for Efficient Multimodal Reasoning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27330","citing_title":"Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20122","citing_title":"ScaffoldAgent: Utility-Guided Dynamic Outline Optimization for Open-Ended Deep Research","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12087","citing_title":"FORT-Searcher: Synthesizing Shortcut-Resistant Search Tasks for Training Deep Search Agents","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09138","citing_title":"Claw-R1: A Step-Level Data Middleware System for Agentic Reinforcement Learning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07689","citing_title":"Struct-Searcher: Agentic Structural Thinking Advances Multimodal Deep Information Seeking","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04703","citing_title":"Rethinking Continual Experience Internalization for Self-Evolving LLM Agents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31504","citing_title":"SimpleSearch-VL: A Simple Recipe for Multimodal Agentic Deep Search","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12191","citing_title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","ref_index":269,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11926","citing_title":"Toward Generalist Autonomous Research via Hypothesis-Tree Refinement","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22138","citing_title":"Efficient Agentic Reasoning Through Self-Regulated Simulative Planning","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20876","citing_title":"Terminal-World: Scaling Terminal-Agent Environments via Agent Skills","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2508.00414","citing_title":"Cognitive Kernel-Pro: A Framework for Deep Research Agents and Agent Foundation Models Training","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02547","citing_title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","ref_index":284,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07969","citing_title":"Mini-o3: Scaling Up Reasoning Patterns and Interaction Turns for Visual Search","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2511.11793","citing_title":"MiroThinker: Pushing the Performance Boundaries of Open-Source Research Agents via Model, Context, and Interactive Scaling","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15808","citing_title":"Inference-Time Scaling of Verification: Self-Evolving Deep Research Agents via Test-Time Rubric-Guided Verification","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2508.05748","citing_title":"WebWatcher: Breaking New Frontier of Vision-Language Deep Research Agent","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2603.04751","citing_title":"Evaluating the Search Agent in a Parallel World","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13034","citing_title":"ViDR: Grounding Multimodal Deep Research Reports in Source Visual Evidence","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03679","citing_title":"LightThinker++: From Reasoning Compression to Memory Management","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06777","citing_title":"Walk the Talk: Bridging the Reasoning-Action Gap for Thinking with Images via Multimodal Agentic Policy Optimization","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14518","citing_title":"Mind DeepResearch Technical Report","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S3BVIFCFG4NELXTFRMVXM7WTW5","json":"https://pith.science/pith/S3BVIFCFG4NELXTFRMVXM7WTW5.json","graph_json":"https://pith.science/api/pith-number/S3BVIFCFG4NELXTFRMVXM7WTW5/graph.json","events_json":"https://pith.science/api/pith-number/S3BVIFCFG4NELXTFRMVXM7WTW5/events.json","paper":"https://pith.science/paper/S3BVIFCF"},"agent_actions":{"view_html":"https://pith.science/pith/S3BVIFCFG4NELXTFRMVXM7WTW5","download_json":"https://pith.science/pith/S3BVIFCFG4NELXTFRMVXM7WTW5.json","view_paper":"https://pith.science/paper/S3BVIFCF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.15061&json=true","fetch_graph":"https://pith.science/api/pith-number/S3BVIFCFG4NELXTFRMVXM7WTW5/graph.json","fetch_events":"https://pith.science/api/pith-number/S3BVIFCFG4NELXTFRMVXM7WTW5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S3BVIFCFG4NELXTFRMVXM7WTW5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S3BVIFCFG4NELXTFRMVXM7WTW5/action/storage_attestation","attest_author":"https://pith.science/pith/S3BVIFCFG4NELXTFRMVXM7WTW5/action/author_attestation","sign_citation":"https://pith.science/pith/S3BVIFCFG4NELXTFRMVXM7WTW5/action/citation_signature","submit_replication":"https://pith.science/pith/S3BVIFCFG4NELXTFRMVXM7WTW5/action/replication_record"}},"created_at":"2026-07-05T11:40:17.911898+00:00","updated_at":"2026-07-05T11:40:17.911898+00:00"}