{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XNA7ASNA23FEFPTW744RMCKIH6","short_pith_number":"pith:XNA7ASNA","schema_version":"1.0","canonical_sha256":"bb41f049a0d6ca42be76ff391609483fa8a9815f5c3235654f0518d2e629b7e8","source":{"kind":"arxiv","id":"2402.01622","version":4},"attestation_state":"computed","paper":{"title":"TravelPlanner: A Benchmark for Real-World Planning with Language Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiangjie Chen, Jian Xie, Kai Zhang, Renze Lou, Tinghui Zhu, Yanghua Xiao, Yuandong Tian, Yu Su","submitted_at":"2024-02-02T18:39:51Z","abstract_excerpt":"Planning has been part of the core pursuit for artificial intelligence since its conception, but earlier AI agents mostly focused on constrained settings because many of the cognitive substrates necessary for human-level planning have been lacking. Recently, language agents powered by large language models (LLMs) have shown interesting capabilities such as tool use and reasoning. Are these language agents capable of planning in more complex settings that are out of the reach of prior AI agents? To advance this investigation, we propose TravelPlanner, a new planning benchmark that focuses on tr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.01622","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-02T18:39:51Z","cross_cats_sorted":[],"title_canon_sha256":"c7ed449dda40f81eca15f0498d598e7d4957d095108781fef73899324851d1ea","abstract_canon_sha256":"68d3228b60e9b6758af5cf6de3881f61e22221ccbf003de95fb108bf1403a825"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:32.534731Z","signature_b64":"GbuG/jFqYHP7vI4q8aMlCgOiUKNKWSnxXA8boHOyjp3xlYgFYO5T5lPAfRmaQFzeGrLL6FXhtEcwyTtxLUFFAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb41f049a0d6ca42be76ff391609483fa8a9815f5c3235654f0518d2e629b7e8","last_reissued_at":"2026-07-05T09:24:32.534227Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:32.534227Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TravelPlanner: A Benchmark for Real-World Planning with Language Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiangjie Chen, Jian Xie, Kai Zhang, Renze Lou, Tinghui Zhu, Yanghua Xiao, Yuandong Tian, Yu Su","submitted_at":"2024-02-02T18:39:51Z","abstract_excerpt":"Planning has been part of the core pursuit for artificial intelligence since its conception, but earlier AI agents mostly focused on constrained settings because many of the cognitive substrates necessary for human-level planning have been lacking. Recently, language agents powered by large language models (LLMs) have shown interesting capabilities such as tool use and reasoning. Are these language agents capable of planning in more complex settings that are out of the reach of prior AI agents? To advance this investigation, we propose TravelPlanner, a new planning benchmark that focuses on tr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.01622","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.01622/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.01622","created_at":"2026-07-05T09:24:32.534287+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.01622v4","created_at":"2026-07-05T09:24:32.534287+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.01622","created_at":"2026-07-05T09:24:32.534287+00:00"},{"alias_kind":"pith_short_12","alias_value":"XNA7ASNA23FE","created_at":"2026-07-05T09:24:32.534287+00:00"},{"alias_kind":"pith_short_16","alias_value":"XNA7ASNA23FEFPTW","created_at":"2026-07-05T09:24:32.534287+00:00"},{"alias_kind":"pith_short_8","alias_value":"XNA7ASNA","created_at":"2026-07-05T09:24:32.534287+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23664","citing_title":"MAS-PromptBench: When Does Prompt Optimization Improve Multi-Agent LLM Systems?","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21169","citing_title":"Trip+: Benchmarking Agents in Personalized Interactive Travel Planning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19613","citing_title":"StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18910","citing_title":"REVES: REvision and VErification--Augmented Training for Test-Time Scaling","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08068","citing_title":"DICE: Entropy-Regularized Equilibrium Selection for Stable Multi-Agent LLM Coordination","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07486","citing_title":"OPENPATH: A Supervisor--Specialist Agent System for Personalized, Accessible, and Multi-stop Urban Trip Planning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04315","citing_title":"Exploring Cross-Scenario Generality of Agentic Memory Systems: Diagnostics and a Strong Baseline","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20256","citing_title":"FBOS-RL: Feedback-Driven Bi-Objective Synergistic Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01046","citing_title":"TravelEval: A Comprehensive Benchmarking Framework for Evaluating LLM-Powered Travel Planning Agents","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2502.16810","citing_title":"AI Realtor: Towards Grounded Persuasive Language Generation for Automated Copywriting","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21460","citing_title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16120","citing_title":"LLM-Powered AI Agent Systems and Their Applications in Industry","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20256","citing_title":"FBOS-RL: Feedback-Driven Bi-Objective Synergistic Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17891","citing_title":"Scaling Diffusion Language Models via Adaptation from Autoregressive Models","ref_index":194,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17829","citing_title":"Interactive Evaluation Requires a Design Science","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2506.00886","citing_title":"Position: Agent Should Invoke External Tools ONLY When Epistemically Necessary","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2509.12626","citing_title":"DoubleAgents: Human-Agent Alignment in a Socially Embedded Workflow","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2510.05307","citing_title":"When Should Users Check? Modeling Confirmation Frequency inMulti-Step Agentic AI Tasks","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2510.07043","citing_title":"COMPASS: Benchmarking Constrained Optimization in LLM Agents","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2602.05048","citing_title":"MINT: Minimal Information Neuro-Symbolic Tree for Objective-Driven Knowledge-Gap Reasoning and Active Elicitation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2603.00977","citing_title":"HiMAC: Hierarchical Macro-Micro Learning for Long-Horizon LLM Agents","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12755","citing_title":"State-Centric Decision Process","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10782","citing_title":"TrajPrism: A Multi-Task Benchmark for Language-Grounded Urban Trajectory Understanding","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10440","citing_title":"TourMart: A Parametric Audit Instrument for Commission Steering in LLM Travel Agents","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00276","citing_title":"Agentic AI for Trip Planning Optimization Application","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XNA7ASNA23FEFPTW744RMCKIH6","json":"https://pith.science/pith/XNA7ASNA23FEFPTW744RMCKIH6.json","graph_json":"https://pith.science/api/pith-number/XNA7ASNA23FEFPTW744RMCKIH6/graph.json","events_json":"https://pith.science/api/pith-number/XNA7ASNA23FEFPTW744RMCKIH6/events.json","paper":"https://pith.science/paper/XNA7ASNA"},"agent_actions":{"view_html":"https://pith.science/pith/XNA7ASNA23FEFPTW744RMCKIH6","download_json":"https://pith.science/pith/XNA7ASNA23FEFPTW744RMCKIH6.json","view_paper":"https://pith.science/paper/XNA7ASNA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.01622&json=true","fetch_graph":"https://pith.science/api/pith-number/XNA7ASNA23FEFPTW744RMCKIH6/graph.json","fetch_events":"https://pith.science/api/pith-number/XNA7ASNA23FEFPTW744RMCKIH6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XNA7ASNA23FEFPTW744RMCKIH6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XNA7ASNA23FEFPTW744RMCKIH6/action/storage_attestation","attest_author":"https://pith.science/pith/XNA7ASNA23FEFPTW744RMCKIH6/action/author_attestation","sign_citation":"https://pith.science/pith/XNA7ASNA23FEFPTW744RMCKIH6/action/citation_signature","submit_replication":"https://pith.science/pith/XNA7ASNA23FEFPTW744RMCKIH6/action/replication_record"}},"created_at":"2026-07-05T09:24:32.534287+00:00","updated_at":"2026-07-05T09:24:32.534287+00:00"}