{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WUCOJB7TFUXCCYU4FAVWQRF5CT","short_pith_number":"pith:WUCOJB7T","schema_version":"1.0","canonical_sha256":"b504e487f32d2e21629c282b6844bd14d0d4ec7321578afad99a9a86939eb813","source":{"kind":"arxiv","id":"2305.14167","version":2},"attestation_state":"computed","paper":{"title":"DetGPT: Detect What You Need via Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Hang Xu, HanZe Dong, Jiahui Gao, Jianhua Han, Jipeng Zhang, Lewei Yao, Lingpeng Kong, Renjie Pi, Rui Pan, Shizhe Diao, Tong Zhang","submitted_at":"2023-05-23T15:37:28Z","abstract_excerpt":"In recent years, the field of computer vision has seen significant advancements thanks to the development of large language models (LLMs). These models have enabled more effective and sophisticated interactions between humans and machines, paving the way for novel techniques that blur the lines between human and machine intelligence. In this paper, we introduce a new paradigm for object detection that we call reasoning-based object detection. Unlike conventional object detection methods that rely on specific object names, our approach enables users to interact with the system using natural lan"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14167","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-05-23T15:37:28Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"42ad010e88e884b1e974e92bc95b0ade7afe2746376799c8aa5af69b64a1ffe8","abstract_canon_sha256":"1fd243424482c9728870d246c3388c818762b1f8f3c736f3d0720b7743acf8aa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:13:22.039531Z","signature_b64":"Pf8vO5OA+Fft828wY/Fx0hSJwSzFXKYwOPh5267csDNXmWCHo6vQk5A78HpdF0X519jM2s0LP7cSvi2BX5DlDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b504e487f32d2e21629c282b6844bd14d0d4ec7321578afad99a9a86939eb813","last_reissued_at":"2026-07-05T06:13:22.039050Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:13:22.039050Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DetGPT: Detect What You Need via Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Hang Xu, HanZe Dong, Jiahui Gao, Jianhua Han, Jipeng Zhang, Lewei Yao, Lingpeng Kong, Renjie Pi, Rui Pan, Shizhe Diao, Tong Zhang","submitted_at":"2023-05-23T15:37:28Z","abstract_excerpt":"In recent years, the field of computer vision has seen significant advancements thanks to the development of large language models (LLMs). These models have enabled more effective and sophisticated interactions between humans and machines, paving the way for novel techniques that blur the lines between human and machine intelligence. In this paper, we introduce a new paradigm for object detection that we call reasoning-based object detection. Unlike conventional object detection methods that rely on specific object names, our approach enables users to interact with the system using natural lan"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14167","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14167/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14167","created_at":"2026-07-05T06:13:22.039108+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14167v2","created_at":"2026-07-05T06:13:22.039108+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14167","created_at":"2026-07-05T06:13:22.039108+00:00"},{"alias_kind":"pith_short_12","alias_value":"WUCOJB7TFUXC","created_at":"2026-07-05T06:13:22.039108+00:00"},{"alias_kind":"pith_short_16","alias_value":"WUCOJB7TFUXCCYU4","created_at":"2026-07-05T06:13:22.039108+00:00"},{"alias_kind":"pith_short_8","alias_value":"WUCOJB7T","created_at":"2026-07-05T06:13:22.039108+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2501.19201","citing_title":"Efficient Reasoning with Hidden Thinking","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":283,"is_internal_anchor":false},{"citing_arxiv_id":"2402.03766","citing_title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2502.04326","citing_title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2401.15947","citing_title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WUCOJB7TFUXCCYU4FAVWQRF5CT","json":"https://pith.science/pith/WUCOJB7TFUXCCYU4FAVWQRF5CT.json","graph_json":"https://pith.science/api/pith-number/WUCOJB7TFUXCCYU4FAVWQRF5CT/graph.json","events_json":"https://pith.science/api/pith-number/WUCOJB7TFUXCCYU4FAVWQRF5CT/events.json","paper":"https://pith.science/paper/WUCOJB7T"},"agent_actions":{"view_html":"https://pith.science/pith/WUCOJB7TFUXCCYU4FAVWQRF5CT","download_json":"https://pith.science/pith/WUCOJB7TFUXCCYU4FAVWQRF5CT.json","view_paper":"https://pith.science/paper/WUCOJB7T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14167&json=true","fetch_graph":"https://pith.science/api/pith-number/WUCOJB7TFUXCCYU4FAVWQRF5CT/graph.json","fetch_events":"https://pith.science/api/pith-number/WUCOJB7TFUXCCYU4FAVWQRF5CT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WUCOJB7TFUXCCYU4FAVWQRF5CT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WUCOJB7TFUXCCYU4FAVWQRF5CT/action/storage_attestation","attest_author":"https://pith.science/pith/WUCOJB7TFUXCCYU4FAVWQRF5CT/action/author_attestation","sign_citation":"https://pith.science/pith/WUCOJB7TFUXCCYU4FAVWQRF5CT/action/citation_signature","submit_replication":"https://pith.science/pith/WUCOJB7TFUXCCYU4FAVWQRF5CT/action/replication_record"}},"created_at":"2026-07-05T06:13:22.039108+00:00","updated_at":"2026-07-05T06:13:22.039108+00:00"}