{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F6GKOEWG2TCEE4LWD5LATFVTII","short_pith_number":"pith:F6GKOEWG","schema_version":"1.0","canonical_sha256":"2f8ca712c6d4c44271761f560996b34226e41e416823569439535c7da24f029e","source":{"kind":"arxiv","id":"2403.14610","version":1},"attestation_state":"computed","paper":{"title":"T-Rex2: Towards Generic Object Detection via Text-Visual Prompt Synergy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feng Li, Lei Zhang, Qing Jiang, Shilong Liu, Tianhe Ren, Zhaoyang Zeng","submitted_at":"2024-03-21T17:57:03Z","abstract_excerpt":"We present T-Rex2, a highly practical model for open-set object detection. Previous open-set object detection methods relying on text prompts effectively encapsulate the abstract concept of common objects, but struggle with rare or complex object representation due to data scarcity and descriptive limitations. Conversely, visual prompts excel in depicting novel objects through concrete visual examples, but fall short in conveying the abstract concept of objects as effectively as text prompts. Recognizing the complementary strengths and weaknesses of both text and visual prompts, we introduce T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.14610","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-21T17:57:03Z","cross_cats_sorted":[],"title_canon_sha256":"cb3ed7e11169fb28a50b5bc4378dbbcff6b1406593c88f4c3d01cd1b2c3b7df5","abstract_canon_sha256":"e90412e33e26e1054f8c2c9b03716d7b3f2b9d956ec13404db4844f868cf18e8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:59:11.342628Z","signature_b64":"QZ+BMhjcC9jft/Lbjn5IUVxzcgjmp7fV4naoRFXMUXx4HQOKzBP+c/sgBI4u5RnXVB4QxpL55bp94Hvxeve6AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f8ca712c6d4c44271761f560996b34226e41e416823569439535c7da24f029e","last_reissued_at":"2026-07-05T07:59:11.342189Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:59:11.342189Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"T-Rex2: Towards Generic Object Detection via Text-Visual Prompt Synergy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feng Li, Lei Zhang, Qing Jiang, Shilong Liu, Tianhe Ren, Zhaoyang Zeng","submitted_at":"2024-03-21T17:57:03Z","abstract_excerpt":"We present T-Rex2, a highly practical model for open-set object detection. Previous open-set object detection methods relying on text prompts effectively encapsulate the abstract concept of common objects, but struggle with rare or complex object representation due to data scarcity and descriptive limitations. Conversely, visual prompts excel in depicting novel objects through concrete visual examples, but fall short in conveying the abstract concept of objects as effectively as text prompts. Recognizing the complementary strengths and weaknesses of both text and visual prompts, we introduce T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.14610","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.14610/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.14610","created_at":"2026-07-05T07:59:11.342247+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.14610v1","created_at":"2026-07-05T07:59:11.342247+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.14610","created_at":"2026-07-05T07:59:11.342247+00:00"},{"alias_kind":"pith_short_12","alias_value":"F6GKOEWG2TCE","created_at":"2026-07-05T07:59:11.342247+00:00"},{"alias_kind":"pith_short_16","alias_value":"F6GKOEWG2TCEE4LW","created_at":"2026-07-05T07:59:11.342247+00:00"},{"alias_kind":"pith_short_8","alias_value":"F6GKOEWG","created_at":"2026-07-05T07:59:11.342247+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08541","citing_title":"VocaDet: Sample-Driven Open-Vocabulary Object Detection and Segmentation via Visual Tokenization and Vector Database Retrieval","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03748","citing_title":"Ultralytics YOLO26: Unified Real-Time End-to-End Vision Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18455","citing_title":"Learning Geometry-Aware Nonprehensile Pushing and Pulling with Dexterous Hands","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F6GKOEWG2TCEE4LWD5LATFVTII","json":"https://pith.science/pith/F6GKOEWG2TCEE4LWD5LATFVTII.json","graph_json":"https://pith.science/api/pith-number/F6GKOEWG2TCEE4LWD5LATFVTII/graph.json","events_json":"https://pith.science/api/pith-number/F6GKOEWG2TCEE4LWD5LATFVTII/events.json","paper":"https://pith.science/paper/F6GKOEWG"},"agent_actions":{"view_html":"https://pith.science/pith/F6GKOEWG2TCEE4LWD5LATFVTII","download_json":"https://pith.science/pith/F6GKOEWG2TCEE4LWD5LATFVTII.json","view_paper":"https://pith.science/paper/F6GKOEWG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.14610&json=true","fetch_graph":"https://pith.science/api/pith-number/F6GKOEWG2TCEE4LWD5LATFVTII/graph.json","fetch_events":"https://pith.science/api/pith-number/F6GKOEWG2TCEE4LWD5LATFVTII/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F6GKOEWG2TCEE4LWD5LATFVTII/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F6GKOEWG2TCEE4LWD5LATFVTII/action/storage_attestation","attest_author":"https://pith.science/pith/F6GKOEWG2TCEE4LWD5LATFVTII/action/author_attestation","sign_citation":"https://pith.science/pith/F6GKOEWG2TCEE4LWD5LATFVTII/action/citation_signature","submit_replication":"https://pith.science/pith/F6GKOEWG2TCEE4LWD5LATFVTII/action/replication_record"}},"created_at":"2026-07-05T07:59:11.342247+00:00","updated_at":"2026-07-05T07:59:11.342247+00:00"}