{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:AKCM3NT2NILHEVNEUGJU3OO3TR","short_pith_number":"pith:AKCM3NT2","schema_version":"1.0","canonical_sha256":"0284cdb67a6a167255a4a1934db9db9c651367017c0c1e9da649d515ed0c0ae8","source":{"kind":"arxiv","id":"2111.12085","version":2},"attestation_state":"computed","paper":{"title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Faisal Ahmed, Jianfeng Wang, Lijuan Wang, Xiaowei Hu, Yumao Lu, Zhe Gan, Zhengyuan Yang, Zicheng Liu","submitted_at":"2021-11-23T18:59:14Z","abstract_excerpt":"We propose UniTAB that Unifies Text And Box outputs for grounded vision-language (VL) modeling. Grounded VL tasks such as grounded captioning require the model to generate a text description and align predicted words with object regions. To achieve this, models must generate desired text and box outputs together, and meanwhile indicate the alignments between words and boxes. In contrast to existing solutions that use multiple separate modules for different outputs, UniTAB represents both text and box outputs with a shared token sequence, and introduces a special <obj> token to naturally indica"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.12085","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-11-23T18:59:14Z","cross_cats_sorted":[],"title_canon_sha256":"72847c44bf8eb807f4aaf9eb937b64bbe5232ae53b21b1d8a3d6ec28f1b8ea65","abstract_canon_sha256":"e49bb09acc3422cf617f754375c98ec589a2b6ce638314c1dbed490b10713b3c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:43:58.374658Z","signature_b64":"Wz3p0viiEMWNJlsrXzjgyGXnoPofZ0B063x92qoLeK8gSYBR5EXAyhTSLnpZp9AX9lFtbxjhSDpewikf2+VLCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0284cdb67a6a167255a4a1934db9db9c651367017c0c1e9da649d515ed0c0ae8","last_reissued_at":"2026-07-05T04:43:58.374099Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:43:58.374099Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Faisal Ahmed, Jianfeng Wang, Lijuan Wang, Xiaowei Hu, Yumao Lu, Zhe Gan, Zhengyuan Yang, Zicheng Liu","submitted_at":"2021-11-23T18:59:14Z","abstract_excerpt":"We propose UniTAB that Unifies Text And Box outputs for grounded vision-language (VL) modeling. Grounded VL tasks such as grounded captioning require the model to generate a text description and align predicted words with object regions. To achieve this, models must generate desired text and box outputs together, and meanwhile indicate the alignments between words and boxes. In contrast to existing solutions that use multiple separate modules for different outputs, UniTAB represents both text and box outputs with a shared token sequence, and introduces a special <obj> token to naturally indica"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.12085","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.12085/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.12085","created_at":"2026-07-05T04:43:58.374158+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.12085v2","created_at":"2026-07-05T04:43:58.374158+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.12085","created_at":"2026-07-05T04:43:58.374158+00:00"},{"alias_kind":"pith_short_12","alias_value":"AKCM3NT2NILH","created_at":"2026-07-05T04:43:58.374158+00:00"},{"alias_kind":"pith_short_16","alias_value":"AKCM3NT2NILHEVNE","created_at":"2026-07-05T04:43:58.374158+00:00"},{"alias_kind":"pith_short_8","alias_value":"AKCM3NT2","created_at":"2026-07-05T04:43:58.374158+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.10026","citing_title":"LaV-CoT: Language-Aware Visual CoT with Multi-Aspect Reward Optimization for Real-World Multilingual VQA","ref_index":68,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AKCM3NT2NILHEVNEUGJU3OO3TR","json":"https://pith.science/pith/AKCM3NT2NILHEVNEUGJU3OO3TR.json","graph_json":"https://pith.science/api/pith-number/AKCM3NT2NILHEVNEUGJU3OO3TR/graph.json","events_json":"https://pith.science/api/pith-number/AKCM3NT2NILHEVNEUGJU3OO3TR/events.json","paper":"https://pith.science/paper/AKCM3NT2"},"agent_actions":{"view_html":"https://pith.science/pith/AKCM3NT2NILHEVNEUGJU3OO3TR","download_json":"https://pith.science/pith/AKCM3NT2NILHEVNEUGJU3OO3TR.json","view_paper":"https://pith.science/paper/AKCM3NT2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.12085&json=true","fetch_graph":"https://pith.science/api/pith-number/AKCM3NT2NILHEVNEUGJU3OO3TR/graph.json","fetch_events":"https://pith.science/api/pith-number/AKCM3NT2NILHEVNEUGJU3OO3TR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AKCM3NT2NILHEVNEUGJU3OO3TR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AKCM3NT2NILHEVNEUGJU3OO3TR/action/storage_attestation","attest_author":"https://pith.science/pith/AKCM3NT2NILHEVNEUGJU3OO3TR/action/author_attestation","sign_citation":"https://pith.science/pith/AKCM3NT2NILHEVNEUGJU3OO3TR/action/citation_signature","submit_replication":"https://pith.science/pith/AKCM3NT2NILHEVNEUGJU3OO3TR/action/replication_record"}},"created_at":"2026-07-05T04:43:58.374158+00:00","updated_at":"2026-07-05T04:43:58.374158+00:00"}