{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:7GFOWEDV4KVRJ2TGEKSKIFNJ4M","short_pith_number":"pith:7GFOWEDV","schema_version":"1.0","canonical_sha256":"f98aeb1075e2ab14ea6622a4a415a9e323a298099576902f801a5bc9fb4ef5e4","source":{"kind":"arxiv","id":"2212.00280","version":1},"attestation_state":"computed","paper":{"title":"GRiT: A Generative Region-to-text Transformer for Object Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jialian Wu, Jianfeng Wang, Junsong Yuan, Lijuan Wang, Zhe Gan, Zhengyuan Yang, Zicheng Liu","submitted_at":"2022-12-01T04:59:44Z","abstract_excerpt":"This paper presents a Generative RegIon-to-Text transformer, GRiT, for object understanding. The spirit of GRiT is to formulate object understanding as <region, text> pairs, where region locates objects and text describes objects. For example, the text in object detection denotes class names while that in dense captioning refers to descriptive sentences. Specifically, GRiT consists of a visual encoder to extract image features, a foreground object extractor to localize objects, and a text decoder to generate open-set object descriptions. With the same model architecture, GRiT can understand ob"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.00280","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-12-01T04:59:44Z","cross_cats_sorted":[],"title_canon_sha256":"a308914e5fbc7c3262425bad27625be5ef040708f541fef1a68261e2a2fa218a","abstract_canon_sha256":"89b9f709dc086577cca92c45a1aeeefd446e22ac935016d0a8cfb72909daff68"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:21:32.809417Z","signature_b64":"9LRugVV8J4pzP1d6ag89vG4Wkp/oMDSyqjxyzJXvq3CL9eC4POzeQzdJvoEg8iyfFJW6iMQCyotlPpw/fCTDDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f98aeb1075e2ab14ea6622a4a415a9e323a298099576902f801a5bc9fb4ef5e4","last_reissued_at":"2026-07-05T05:21:32.809004Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:21:32.809004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GRiT: A Generative Region-to-text Transformer for Object Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jialian Wu, Jianfeng Wang, Junsong Yuan, Lijuan Wang, Zhe Gan, Zhengyuan Yang, Zicheng Liu","submitted_at":"2022-12-01T04:59:44Z","abstract_excerpt":"This paper presents a Generative RegIon-to-Text transformer, GRiT, for object understanding. The spirit of GRiT is to formulate object understanding as <region, text> pairs, where region locates objects and text describes objects. For example, the text in object detection denotes class names while that in dense captioning refers to descriptive sentences. Specifically, GRiT consists of a visual encoder to extract image features, a foreground object extractor to localize objects, and a text decoder to generate open-set object descriptions. With the same model architecture, GRiT can understand ob"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.00280","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.00280/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.00280","created_at":"2026-07-05T05:21:32.809059+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.00280v1","created_at":"2026-07-05T05:21:32.809059+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.00280","created_at":"2026-07-05T05:21:32.809059+00:00"},{"alias_kind":"pith_short_12","alias_value":"7GFOWEDV4KVR","created_at":"2026-07-05T05:21:32.809059+00:00"},{"alias_kind":"pith_short_16","alias_value":"7GFOWEDV4KVRJ2TG","created_at":"2026-07-05T05:21:32.809059+00:00"},{"alias_kind":"pith_short_8","alias_value":"7GFOWEDV","created_at":"2026-07-05T05:21:32.809059+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03748","citing_title":"Ultralytics YOLO26: Unified Real-Time End-to-End Vision Models","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2309.17421","citing_title":"The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision)","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2305.06355","citing_title":"VideoChat: Chat-Centric Video Understanding","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2307.16125","citing_title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03475","citing_title":"WorldJen: An End-to-End Multi-Dimensional Benchmark for Generative Video Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03475","citing_title":"WorldJen: An End-to-End Multi-Dimensional Benchmark for Generative Video Models","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7GFOWEDV4KVRJ2TGEKSKIFNJ4M","json":"https://pith.science/pith/7GFOWEDV4KVRJ2TGEKSKIFNJ4M.json","graph_json":"https://pith.science/api/pith-number/7GFOWEDV4KVRJ2TGEKSKIFNJ4M/graph.json","events_json":"https://pith.science/api/pith-number/7GFOWEDV4KVRJ2TGEKSKIFNJ4M/events.json","paper":"https://pith.science/paper/7GFOWEDV"},"agent_actions":{"view_html":"https://pith.science/pith/7GFOWEDV4KVRJ2TGEKSKIFNJ4M","download_json":"https://pith.science/pith/7GFOWEDV4KVRJ2TGEKSKIFNJ4M.json","view_paper":"https://pith.science/paper/7GFOWEDV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.00280&json=true","fetch_graph":"https://pith.science/api/pith-number/7GFOWEDV4KVRJ2TGEKSKIFNJ4M/graph.json","fetch_events":"https://pith.science/api/pith-number/7GFOWEDV4KVRJ2TGEKSKIFNJ4M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7GFOWEDV4KVRJ2TGEKSKIFNJ4M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7GFOWEDV4KVRJ2TGEKSKIFNJ4M/action/storage_attestation","attest_author":"https://pith.science/pith/7GFOWEDV4KVRJ2TGEKSKIFNJ4M/action/author_attestation","sign_citation":"https://pith.science/pith/7GFOWEDV4KVRJ2TGEKSKIFNJ4M/action/citation_signature","submit_replication":"https://pith.science/pith/7GFOWEDV4KVRJ2TGEKSKIFNJ4M/action/replication_record"}},"created_at":"2026-07-05T05:21:32.809059+00:00","updated_at":"2026-07-05T05:21:32.809059+00:00"}