{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ULWDK4B7AGLYYEW3L5VHPMLRT6","short_pith_number":"pith:ULWDK4B7","schema_version":"1.0","canonical_sha256":"a2ec35703f01978c12db5f6a77b1719fa0ee668acfc64d834fae1c99241b2b71","source":{"kind":"arxiv","id":"2501.13893","version":1},"attestation_state":"computed","paper":{"title":"Pix2Cap-COCO: Advancing Visual Comprehension via Pixel-Level Captioning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bo He, Junke Wang, Lingyu Kong, Zuxuan Wu, Zuyao You","submitted_at":"2025-01-23T18:08:57Z","abstract_excerpt":"We present Pix2Cap-COCO, the first panoptic pixel-level caption dataset designed to advance fine-grained visual understanding. To achieve this, we carefully design an automated annotation pipeline that prompts GPT-4V to generate pixel-aligned, instance-specific captions for individual objects within images, enabling models to learn more granular relationships between objects and their contexts. This approach results in 167,254 detailed captions, with an average of 22.94 words per caption. Building on Pix2Cap-COCO, we introduce a novel task, panoptic segmentation-captioning, which challenges mo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.13893","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-23T18:08:57Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"65dbcc1292bb42c685c692c25797ccf1fe9295e084bac6527a3dbaa58af56a47","abstract_canon_sha256":"b68089dabee83f2ce3356a06325da3c503b7909564b3a26e74cdb21f88c2c4e1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:35.732533Z","signature_b64":"NL4VRNQh9PfSy/3bbcsWg+dgo2itR+ZkBzM+5GFC4BGOyzzRds2uoW8q8ps6zjMHISnpIwBssAgpuyX6RyDQBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2ec35703f01978c12db5f6a77b1719fa0ee668acfc64d834fae1c99241b2b71","last_reissued_at":"2026-07-05T10:04:35.732156Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:35.732156Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pix2Cap-COCO: Advancing Visual Comprehension via Pixel-Level Captioning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bo He, Junke Wang, Lingyu Kong, Zuxuan Wu, Zuyao You","submitted_at":"2025-01-23T18:08:57Z","abstract_excerpt":"We present Pix2Cap-COCO, the first panoptic pixel-level caption dataset designed to advance fine-grained visual understanding. To achieve this, we carefully design an automated annotation pipeline that prompts GPT-4V to generate pixel-aligned, instance-specific captions for individual objects within images, enabling models to learn more granular relationships between objects and their contexts. This approach results in 167,254 detailed captions, with an average of 22.94 words per caption. Building on Pix2Cap-COCO, we introduce a novel task, panoptic segmentation-captioning, which challenges mo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.13893","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.13893/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.13893","created_at":"2026-07-05T10:04:35.732205+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.13893v1","created_at":"2026-07-05T10:04:35.732205+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.13893","created_at":"2026-07-05T10:04:35.732205+00:00"},{"alias_kind":"pith_short_12","alias_value":"ULWDK4B7AGLY","created_at":"2026-07-05T10:04:35.732205+00:00"},{"alias_kind":"pith_short_16","alias_value":"ULWDK4B7AGLYYEW3","created_at":"2026-07-05T10:04:35.732205+00:00"},{"alias_kind":"pith_short_8","alias_value":"ULWDK4B7","created_at":"2026-07-05T10:04:35.732205+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.24102","citing_title":"DenseWorld-1M: Towards Detailed Dense Grounded Caption in the Real World","ref_index":85,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ULWDK4B7AGLYYEW3L5VHPMLRT6","json":"https://pith.science/pith/ULWDK4B7AGLYYEW3L5VHPMLRT6.json","graph_json":"https://pith.science/api/pith-number/ULWDK4B7AGLYYEW3L5VHPMLRT6/graph.json","events_json":"https://pith.science/api/pith-number/ULWDK4B7AGLYYEW3L5VHPMLRT6/events.json","paper":"https://pith.science/paper/ULWDK4B7"},"agent_actions":{"view_html":"https://pith.science/pith/ULWDK4B7AGLYYEW3L5VHPMLRT6","download_json":"https://pith.science/pith/ULWDK4B7AGLYYEW3L5VHPMLRT6.json","view_paper":"https://pith.science/paper/ULWDK4B7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.13893&json=true","fetch_graph":"https://pith.science/api/pith-number/ULWDK4B7AGLYYEW3L5VHPMLRT6/graph.json","fetch_events":"https://pith.science/api/pith-number/ULWDK4B7AGLYYEW3L5VHPMLRT6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ULWDK4B7AGLYYEW3L5VHPMLRT6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ULWDK4B7AGLYYEW3L5VHPMLRT6/action/storage_attestation","attest_author":"https://pith.science/pith/ULWDK4B7AGLYYEW3L5VHPMLRT6/action/author_attestation","sign_citation":"https://pith.science/pith/ULWDK4B7AGLYYEW3L5VHPMLRT6/action/citation_signature","submit_replication":"https://pith.science/pith/ULWDK4B7AGLYYEW3L5VHPMLRT6/action/replication_record"}},"created_at":"2026-07-05T10:04:35.732205+00:00","updated_at":"2026-07-05T10:04:35.732205+00:00"}