{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:UARLCJZ5SZK4VUG6YOVKBIRXJI","short_pith_number":"pith:UARLCJZ5","schema_version":"1.0","canonical_sha256":"a022b1273d9655cad0dec3aaa0a2374a3771c8d2004895134300114af5f7bf58","source":{"kind":"arxiv","id":"2106.09707","version":1},"attestation_state":"computed","paper":{"title":"Learning to Predict Visual Attributes in the Wild","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abhinav Shrivastava, Khoi Pham, Kushal Kafle, Quan Tran, Scott Cohen, Zhe Lin, Zhihong Ding","submitted_at":"2021-06-17T17:58:02Z","abstract_excerpt":"Visual attributes constitute a large portion of information contained in a scene. Objects can be described using a wide variety of attributes which portray their visual appearance (color, texture), geometry (shape, size, posture), and other intrinsic properties (state, action). Existing work is mostly limited to study of attribute prediction in specific domains. In this paper, we introduce a large-scale in-the-wild visual attribute prediction dataset consisting of over 927K attribute annotations for over 260K object instances. Formally, object attribute prediction is a multi-label classificati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.09707","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-06-17T17:58:02Z","cross_cats_sorted":[],"title_canon_sha256":"ddea4663bb03d54bb8c05708675328e41e2accc9ad5782cbf0d83143de393c7e","abstract_canon_sha256":"d1eb4433e1392cdeb0e6fab673e3edc595a4dcb04bb08cae2daa17a0797607ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:50:18.915293Z","signature_b64":"g1nh0Zci4VqnP9zMvO8NhEtTYgMF0Hd76k4AgW1MSDPaySxp7dC385ql7JyvQELhQZEL6tkfI0jQPy7jL0YnAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a022b1273d9655cad0dec3aaa0a2374a3771c8d2004895134300114af5f7bf58","last_reissued_at":"2026-07-05T02:50:18.914756Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:50:18.914756Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Predict Visual Attributes in the Wild","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abhinav Shrivastava, Khoi Pham, Kushal Kafle, Quan Tran, Scott Cohen, Zhe Lin, Zhihong Ding","submitted_at":"2021-06-17T17:58:02Z","abstract_excerpt":"Visual attributes constitute a large portion of information contained in a scene. Objects can be described using a wide variety of attributes which portray their visual appearance (color, texture), geometry (shape, size, posture), and other intrinsic properties (state, action). Existing work is mostly limited to study of attribute prediction in specific domains. In this paper, we introduce a large-scale in-the-wild visual attribute prediction dataset consisting of over 927K attribute annotations for over 260K object instances. Formally, object attribute prediction is a multi-label classificati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.09707","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.09707/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.09707","created_at":"2026-07-05T02:50:18.914820+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.09707v1","created_at":"2026-07-05T02:50:18.914820+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.09707","created_at":"2026-07-05T02:50:18.914820+00:00"},{"alias_kind":"pith_short_12","alias_value":"UARLCJZ5SZK4","created_at":"2026-07-05T02:50:18.914820+00:00"},{"alias_kind":"pith_short_16","alias_value":"UARLCJZ5SZK4VUG6","created_at":"2026-07-05T02:50:18.914820+00:00"},{"alias_kind":"pith_short_8","alias_value":"UARLCJZ5","created_at":"2026-07-05T02:50:18.914820+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.18695","citing_title":"Attributes Should Come from Images, Not Class Names: Distribution-Conditioned Attribute Selection for Vision-Language Models","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UARLCJZ5SZK4VUG6YOVKBIRXJI","json":"https://pith.science/pith/UARLCJZ5SZK4VUG6YOVKBIRXJI.json","graph_json":"https://pith.science/api/pith-number/UARLCJZ5SZK4VUG6YOVKBIRXJI/graph.json","events_json":"https://pith.science/api/pith-number/UARLCJZ5SZK4VUG6YOVKBIRXJI/events.json","paper":"https://pith.science/paper/UARLCJZ5"},"agent_actions":{"view_html":"https://pith.science/pith/UARLCJZ5SZK4VUG6YOVKBIRXJI","download_json":"https://pith.science/pith/UARLCJZ5SZK4VUG6YOVKBIRXJI.json","view_paper":"https://pith.science/paper/UARLCJZ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.09707&json=true","fetch_graph":"https://pith.science/api/pith-number/UARLCJZ5SZK4VUG6YOVKBIRXJI/graph.json","fetch_events":"https://pith.science/api/pith-number/UARLCJZ5SZK4VUG6YOVKBIRXJI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UARLCJZ5SZK4VUG6YOVKBIRXJI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UARLCJZ5SZK4VUG6YOVKBIRXJI/action/storage_attestation","attest_author":"https://pith.science/pith/UARLCJZ5SZK4VUG6YOVKBIRXJI/action/author_attestation","sign_citation":"https://pith.science/pith/UARLCJZ5SZK4VUG6YOVKBIRXJI/action/citation_signature","submit_replication":"https://pith.science/pith/UARLCJZ5SZK4VUG6YOVKBIRXJI/action/replication_record"}},"created_at":"2026-07-05T02:50:18.914820+00:00","updated_at":"2026-07-05T02:50:18.914820+00:00"}