{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2016:WCJMRLKK2FN5WSCHK7IUZTPIPX","short_pith_number":"pith:WCJMRLKK","schema_version":"1.0","canonical_sha256":"b092c8ad4ad15bdb484757d14ccde87deae488e94c89cbded258a8931b97456d","source":{"kind":"arxiv","id":"1608.07639","version":1},"attestation_state":"computed","paper":{"title":"Learning to generalize to new compositions in image understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Amir Globerson, Gal Chechik, Jonathan Berant, Vahid Kezami, Yuval Atzmon","submitted_at":"2016-08-27T00:34:00Z","abstract_excerpt":"Recurrent neural networks have recently been used for learning to describe images using natural language. However, it has been observed that these models generalize poorly to scenes that were not observed during training, possibly depending too strongly on the statistics of the text in the training data. Here we propose to describe images using short structured representations, aiming to capture the crux of a description. These structured representations allow us to tease-out and evaluate separately two types of generalization: standard generalization to new images with similar scenes, and gen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1608.07639","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2016-08-27T00:34:00Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"710bb764777493e39380a6e01286fe1f6a8f7ad6669356b26ce8984899ed5245","abstract_canon_sha256":"fc9fab04d359292d964af4e337b1b1c050f84e3cb6902b2f10e7c3252db9fd8b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T01:07:50.207660Z","signature_b64":"0ma05GW0jbG41BSvbj/Zis2Dr+cZoi57y/wvAI7uRUODrF96z9zDS/55UmIdbbjpuAAFSd03uWfLp8rU2FBmAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b092c8ad4ad15bdb484757d14ccde87deae488e94c89cbded258a8931b97456d","last_reissued_at":"2026-05-18T01:07:50.207009Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T01:07:50.207009Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to generalize to new compositions in image understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Amir Globerson, Gal Chechik, Jonathan Berant, Vahid Kezami, Yuval Atzmon","submitted_at":"2016-08-27T00:34:00Z","abstract_excerpt":"Recurrent neural networks have recently been used for learning to describe images using natural language. However, it has been observed that these models generalize poorly to scenes that were not observed during training, possibly depending too strongly on the statistics of the text in the training data. Here we propose to describe images using short structured representations, aiming to capture the crux of a description. These structured representations allow us to tease-out and evaluate separately two types of generalization: standard generalization to new images with similar scenes, and gen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1608.07639","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1608.07639","created_at":"2026-05-18T01:07:50.207095+00:00"},{"alias_kind":"arxiv_version","alias_value":"1608.07639v1","created_at":"2026-05-18T01:07:50.207095+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1608.07639","created_at":"2026-05-18T01:07:50.207095+00:00"},{"alias_kind":"pith_short_12","alias_value":"WCJMRLKK2FN5","created_at":"2026-05-18T12:30:48.956258+00:00"},{"alias_kind":"pith_short_16","alias_value":"WCJMRLKK2FN5WSCH","created_at":"2026-05-18T12:30:48.956258+00:00"},{"alias_kind":"pith_short_8","alias_value":"WCJMRLKK","created_at":"2026-05-18T12:30:48.956258+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.20986","citing_title":"EVA: Mixture-of-Experts Semantic Variant Alignment for Compositional Zero-Shot Learning","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WCJMRLKK2FN5WSCHK7IUZTPIPX","json":"https://pith.science/pith/WCJMRLKK2FN5WSCHK7IUZTPIPX.json","graph_json":"https://pith.science/api/pith-number/WCJMRLKK2FN5WSCHK7IUZTPIPX/graph.json","events_json":"https://pith.science/api/pith-number/WCJMRLKK2FN5WSCHK7IUZTPIPX/events.json","paper":"https://pith.science/paper/WCJMRLKK"},"agent_actions":{"view_html":"https://pith.science/pith/WCJMRLKK2FN5WSCHK7IUZTPIPX","download_json":"https://pith.science/pith/WCJMRLKK2FN5WSCHK7IUZTPIPX.json","view_paper":"https://pith.science/paper/WCJMRLKK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1608.07639&json=true","fetch_graph":"https://pith.science/api/pith-number/WCJMRLKK2FN5WSCHK7IUZTPIPX/graph.json","fetch_events":"https://pith.science/api/pith-number/WCJMRLKK2FN5WSCHK7IUZTPIPX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WCJMRLKK2FN5WSCHK7IUZTPIPX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WCJMRLKK2FN5WSCHK7IUZTPIPX/action/storage_attestation","attest_author":"https://pith.science/pith/WCJMRLKK2FN5WSCHK7IUZTPIPX/action/author_attestation","sign_citation":"https://pith.science/pith/WCJMRLKK2FN5WSCHK7IUZTPIPX/action/citation_signature","submit_replication":"https://pith.science/pith/WCJMRLKK2FN5WSCHK7IUZTPIPX/action/replication_record"}},"created_at":"2026-05-18T01:07:50.207095+00:00","updated_at":"2026-05-18T01:07:50.207095+00:00"}