{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ELWRHYLELAS3OMBDZQALLRBTTN","short_pith_number":"pith:ELWRHYLE","schema_version":"1.0","canonical_sha256":"22ed13e1645825b73023cc00b5c4339b6de9f9e44f7a9b5626f3601991d93918","source":{"kind":"arxiv","id":"2402.05937","version":3},"attestation_state":"computed","paper":{"title":"InstaGen: Enhancing Object Detection by Training on Synthetic Dataset","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengjian Feng, Lin Ma, Weidi Xie, Yujie Zhong, Zequn Jie","submitted_at":"2024-02-08T18:59:53Z","abstract_excerpt":"In this paper, we present a novel paradigm to enhance the ability of object detector, e.g., expanding categories or improving detection performance, by training on synthetic dataset generated from diffusion models. Specifically, we integrate an instance-level grounding head into a pre-trained, generative diffusion model, to augment it with the ability of localising instances in the generated images. The grounding head is trained to align the text embedding of category names with the regional visual feature of the diffusion model, using supervision from an off-the-shelf object detector, and a n"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05937","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-02-08T18:59:53Z","cross_cats_sorted":[],"title_canon_sha256":"258f296d713fd2591827e55be0db6e2e24edcdb0b3d95fc9331fe0005c4de547","abstract_canon_sha256":"cd44daefee9557886df89f70c46c889d87eaf3f435deee0b7e17387843cbfa82"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:10.245922Z","signature_b64":"JIJVJaqzZny1OaGywJ1Hi8gQvU6eHhuZM/ZomCDdU9933+QWiDnWk5SAl43GNqK/xyYxlmuJaMvjWiA0Nbg1AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"22ed13e1645825b73023cc00b5c4339b6de9f9e44f7a9b5626f3601991d93918","last_reissued_at":"2026-07-05T08:05:10.245428Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:10.245428Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InstaGen: Enhancing Object Detection by Training on Synthetic Dataset","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengjian Feng, Lin Ma, Weidi Xie, Yujie Zhong, Zequn Jie","submitted_at":"2024-02-08T18:59:53Z","abstract_excerpt":"In this paper, we present a novel paradigm to enhance the ability of object detector, e.g., expanding categories or improving detection performance, by training on synthetic dataset generated from diffusion models. Specifically, we integrate an instance-level grounding head into a pre-trained, generative diffusion model, to augment it with the ability of localising instances in the generated images. The grounding head is trained to align the text embedding of category names with the regional visual feature of the diffusion model, using supervision from an off-the-shelf object detector, and a n"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05937","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05937/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05937","created_at":"2026-07-05T08:05:10.245482+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05937v3","created_at":"2026-07-05T08:05:10.245482+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05937","created_at":"2026-07-05T08:05:10.245482+00:00"},{"alias_kind":"pith_short_12","alias_value":"ELWRHYLELAS3","created_at":"2026-07-05T08:05:10.245482+00:00"},{"alias_kind":"pith_short_16","alias_value":"ELWRHYLELAS3OMBD","created_at":"2026-07-05T08:05:10.245482+00:00"},{"alias_kind":"pith_short_8","alias_value":"ELWRHYLE","created_at":"2026-07-05T08:05:10.245482+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04955","citing_title":"EXPOTION: Facial Expression and Motion Control for Multimodal Music Generation","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ELWRHYLELAS3OMBDZQALLRBTTN","json":"https://pith.science/pith/ELWRHYLELAS3OMBDZQALLRBTTN.json","graph_json":"https://pith.science/api/pith-number/ELWRHYLELAS3OMBDZQALLRBTTN/graph.json","events_json":"https://pith.science/api/pith-number/ELWRHYLELAS3OMBDZQALLRBTTN/events.json","paper":"https://pith.science/paper/ELWRHYLE"},"agent_actions":{"view_html":"https://pith.science/pith/ELWRHYLELAS3OMBDZQALLRBTTN","download_json":"https://pith.science/pith/ELWRHYLELAS3OMBDZQALLRBTTN.json","view_paper":"https://pith.science/paper/ELWRHYLE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05937&json=true","fetch_graph":"https://pith.science/api/pith-number/ELWRHYLELAS3OMBDZQALLRBTTN/graph.json","fetch_events":"https://pith.science/api/pith-number/ELWRHYLELAS3OMBDZQALLRBTTN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ELWRHYLELAS3OMBDZQALLRBTTN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ELWRHYLELAS3OMBDZQALLRBTTN/action/storage_attestation","attest_author":"https://pith.science/pith/ELWRHYLELAS3OMBDZQALLRBTTN/action/author_attestation","sign_citation":"https://pith.science/pith/ELWRHYLELAS3OMBDZQALLRBTTN/action/citation_signature","submit_replication":"https://pith.science/pith/ELWRHYLELAS3OMBDZQALLRBTTN/action/replication_record"}},"created_at":"2026-07-05T08:05:10.245482+00:00","updated_at":"2026-07-05T08:05:10.245482+00:00"}