{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RCKAOG7GREVFZU6GYFOG5P4UMU","short_pith_number":"pith:RCKAOG7G","schema_version":"1.0","canonical_sha256":"8894071be6892a5cd3c6c15c6ebf946532813b2912983538798f23ee56e8840c","source":{"kind":"arxiv","id":"2410.23775","version":3},"attestation_state":"computed","paper":{"title":"In-Context LoRA for Diffusion Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GR"],"primary_cat":"cs.CV","authors_text":"Chen Liang, Huanzhang Dou, Jingren Zhou, Lianghua Huang, Wei Wang, Yu Liu, Yupeng Shi, Yutong Feng, Zhi-Fan Wu","submitted_at":"2024-10-31T09:45:00Z","abstract_excerpt":"Recent research arXiv:2410.15027 has explored the use of diffusion transformers (DiTs) for task-agnostic image generation by simply concatenating attention tokens across images. However, despite substantial computational resources, the fidelity of the generated images remains suboptimal. In this study, we reevaluate and streamline this framework by hypothesizing that text-to-image DiTs inherently possess in-context generation capabilities, requiring only minimal tuning to activate them. Through diverse task experiments, we qualitatively demonstrate that existing text-to-image DiTs can effectiv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.23775","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-31T09:45:00Z","cross_cats_sorted":["cs.GR"],"title_canon_sha256":"d5765fd9f75b73c73e06c33cb8f5d9cb30ac61f1909bdb388de5e1d5192b4b02","abstract_canon_sha256":"c682f1e644cc4a0b50a97c7b04423885bfaf0279561b589cae17d58e7af020fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:10.307823Z","signature_b64":"vz9C1YNSBncXVGQBQHR/2AQu0E6UWHPO1oTwcwkzay80NQeCJ/gB/S5yeVy3KCYxNQBkqAHSh5SXRCpVOqhxDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8894071be6892a5cd3c6c15c6ebf946532813b2912983538798f23ee56e8840c","last_reissued_at":"2026-07-05T09:31:10.307331Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:10.307331Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"In-Context LoRA for Diffusion Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GR"],"primary_cat":"cs.CV","authors_text":"Chen Liang, Huanzhang Dou, Jingren Zhou, Lianghua Huang, Wei Wang, Yu Liu, Yupeng Shi, Yutong Feng, Zhi-Fan Wu","submitted_at":"2024-10-31T09:45:00Z","abstract_excerpt":"Recent research arXiv:2410.15027 has explored the use of diffusion transformers (DiTs) for task-agnostic image generation by simply concatenating attention tokens across images. However, despite substantial computational resources, the fidelity of the generated images remains suboptimal. In this study, we reevaluate and streamline this framework by hypothesizing that text-to-image DiTs inherently possess in-context generation capabilities, requiring only minimal tuning to activate them. Through diverse task experiments, we qualitatively demonstrate that existing text-to-image DiTs can effectiv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.23775","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.23775/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.23775","created_at":"2026-07-05T09:31:10.307408+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.23775v3","created_at":"2026-07-05T09:31:10.307408+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.23775","created_at":"2026-07-05T09:31:10.307408+00:00"},{"alias_kind":"pith_short_12","alias_value":"RCKAOG7GREVF","created_at":"2026-07-05T09:31:10.307408+00:00"},{"alias_kind":"pith_short_16","alias_value":"RCKAOG7GREVFZU6G","created_at":"2026-07-05T09:31:10.307408+00:00"},{"alias_kind":"pith_short_8","alias_value":"RCKAOG7G","created_at":"2026-07-05T09:31:10.307408+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25907","citing_title":"In-context Region-based Drag: Drag Any Region to Any Shape","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21661","citing_title":"UnityShots: Memory-Driven Multi-Shot Audio-Video Generation with Boundary-Aware Gating","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01677","citing_title":"ICDepth: Taming Video Diffusion Models for Video Depth Estimation via In-Context Conditioning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08514","citing_title":"OmniTryOn: Video Try-On Anything at Once!","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07891","citing_title":"C3VD-DEFCOL: A Deformable Colonoscopy Dataset with Time-Resolved 3D Ground Truth and Realistic Appearance","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04945","citing_title":"STaR-Quant: State-Time Consistent Post-Training Quantization for Diffusion Large Language Models","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06498","citing_title":"Semantic-Structural Alignment for Generative Pictorial Charts","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29319","citing_title":"FDM-MFVT: Few-step Sampling Diffusion Model for Mask-Free Virtual Try-On","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29473","citing_title":"MAVIN: Multi-Shot Audio-Visual Generation with Customized Narrative Control","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25378","citing_title":"CollectionLoRA: Collecting 50 Effects in 1 LoRA via Multi-Teacher On-Policy Distillation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11596","citing_title":"HorizonDrive: Self-Corrective Autoregressive World Model for Long-horizon Driving Simulation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16649","citing_title":"AtlasVid: Efficient Ultra-High-Resolution Long Video Generation via Decoupled Global-Local Modeling","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07469","citing_title":"VideoCoF: Unified Video Editing with Temporal Reasoner","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20690","citing_title":"In-Context Edit: Enabling Instructional Image Editing with In-Context Generation in Large Scale Diffusion Transformer","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22143","citing_title":"JUST-DUB-IT: Video Dubbing via Joint Audio-Visual Diffusion","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14209","citing_title":"ChArtist: Generating Pictorial Charts with Unified Spatial and Subject Control","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03315","citing_title":"StoryBlender: Inter-Shot Consistent and Editable 3D Storyboard with Spatial-temporal Dynamics","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11596","citing_title":"HorizonDrive: Self-Corrective Autoregressive World Model for Long-horizon Driving Simulation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25184","citing_title":"Enabling High Error Tolerance in Satellite Video Transmissions by Generative Semantic Communication","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07455","citing_title":"EditTransfer++: Toward Faithful and Efficient Visual-Prompt-Guided Image Editing","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04911","citing_title":"SpatialEdit: Benchmarking Fine-Grained Image Spatial Editing","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15742","citing_title":"FLUX.1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RCKAOG7GREVFZU6GYFOG5P4UMU","json":"https://pith.science/pith/RCKAOG7GREVFZU6GYFOG5P4UMU.json","graph_json":"https://pith.science/api/pith-number/RCKAOG7GREVFZU6GYFOG5P4UMU/graph.json","events_json":"https://pith.science/api/pith-number/RCKAOG7GREVFZU6GYFOG5P4UMU/events.json","paper":"https://pith.science/paper/RCKAOG7G"},"agent_actions":{"view_html":"https://pith.science/pith/RCKAOG7GREVFZU6GYFOG5P4UMU","download_json":"https://pith.science/pith/RCKAOG7GREVFZU6GYFOG5P4UMU.json","view_paper":"https://pith.science/paper/RCKAOG7G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.23775&json=true","fetch_graph":"https://pith.science/api/pith-number/RCKAOG7GREVFZU6GYFOG5P4UMU/graph.json","fetch_events":"https://pith.science/api/pith-number/RCKAOG7GREVFZU6GYFOG5P4UMU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RCKAOG7GREVFZU6GYFOG5P4UMU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RCKAOG7GREVFZU6GYFOG5P4UMU/action/storage_attestation","attest_author":"https://pith.science/pith/RCKAOG7GREVFZU6GYFOG5P4UMU/action/author_attestation","sign_citation":"https://pith.science/pith/RCKAOG7GREVFZU6GYFOG5P4UMU/action/citation_signature","submit_replication":"https://pith.science/pith/RCKAOG7GREVFZU6GYFOG5P4UMU/action/replication_record"}},"created_at":"2026-07-05T09:31:10.307408+00:00","updated_at":"2026-07-05T09:31:10.307408+00:00"}