{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DI3O6WO5RMKIBCFHRYBJX4PXFA","short_pith_number":"pith:DI3O6WO5","schema_version":"1.0","canonical_sha256":"1a36ef59dd8b148088a78e029bf1f72801b23940df09d55dcffc2d7cc31dde03","source":{"kind":"arxiv","id":"2406.10797","version":4},"attestation_state":"computed","paper":{"title":"STAR: Scale-wise Text-conditioned AutoRegressive image generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Biye Li, Huaian Chen, Mohan Zhou, Tao Liang, Tiejun Zhao, Xiaoxiao Ma, Yalong Bai, Yi Jin","submitted_at":"2024-06-16T03:45:45Z","abstract_excerpt":"We introduce STAR, a text-to-image model that employs a scale-wise auto-regressive paradigm. Unlike VAR, which is constrained to class-conditioned synthesis for images up to 256$\\times$256, STAR enables text-driven image generation up to 1024$\\times$1024 through three key designs. First, we introduce a pre-trained text encoder to extract and adopt representations for textual constraints, enhancing details and generalizability. Second, given the inherent structural correlation across different scales, we leverage 2D Rotary Positional Encoding (RoPE) and tweak it into a normalized version, ensur"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10797","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-16T03:45:45Z","cross_cats_sorted":[],"title_canon_sha256":"305456ae02ed98a5b0694bbacc2f9630d7178f772da5b4873fce1131f41d1716","abstract_canon_sha256":"3c59caa62d90ee3a779f9d60ad07c11e85b02a72ec0cff9d08abca451ef1d3cf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:16:34.434076Z","signature_b64":"FRACaQahJteYqkgAcQTQRqUP2xkpCIk9dyafqCCuFUiogRJTzjfe+86MyYNQhdT5QAyKkFeZPFQW/SD8VPWWAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a36ef59dd8b148088a78e029bf1f72801b23940df09d55dcffc2d7cc31dde03","last_reissued_at":"2026-07-05T10:16:34.433590Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:16:34.433590Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"STAR: Scale-wise Text-conditioned AutoRegressive image generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Biye Li, Huaian Chen, Mohan Zhou, Tao Liang, Tiejun Zhao, Xiaoxiao Ma, Yalong Bai, Yi Jin","submitted_at":"2024-06-16T03:45:45Z","abstract_excerpt":"We introduce STAR, a text-to-image model that employs a scale-wise auto-regressive paradigm. Unlike VAR, which is constrained to class-conditioned synthesis for images up to 256$\\times$256, STAR enables text-driven image generation up to 1024$\\times$1024 through three key designs. First, we introduce a pre-trained text encoder to extract and adopt representations for textual constraints, enhancing details and generalizability. Second, given the inherent structural correlation across different scales, we leverage 2D Rotary Positional Encoding (RoPE) and tweak it into a normalized version, ensur"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10797","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10797/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10797","created_at":"2026-07-05T10:16:34.433648+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10797v4","created_at":"2026-07-05T10:16:34.433648+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10797","created_at":"2026-07-05T10:16:34.433648+00:00"},{"alias_kind":"pith_short_12","alias_value":"DI3O6WO5RMKI","created_at":"2026-07-05T10:16:34.433648+00:00"},{"alias_kind":"pith_short_16","alias_value":"DI3O6WO5RMKIBCFH","created_at":"2026-07-05T10:16:34.433648+00:00"},{"alias_kind":"pith_short_8","alias_value":"DI3O6WO5","created_at":"2026-07-05T10:16:34.433648+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14891","citing_title":"Hierarchical Image Tokenization for Multi-Scale Image Super Resolution","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26089","citing_title":"Channel-wise Vector Quantization","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2601.01593","citing_title":"Beyond Patches: Global-aware Autoregressive Model for Multimodal Few-Shot Font Generation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03799","citing_title":"Next-Scale Autoregressive Models for Text-to-Motion Generation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21450","citing_title":"VARestorer: One-Step VAR Distillation for Real-World Image Super-Resolution","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14591","citing_title":"Prompt-Guided Image Editing with Masked Logit Nudging in Visual Autoregressive Models","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DI3O6WO5RMKIBCFHRYBJX4PXFA","json":"https://pith.science/pith/DI3O6WO5RMKIBCFHRYBJX4PXFA.json","graph_json":"https://pith.science/api/pith-number/DI3O6WO5RMKIBCFHRYBJX4PXFA/graph.json","events_json":"https://pith.science/api/pith-number/DI3O6WO5RMKIBCFHRYBJX4PXFA/events.json","paper":"https://pith.science/paper/DI3O6WO5"},"agent_actions":{"view_html":"https://pith.science/pith/DI3O6WO5RMKIBCFHRYBJX4PXFA","download_json":"https://pith.science/pith/DI3O6WO5RMKIBCFHRYBJX4PXFA.json","view_paper":"https://pith.science/paper/DI3O6WO5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10797&json=true","fetch_graph":"https://pith.science/api/pith-number/DI3O6WO5RMKIBCFHRYBJX4PXFA/graph.json","fetch_events":"https://pith.science/api/pith-number/DI3O6WO5RMKIBCFHRYBJX4PXFA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DI3O6WO5RMKIBCFHRYBJX4PXFA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DI3O6WO5RMKIBCFHRYBJX4PXFA/action/storage_attestation","attest_author":"https://pith.science/pith/DI3O6WO5RMKIBCFHRYBJX4PXFA/action/author_attestation","sign_citation":"https://pith.science/pith/DI3O6WO5RMKIBCFHRYBJX4PXFA/action/citation_signature","submit_replication":"https://pith.science/pith/DI3O6WO5RMKIBCFHRYBJX4PXFA/action/replication_record"}},"created_at":"2026-07-05T10:16:34.433648+00:00","updated_at":"2026-07-05T10:16:34.433648+00:00"}