File size: 15,026 Bytes
5caac8c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9db98e9
 
 
 
 
 
 
 
2588fce
9db98e9
 
 
 
 
 
 
 
 
 
 
2588fce
 
 
 
 
 
 
 
 
 
9db98e9
 
 
 
5caac8c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9db98e9
bdc3944
 
5caac8c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
from __future__ import annotations

import logging
import os
import random
import threading
import traceback
from dataclasses import dataclass
from pathlib import Path
from typing import Any

# Keep DiffSynth on the Hugging Face download path and make cache locations writable
# on both Spaces and a local checkout.
os.environ.setdefault("DIFFSYNTH_DOWNLOAD_SOURCE", "huggingface")
os.environ.setdefault("DIFFSYNTH_SKIP_DOWNLOAD", "True")
os.environ.setdefault("HF_HUB_ENABLE_HF_TRANSFER", "1")
os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True")

import gradio as gr
import torch
from huggingface_hub import hf_hub_download, snapshot_download
from safetensors.torch import load_file

try:
    import spaces
except ImportError:  # Keeps the app importable when developing outside Spaces.
    class _LocalSpaces:
        @staticmethod
        def GPU(*_args: Any, **_kwargs: Any):
            def decorator(function):
                return function

            return decorator

    spaces = _LocalSpaces()

from diffsynth.pipelines.anima_image import AnimaImagePipeline, ModelConfig
from lycoris import create_lycoris_from_weights

logging.basicConfig(level=logging.INFO)
LOGGER = logging.getLogger("anima-telescopa")

SPACE_MODEL_ID = "RicemanT/Anima-Telescopa"
BASE_MODEL_ID = "circlestone-labs/Anima"

BASE_FILES = {
    "diffusion": "split_files/diffusion_models/anima-base-v1.0.safetensors",
    "text_encoder": "split_files/text_encoders/qwen_3_06b_base.safetensors",
    "vae": "split_files/vae/qwen_image_vae.safetensors",
}

# The finetune file in the source repository is intentionally not exposed here:
# this app loads LoKr adapters onto the published Anima base, while that file is a
# separate full-finetune artifact with different loading requirements.
ADAPTERS = {
    "Recommended · LoKr v0.5 · Epoch 10": "Anima-TelescopaLOKRV0.5-Epoch10.safetensors",
    "Earlier · LoKr v0.1 · Epoch 3": "Anima-TelescopaLOKRV0.1-Epoch3.safetensors",
}

DEFAULT_PROMPT = (
    "1girl, solo, long silver hair, blue eyes, blue dress, underwater, "
    "floating hair, refraction, detailed anime background, cinematic composition"
)
DEFAULT_NEGATIVE = (
    "low quality, worst quality, blurry, jpeg artifacts, watermark, signature, "
    "text, logo, distorted anatomy, extra fingers"
)
RECOMMENDED_PREFIX = "(masterpiece, best quality, highres, detailed background:1.2), "


def _writable_directory(preferred: str, fallback: str) -> Path:
    for candidate in (Path(preferred), Path(fallback)):
        try:
            candidate.mkdir(parents=True, exist_ok=True)
            probe = candidate / ".write-test"
            probe.touch()
            probe.unlink()
            return candidate
        except OSError:
            continue
    raise RuntimeError("No writable model/cache directory is available.")


HF_HOME = _writable_directory(
    os.environ.get("HF_HOME", "/data/.cache/huggingface"),
    "/tmp/.cache/huggingface",
)
MODEL_DIR = _writable_directory(
    os.environ.get("ANIMA_LOCAL_MODEL_DIR", "/data/models/anima-telescopa"),
    "/tmp/models/anima-telescopa",
)
os.environ["HF_HOME"] = str(HF_HOME)
os.environ.setdefault("DIFFSYNTH_MODEL_BASE_PATH", str(MODEL_DIR / "diffsynth"))


@dataclass
class Assets:
    diffusion: str
    text_encoder: str
    vae: str
    qwen_tokenizer_dir: str
    t5_tokenizer_dir: str


_PIPE: AnimaImagePipeline | None = None
_ADAPTER_NETWORK: Any | None = None
_ACTIVE_ADAPTER: str | None = None
_RUNTIME_LOCK = threading.RLock()


def _download_assets(progress: gr.Progress | None = None) -> Assets:
    def download(repo_id: str, filename: str) -> str:
        if progress:
            progress(0, desc=f"Preparing {Path(filename).name}")
        return hf_hub_download(repo_id=repo_id, filename=filename, cache_dir=str(HF_HOME))

    qwen_tokenizer_dir = snapshot_download(
        repo_id="Qwen/Qwen3-0.6B",
        cache_dir=str(HF_HOME),
        allow_patterns=["tokenizer*", "*.json", "*.model"],
    )
    t5_tokenizer_dir = snapshot_download(
        repo_id="google/t5-v1_1-xxl",
        cache_dir=str(HF_HOME),
        allow_patterns=["tokenizer*", "*.json", "*.model"],
    )
    return Assets(
        diffusion=download(BASE_MODEL_ID, BASE_FILES["diffusion"]),
        text_encoder=download(BASE_MODEL_ID, BASE_FILES["text_encoder"]),
        vae=download(BASE_MODEL_ID, BASE_FILES["vae"]),
        qwen_tokenizer_dir=qwen_tokenizer_dir,
        t5_tokenizer_dir=t5_tokenizer_dir,
    )


def _load_pipeline(progress: gr.Progress | None = None) -> AnimaImagePipeline:
    global _PIPE
    with _RUNTIME_LOCK:
        if _PIPE is not None:
            return _PIPE
        if not torch.cuda.is_available():
            raise RuntimeError(
                "Anima requires a CUDA GPU for practical inference. "
                "Run this Space on a GPU-enabled hardware tier (ZeroGPU, T4, A10G, or better)."
            )
        assets = _download_assets(progress)
        if progress:
            progress(0.45, desc="Loading Anima base model")
        pipeline_kwargs = {
            "torch_dtype": torch.bfloat16,
            "device": "cuda",
            "model_configs": [
                ModelConfig(path=assets.diffusion),
                ModelConfig(path=assets.text_encoder),
                ModelConfig(path=assets.vae),
            ],
            "tokenizer_config": ModelConfig(path=assets.qwen_tokenizer_dir),
            "tokenizer_t5xxl_config": ModelConfig(path=assets.t5_tokenizer_dir),
        }
        vram_limit = os.environ.get("ANIMA_VRAM_LIMIT")
        if vram_limit:
            pipeline_kwargs["vram_limit"] = int(vram_limit)
        _PIPE = AnimaImagePipeline.from_pretrained(**pipeline_kwargs)
        if progress:
            progress(0.7, desc="Anima base model ready")
        return _PIPE


def _translate_lokr_state_dict(weights: dict[str, torch.Tensor]) -> dict[str, torch.Tensor]:
    """Translate diffusion-pipe/ComfyUI LoKr names to LyCORIS names.

    Telescopa stores keys like ``diffusion_model.blocks.0.self_attn.q_proj.lokr_w1``.
    LyCORIS resolves target modules from ``pipe.dit.named_modules()`` and expects
    the same module path flattened under its ``lycoris_`` prefix.
    """
    translated: dict[str, torch.Tensor] = {}
    module_prefixes: set[str] = set()
    prefix = "diffusion_model."
    for key, value in weights.items():
        if not key.startswith(prefix):
            translated[key] = value
            continue
        target = key[len(prefix) :]
        if "." not in target:
            continue
        module_name, suffix = target.rsplit(".", 1)
        lycoris_name = "lycoris_" + module_name.replace(".", "_")
        translated[f"{lycoris_name}.{suffix}"] = value
        if suffix in {"lokr_w1", "lokr_w2", "lokr_w1_a", "lokr_w2_a"}:
            module_prefixes.add(lycoris_name)

    # Telescopa was trained with LoKr alpha 16, but its exported safetensors
    # contain no alpha tensors. LyCORIS requires `<prefix>.alpha` and otherwise
    # attempts float(None) while constructing a full-matrix LoKr module.
    for module_prefix in module_prefixes:
        translated.setdefault(f"{module_prefix}.alpha", torch.tensor(16.0))

    if not translated or not module_prefixes:
        raise RuntimeError("The Telescopa LoKr file did not contain translatable adapter weights.")
    return translated


def _activate_adapter(adapter_label: str, scale: float, progress: gr.Progress | None = None) -> None:
    global _ADAPTER_NETWORK, _ACTIVE_ADAPTER
    if adapter_label not in ADAPTERS:
        raise ValueError("Unknown Telescopa adapter variant.")
    pipe = _load_pipeline(progress)
    adapter_filename = ADAPTERS[adapter_label]
    if _ACTIVE_ADAPTER == adapter_filename and _ADAPTER_NETWORK is not None:
        _ADAPTER_NETWORK.multiplier = float(scale)
        return

    with _RUNTIME_LOCK:
        if _ADAPTER_NETWORK is not None:
            _ADAPTER_NETWORK.restore()
            _ADAPTER_NETWORK = None
            _ACTIVE_ADAPTER = None
        if progress:
            progress(0.78, desc=f"Loading {adapter_label}")
        adapter_path = hf_hub_download(
            repo_id=SPACE_MODEL_ID,
            filename=adapter_filename,
            cache_dir=str(HF_HOME),
        )
        # The repository contains diffusion-pipe/Kohya-compatible full-matrix
        # LoKr weights. LyCORIS maps those keys onto DiffSynth's Anima DiT.
        weights = _translate_lokr_state_dict(load_file(adapter_path, device="cpu"))
        # LyCORIS 3.4 returns (network, state_dict), not the network alone.
        _ADAPTER_NETWORK, _ = create_lycoris_from_weights(
            multiplier=float(scale),
            file=adapter_path,
            module=pipe.dit,
            weights_sd=weights,
        )
        _ADAPTER_NETWORK.apply_to()
        matched_loras = getattr(_ADAPTER_NETWORK, "loras", None)
        if not matched_loras:
            _ADAPTER_NETWORK.restore()
            _ADAPTER_NETWORK = None
            raise RuntimeError(
                "The selected LoKr file did not match any Anima DiT layers. "
                "The adapter/runtime versions may be incompatible."
            )
        _ACTIVE_ADAPTER = adapter_filename


def _normalize_dimension(value: int) -> int:
    return max(512, min(1280, int(round(int(value) / 16) * 16)))


def _normalize_seed(seed: int | None) -> int:
    try:
        value = int(seed) if seed is not None else -1
    except (TypeError, ValueError):
        value = -1
    return random.randint(0, 2**31 - 1) if value < 0 else value


def _prepare_prompt(prompt: str, use_prefix: bool) -> str:
    prompt = (prompt or "").strip() or DEFAULT_PROMPT
    if use_prefix and not prompt.lower().startswith(RECOMMENDED_PREFIX.lower()):
        prompt = RECOMMENDED_PREFIX + prompt
    return prompt


@spaces.GPU(duration=180)
def generate(
    prompt: str,
    negative_prompt: str,
    adapter_label: str,
    adapter_scale: float,
    width: int,
    height: int,
    steps: int,
    cfg_scale: float,
    sigma_shift: float,
    seed: int,
    use_prefix: bool,
    progress: gr.Progress = gr.Progress(track_tqdm=False),
):
    try:
        prompt = _prepare_prompt(prompt, use_prefix)
        negative_prompt = (negative_prompt or DEFAULT_NEGATIVE).strip()
        width = _normalize_dimension(width)
        height = _normalize_dimension(height)
        steps = max(10, min(45, int(steps)))
        cfg_scale = max(1.0, min(8.0, float(cfg_scale)))
        sigma_shift = float(sigma_shift)
        seed = _normalize_seed(seed)

        _activate_adapter(adapter_label, adapter_scale, progress)
        pipe = _load_pipeline(progress)
        if progress:
            progress(0.82, desc="Generating Telescopa image")
        with torch.inference_mode():
            image = pipe(
                prompt=prompt,
                negative_prompt=negative_prompt,
                cfg_scale=cfg_scale,
                height=height,
                width=width,
                seed=seed,
                num_inference_steps=steps,
                sigma_shift=None if sigma_shift <= 0 else sigma_shift,
                progress_bar_cmd=lambda value: progress(0.82 + 0.17 * float(value), desc="Sampling"),
            )
        info = (
            f"**Seed:** `{seed}`  ·  **Variant:** {adapter_label}  ·  "
            f"**Size:** {width}×{height}  ·  **Steps:** {steps}"
        )
        return image, info
    except Exception as exc:
        LOGGER.error("Telescopa generation failed: %s", exc)
        LOGGER.debug(traceback.format_exc())
        return None, (
            "**Generation failed.** The base Anima runtime or LoKr adapter could not be loaded. "
            f"`{type(exc).__name__}: {exc}`"
        )


with gr.Blocks(title="Anima · Telescopa", theme=gr.themes.Soft()) as demo:
    gr.Markdown(
        """
# Anima · Telescopa

Generate anime-style illustrations with **[RicemanT/Anima-Telescopa](https://huggingface.co/RicemanT/Anima-Telescopa)**, a full-matrix **LoKr** fine-tune on [Anima](https://huggingface.co/circlestone-labs/Anima).

The recommended settings from the model card are pre-filled: **28 steps · CFG 4 · shift 5**. The first generation downloads the Anima base components, tokenizers, and selected adapter into the Space cache.
        """
    )
    with gr.Row():
        with gr.Column(scale=1):
            prompt = gr.Textbox(label="Prompt", value=DEFAULT_PROMPT, lines=5)
            negative_prompt = gr.Textbox(label="Negative prompt", value=DEFAULT_NEGATIVE, lines=3)
            use_prefix = gr.Checkbox(
                label="Add quality/background prefix",
                value=True,
                info="Adds a compact quality prompt recommended for this fine-tune.",
            )
            adapter_label = gr.Dropdown(
                choices=list(ADAPTERS),
                value=list(ADAPTERS)[0],
                label="Telescopa variant",
            )
            adapter_scale = gr.Slider(0.0, 1.5, value=1.0, step=0.05, label="LoKr strength")
            with gr.Row():
                width = gr.Slider(512, 1280, value=1024, step=16, label="Width")
                height = gr.Slider(512, 1280, value=1024, step=16, label="Height")
            with gr.Row():
                steps = gr.Slider(10, 45, value=28, step=1, label="Steps")
                cfg_scale = gr.Slider(1, 8, value=4, step=0.1, label="CFG")
            with gr.Row():
                sigma_shift = gr.Slider(0, 8, value=5, step=0.1, label="AuraFlow shift")
                seed = gr.Number(value=-1, precision=0, label="Seed (-1 = random)")
            generate_button = gr.Button("Generate image", variant="primary")
        with gr.Column(scale=1):
            output = gr.Image(label="Generated image", type="pil")
            info = gr.Markdown("Choose a prompt and generate an image.\n\n*GPU inference is required.*")

    gr.Examples(
        examples=[
            ["1girl, solo, red hair, school uniform, sunset rooftop, city skyline, wind, dramatic clouds"],
            ["ancient library inside a giant tree, warm sunlight, floating books, intricate anime background"],
            ["small coastal train station at night, glowing vending machines, rain, cinematic anime background"],
        ],
        inputs=[prompt],
        label="Prompt ideas",
    )
    gr.Markdown(
        "**License note:** the model weights are distributed under the CircleStone Labs Non-Commercial License v1.1. "
        "Review the [model card](https://huggingface.co/RicemanT/Anima-Telescopa) before deploying or using this Space."
    )

    generate_button.click(
        fn=generate,
        inputs=[
            prompt,
            negative_prompt,
            adapter_label,
            adapter_scale,
            width,
            height,
            steps,
            cfg_scale,
            sigma_shift,
            seed,
            use_prefix,
        ],
        outputs=[output, info],
    )


if __name__ == "__main__":
    demo.queue(max_size=12).launch()