from __future__ import annotations
from collections.abc import Iterator
from contextlib import contextmanager
import json
from pathlib import Path
from typing import cast
import pandas as pd
from openpyxl import load_workbook
from PIL import Image
import pytest
from typer.testing import CliRunner
import kneiff.cli.dataset as cli_dataset
import kneiff.datasets.export.resolution_plan as resolution_plan
import kneiff.datasets.export.writer as dataset_writer
from kneiff.datasets.export.image_grid import TRAINING_IMAGE_GRID_FILENAME
from kneiff.datasets.export.workflow import build_dataset_sync_plan
from kneiff.datasets.manifest.block_schema import BLOCK_HEADERS
from kneiff.datasets.manifest.block_workbook import (
BLOCK_METADATA_SHEET_NAME,
read_block_manifest_workbook,
)
from kneiff.datasets.manifest.schema import (
COL_RELATIVE_PATH,
COL_RESOLUTION,
)
from tests._cli_helpers import invoke_cli
from tests.dataset._export_helpers import (
_manifest_row,
_write_image,
_write_manifest,
_write_light_config,
)
runner = CliRunner()
pytestmark = pytest.mark.usefixtures("isolated_cli_project")
ROOK_PROJECT_ROOT = Path(__file__).parents[1] / "fixtures" / "projects" / "rook"
ROOK_VOCABULARY_PATH = ROOK_PROJECT_ROOT / "vocabulary.knf.yaml"
def _config_path(project_root: Path, name: str = "ready") -> Path:
path = project_root / "configs" / f"{name}.knf.yaml"
path.parent.mkdir(parents=True, exist_ok=True)
return path
def _source_root(project_root: Path) -> Path:
return project_root / "SOURCE"
def _export_root(project_root: Path, name: str = "ready") -> Path:
return project_root / "HF" / name
def _write_export_file(
project_root: Path,
*,
subset: str = "fullbody",
) -> None:
"""Write one exported image and caption pair for training tests."""
image_path = _export_root(project_root) / subset / "scene.png"
image_path.parent.mkdir(parents=True, exist_ok=True)
image_path.write_bytes(b"image")
image_path.with_suffix(".txt").write_text("caption", encoding="utf-8")
def _training_root(project_root: Path, name: str = "ready", run: int = 1) -> Path:
return project_root / "TRAINING" / f"{name}_{run}"
def _manifest_path(project_root: Path) -> Path:
return project_root / "MANIFEST.knf.xlsx"
def _write_source_image(
project_root: Path,
rel_path: str,
*,
color: str = "white",
size: tuple[int, int] = (32, 32),
) -> None:
_write_image(_source_root(project_root) / rel_path, color=color, size=size)
def _write_project_manifest(
project_root: Path,
rows: list[dict[str, str]],
) -> None:
_ensure_rook_project_resources(project_root)
_write_manifest(project_root, rows)
def _ensure_rook_project_resources(project_root: Path) -> None:
"""Write the vocabulary required by Rook-shaped manifest rows."""
project_root.mkdir(parents=True, exist_ok=True)
vocabulary_path = project_root / "vocabulary.knf.yaml"
vocabulary_path.write_text(
ROOK_VOCABULARY_PATH.read_text(encoding="utf-8"),
encoding="utf-8",
)
def _manifest_records(project_root: Path) -> dict[str, dict[str, object]]:
return {
record.relative_path: {"Relative_path": record.relative_path}
for record in read_block_manifest_workbook(_manifest_path(project_root)).records
}
def _write_project_light_config(project_root: Path, name: str = "ready") -> Path:
_ensure_rook_project_resources(project_root)
config_path = _config_path(project_root, name)
_write_light_config(config_path)
return config_path
def _read_jsonl(path: Path) -> list[dict[str, object]]:
"""Read a newline-delimited JSON metadata file."""
return [
json.loads(line)
for line in path.read_text(encoding="utf-8").splitlines()
if line.strip()
]
def test_dataset_export_command_is_removed() -> None:
result = invoke_cli(runner, ["dataset", "export", "--help"])
assert result.exit_code != 0
def test_dataset_init_command_is_removed() -> None:
result = invoke_cli(runner, ["dataset", "init", "--help"])
assert result.exit_code != 0
def test_dataset_sync_plan_uses_explicit_workers(tmp_path: Path) -> None:
_write_project_light_config(tmp_path)
plan = build_dataset_sync_plan(workdir=tmp_path, workers=5)
assert plan.export_config.workers == 5
assert plan.dataset_name == "ready"
assert plan.source_root == tmp_path / "SOURCE"
assert plan.manifest_path == tmp_path / "MANIFEST.knf.xlsx"
assert plan.export_root == tmp_path / "HF" / "ready"
assert plan.training_workspace_root == tmp_path / "TRAINING" / "ready_1"
def test_dataset_plan_cli_loads_workers_from_apprc_storage(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
config_path = _write_project_light_config(tmp_path)
(tmp_path / ".env.apprc-storage").write_text(
"KNF_WORKERS=5\n",
encoding="utf-8",
)
captured_workers: list[int] = []
run_plan = cli_dataset._run_dataset_plan_for_plan
def capture_workers(plan):
"""Record the AppRC value after the CLI has built its dataset plan."""
captured_workers.append(plan.export_config.workers)
return run_plan(plan)
monkeypatch.setattr(cli_dataset, "_run_dataset_plan_for_plan", capture_workers)
result = invoke_cli(runner, ["dataset", "plan", str(config_path)])
assert result.exit_code == 0, result.output
assert captured_workers == [5]
def test_dataset_sync_cli_rejects_directory_input_with_config_hint(
tmp_path: Path,
) -> None:
result = invoke_cli(runner, ["dataset", "sync", str(tmp_path)])
assert result.exit_code == 1
assert "configs/*.knf.yaml file, not a directory" in result.stderr
def test_dataset_sync_cli_prefixes_missing_config_errors(tmp_path: Path) -> None:
config_path = _config_path(tmp_path, "missing")
result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert result.exit_code == 1
assert "[missing | dataset sync]" in result.stderr
assert str(config_path) in result.stderr
@pytest.mark.parametrize(
"path_key,path_value",
[
("source_root", "."),
("manifest_path", "MANIFEST.knf.xlsx"),
("export_root", "HF/ready"),
("allow_export_inside_source", "true"),
],
)
def test_dataset_sync_cli_rejects_config_owned_path_fields(
tmp_path: Path,
path_key: str,
path_value: str,
) -> None:
config_path = _config_path(tmp_path)
config_path.parent.mkdir(parents=True, exist_ok=True)
config_path.write_text(
f"""
{path_key}: {path_value}
mappings:
fullbody:
- "0-FULLBODY"
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert result.exit_code == 1
assert "[ready | dataset sync]" in result.stderr
assert "remove these key(s)" in result.stderr
assert path_key in result.stderr
def test_dataset_sync_cli_warns_without_image_resize(tmp_path: Path) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _write_project_light_config(tmp_path, "alpha")
result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert result.exit_code == 0
assert "Config: alpha" in result.stdout
assert f"Project root: {tmp_path}" in result.stdout
assert f"Source root: {_source_root(tmp_path)}" in result.stdout
assert f"Manifest: {_manifest_path(tmp_path)}" in result.stdout
assert f"Export root: {_export_root(tmp_path, 'alpha')}" in result.stdout
assert "Training root:" not in result.stdout
assert (
"Warning: alpha has no image_resize block; exported images "
"will keep source dimensions."
) in result.stdout
def test_dataset_sync_cli_does_not_warn_with_image_resize(tmp_path: Path) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _config_path(tmp_path, "alpha")
config_path.parent.mkdir(parents=True, exist_ok=True)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
image_resize:
min_pixel_area: 16
max_pixel_area: 64
augmentations:
mirrored_extra: false
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert result.exit_code == 0
assert "has no image_resize block" not in result.stdout
def test_dataset_sync_cli_accepts_config_file_and_ignores_sibling_configs(
tmp_path: Path,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
alpha_config = _write_project_light_config(tmp_path, "alpha")
_write_project_light_config(tmp_path, "beta")
result = invoke_cli(runner, ["dataset", "sync", str(alpha_config)])
assert result.exit_code == 0
assert "Config: alpha" in result.stdout
assert (_export_root(tmp_path, "alpha") / "fullbody" / "scene__orig.png").exists()
assert not _export_root(tmp_path, "beta").exists()
def test_dataset_sync_cli_writes_fixed_block_manifest(tmp_path: Path) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
config_path = _write_project_light_config(tmp_path)
result = invoke_cli(runner, ["dataset", "sync", str(config_path), "-m"])
assert result.exit_code == 0, result.output
workbook = load_workbook(_manifest_path(tmp_path), read_only=True)
try:
visible_sheet = next(
worksheet
for worksheet in workbook.worksheets
if worksheet.title != BLOCK_METADATA_SHEET_NAME
)
assert (
tuple(visible_sheet.cell(1, column).value for column in range(1, 9))
== BLOCK_HEADERS
)
assert visible_sheet["C2"].value == "0-FULLBODY/scene.png"
assert workbook[BLOCK_METADATA_SHEET_NAME].sheet_state == "veryHidden"
assert "_caption_previews" not in workbook.sheetnames
finally:
workbook.close()
assert (tmp_path / "MANIFEST.yaml").exists()
def test_resolution_plan_uses_cached_manifest_resolution(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png", size=(640, 320))
config_path = _write_project_light_config(tmp_path)
plan = build_dataset_sync_plan(workdir=tmp_path, config_path=config_path)
dataframe = pd.DataFrame(
[
{
COL_RELATIVE_PATH: "0-FULLBODY/scene.png",
COL_RESOLUTION: "640x320",
}
]
)
def fail_image_size(_path: Path) -> tuple[int, int]:
raise AssertionError("cached source size should not be probed")
monkeypatch.setattr(resolution_plan, "image_size", fail_image_size)
result = resolution_plan.build_manifest_resolution_plan_sheet(
dataframe,
plan,
metadata_workers=4,
)
record = result.to_dict("records")[0]
assert record["Source size"] == "640x320"
assert record["Kneiff output"] == "copy 640x320"
def test_resolution_plan_probes_missing_manifest_resolution(tmp_path: Path) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png", size=(320, 160))
config_path = _write_project_light_config(tmp_path)
plan = build_dataset_sync_plan(workdir=tmp_path, config_path=config_path)
dataframe = pd.DataFrame([{COL_RELATIVE_PATH: "0-FULLBODY/scene.png"}])
result = resolution_plan.build_manifest_resolution_plan_sheet(
dataframe,
plan,
metadata_workers=2,
)
record = result.to_dict("records")[0]
assert record["Source size"] == "320x160"
assert record["Kneiff output"] == "copy 320x160"
def test_resolution_plan_skips_simpletuner_artifacts_for_unmapped_config(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png", size=(640, 320))
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
head:
- "2-HEAD"
training:
enabled: true
simpletuner:
trainer:
max_train_steps: 1
""",
encoding="utf-8",
)
plan = build_dataset_sync_plan(workdir=tmp_path, config_path=config_path)
dataframe = pd.DataFrame(
[
{
COL_RELATIVE_PATH: "0-FULLBODY/scene.png",
COL_RESOLUTION: "640x320",
}
]
)
def fail_artifact_build(*args, **kwargs):
raise AssertionError("unmapped configs should not build trainer artifacts")
monkeypatch.setattr(
resolution_plan,
"build_simpletuner_artifacts",
fail_artifact_build,
)
result = resolution_plan.build_manifest_resolution_plan_sheet(dataframe, plan)
record = result.to_dict("records")[0]
assert record["Status"] == "not selected"
assert record["Reason"] == "not matched by config mappings"
def test_dataset_sync_cli_manifest_only_with_config_skips_export(
tmp_path: Path,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
config_path = _write_project_light_config(tmp_path)
export_root = _export_root(tmp_path)
export_root.mkdir(parents=True)
keep_file = export_root / "keep.txt"
keep_file.write_text("keep\n", encoding="utf-8")
result = invoke_cli(runner, ["dataset", "sync", str(config_path), "-m"])
assert result.exit_code == 0
assert _manifest_path(tmp_path).exists()
assert keep_file.exists()
assert not (export_root / "fullbody" / "scene__orig.png").exists()
assert "Exported" not in result.stdout
def test_dataset_sync_cli_crawls_only_source_root(tmp_path: Path) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_image(_export_root(tmp_path) / "fullbody" / "old.png")
_write_image(_training_root(tmp_path) / "dataset" / "old.png")
config_path = _write_project_light_config(tmp_path)
result = invoke_cli(runner, ["dataset", "sync", str(config_path), "-m"])
assert result.exit_code == 0
records = _manifest_records(tmp_path)
assert tuple(records) == ("0-FULLBODY/scene.png",)
def test_dataset_sync_cli_updates_incrementally_without_prompt(tmp_path: Path) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _write_project_light_config(tmp_path)
export_root = _export_root(tmp_path)
export_root.mkdir(parents=True)
keep_file = export_root / "keep.txt"
keep_file.write_text("keep\n", encoding="utf-8")
result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert result.exit_code == 0
assert keep_file.exists()
assert (export_root / "fullbody" / "scene__orig.png").exists()
assert "Preparing dataset export" in result.stdout
assert "Delete and rebuild it?" not in result.stdout
assert "Filesystem work: incremental" in result.stdout
def test_dataset_sync_cli_reports_resolution_and_changed_image_details(
tmp_path: Path,
) -> None:
_write_source_image(
tmp_path,
"0-FULLBODY/scene.png",
size=(2000, 1000),
)
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
image_resize:
max_pixel_area: 1000
augmentations:
mirrored_extra: false
training:
enabled: true
simpletuner:
dataset:
crop: false
crop_aspect: preserve
resolution: 1024
minimum_image_size: 512
resolution_type: pixel_area
trainer:
aspect_bucket_alignment: 64
max_train_steps: 1
subsets:
fullbody:
probability: 1.0
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert result.exit_code == 0
assert "Image resolutions" in result.stdout
assert "Source resolutions:" in result.stdout
assert "2000x1000" in result.stdout
assert "Export resolutions:" in result.stdout
assert "1414x707" in result.stdout
assert "Source resolution plan" in result.stdout
assert "SimpleTuner behavior" in result.stdout
assert "1472x704 resize" in result.stdout
assert "Changed images:" in result.stdout
assert (
"[resize] fullbody/scene__orig.png <- 0-FULLBODY/scene.png; "
"2000x1000 -> 1414x707; scale 0.707; aspect kept"
) in result.stdout
def test_dataset_sync_cli_reports_kicked_resolution_images(
tmp_path: Path,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/tiny.png", size=(256, 256))
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/tiny.png")])
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
training:
enabled: true
simpletuner:
dataset:
crop: false
crop_aspect: preserve
resolution: 1024
minimum_image_size: 512
resolution_type: pixel
trainer:
max_train_steps: 1
subsets:
fullbody:
probability: 1.0
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert result.exit_code == 0
assert "Resolution health" in result.stdout
assert "KICKED" in result.stdout
assert "Kicked-out images" in result.stdout
assert "0-FULLBODY/tiny.png" in result.stdout
assert "minimum_image_size" in result.stdout
def test_dataset_sync_cli_dry_run_reports_kicked_resolution_images(
tmp_path: Path,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/tiny.png", size=(256, 256))
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/tiny.png")])
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
training:
enabled: true
simpletuner:
dataset:
crop: false
crop_aspect: preserve
resolution: 1024
minimum_image_size: 512
resolution_type: pixel
trainer:
max_train_steps: 1
subsets:
fullbody:
probability: 1.0
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path), "--dry-run"])
assert result.exit_code == 0
assert "Resolution health" in result.stdout
assert "KICKED" in result.stdout
assert "Kicked-out images" in result.stdout
assert "minimum_image_size" in result.stdout
assert "Changed images:" not in result.stdout
def test_dataset_sync_cli_caps_changed_images_by_default(
tmp_path: Path,
) -> None:
rows = []
for index in range(21):
rel_path = f"0-FULLBODY/image_{index:02d}.png"
_write_source_image(tmp_path, rel_path, size=(32 + index, 32))
rows.append(_manifest_row(rel_path))
_write_project_manifest(tmp_path, rows)
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
image_resize:
min_pixel_area: 16
max_pixel_area: 1024
augmentations:
mirrored_extra: false
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
changed_lines = [
line for line in result.stdout.splitlines() if line.startswith("[copy] ")
]
assert result.exit_code == 0
assert len(changed_lines) == 20
assert "image_20__orig.png" not in result.stdout
assert "... 1 more changed images omitted; rerun with --verbose" in result.stdout
assert "other" in result.stdout
def test_dataset_sync_cli_verbose_uncaps_changed_images_and_buckets(
tmp_path: Path,
) -> None:
rows = []
for index in range(21):
rel_path = f"0-FULLBODY/image_{index:02d}.png"
_write_source_image(tmp_path, rel_path, size=(32 + index, 32))
rows.append(_manifest_row(rel_path))
_write_project_manifest(tmp_path, rows)
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
image_resize:
min_pixel_area: 16
max_pixel_area: 1024
augmentations:
mirrored_extra: false
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path), "--verbose"])
changed_lines = [
line for line in result.stdout.splitlines() if line.startswith("[copy] ")
]
assert result.exit_code == 0
assert len(changed_lines) == 21
assert "image_20__orig.png" in result.stdout
assert "52x32" in result.stdout
assert "more changed images omitted" not in result.stdout
assert "other" not in result.stdout
def test_dataset_sync_cli_rebuild_prompts_before_cleanup(tmp_path: Path) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _write_project_light_config(tmp_path)
export_root = _export_root(tmp_path)
export_root.mkdir(parents=True)
keep_file = export_root / "keep.txt"
keep_file.write_text("keep\n", encoding="utf-8")
no_result = invoke_cli(
runner,
["dataset", "sync", str(config_path), "--rebuild"],
input="n\n",
)
assert no_result.exit_code == 0
assert keep_file.exists()
assert not (export_root / "fullbody" / "scene__orig.png").exists()
assert "Preparing dataset export" in no_result.stdout
assert f"{export_root} is not empty. Delete and rebuild it?" in no_result.stdout
assert "skipped" in no_result.stdout
assert "Total exported: 0 source images, 0 images, 0 captions" in no_result.stdout
def test_dataset_export_suspends_progress_around_rebuild_prompt(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _write_project_light_config(tmp_path)
export_root = _export_root(tmp_path)
export_root.mkdir(parents=True)
(export_root / "keep.txt").write_text("keep\n", encoding="utf-8")
plan = build_dataset_sync_plan(workdir=tmp_path, config_path=config_path)
events: list[str] = []
class FakeProgress:
"""Record prompt/progress sequencing without Rich terminal rendering."""
def message(self, text: str, *, echo: bool = True) -> None:
"""Record visible progress messages."""
events.append(f"message:{text}:{echo}")
@contextmanager
def suspend(self) -> Iterator[None]:
"""Record suspension boundaries around the prompt."""
events.append("suspend-enter")
try:
yield
finally:
events.append("suspend-exit")
def confirm(prompt: str, *, default: bool) -> bool:
"""Return no while recording the prompt location."""
events.append(f"confirm:{prompt}:{default}")
return False
def fail_export(*_args: object, **_kwargs: object) -> None:
"""Reject export when the user declines rebuild cleanup."""
raise AssertionError("export should not run after declined rebuild")
monkeypatch.setattr(cli_dataset.typer, "confirm", confirm)
monkeypatch.setattr(cli_dataset, "export_training_dataset", fail_export)
cli_dataset._run_dataset_export(
plan,
yes=False,
rebuild=True,
verbose=False,
progress=cast(cli_dataset.CliProgress, FakeProgress()),
)
assert events == [
"message:Preparing dataset export:True",
"suspend-enter",
f"confirm:{export_root} is not empty. Delete and rebuild it?:False",
"suspend-exit",
]
def test_dataset_sync_cli_yes_does_not_rebuild_without_rebuild(tmp_path: Path) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _write_project_light_config(tmp_path)
export_root = _export_root(tmp_path)
export_root.mkdir(parents=True)
keep_file = export_root / "keep.txt"
keep_file.write_text("keep\n", encoding="utf-8")
result = invoke_cli(runner, ["dataset", "sync", str(config_path), "-y"])
assert result.exit_code == 0
assert keep_file.exists()
assert (export_root / "fullbody" / "scene__orig.png").exists()
def test_dataset_sync_cli_rebuild_yes_cleans_without_prompt(tmp_path: Path) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _write_project_light_config(tmp_path)
export_root = _export_root(tmp_path)
export_root.mkdir(parents=True)
keep_file = export_root / "keep.txt"
keep_file.write_text("keep\n", encoding="utf-8")
result = invoke_cli(
runner,
["dataset", "sync", str(config_path), "--rebuild", "-y"],
)
assert result.exit_code == 0
assert not keep_file.exists()
assert (export_root / "fullbody" / "scene__orig.png").exists()
assert "Delete and rebuild it?" not in result.stdout
def test_dataset_sync_cli_report_counts_augmented_and_separate_outputs(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _config_path(tmp_path)
config_path.parent.mkdir(parents=True, exist_ok=True)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
caption_outputs:
mode: separate_txt
formats: [tags, natural]
augmentations:
mirrored_extra: true
seed: 12345
""",
encoding="utf-8",
)
monkeypatch.setattr(
dataset_writer,
"_mirrored_augmented_image",
lambda _source_path, _seed: Image.new("RGB", (32, 32), color="black"),
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path), "-y"])
assert result.exit_code == 0
assert "ready" in result.stdout
assert "Dataset/Subdir" in result.stdout
assert "Sources" in result.stdout
assert "Images" in result.stdout
assert "Captions" in result.stdout
assert "fullbody" in result.stdout
assert "4.00x" in result.stdout
assert "4-4" in result.stdout
assert "Total exported: 1 source images, 4 images, 4 captions" in result.stdout
export_root = _export_root(tmp_path) / "fullbody"
assert (export_root / "scene__orig__tags.png").exists()
assert (export_root / "scene__orig__natural.png").exists()
assert (export_root / "scene__aug-mirror__tags.png").exists()
assert (export_root / "scene__aug-mirror__natural.png").exists()
def test_dataset_sync_cli_writes_hub_artifacts_and_preserves_training(
tmp_path: Path,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png", color="red")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
export_root = _export_root(tmp_path)
training_root = _training_root(tmp_path)
local_model_path = "/tmp/demo-models/chroma/model.safetensors"
preserved_output = training_root / "_simpletuner-output" / "keep.bin"
preserved_output.parent.mkdir(parents=True)
preserved_output.write_text("trained", encoding="utf-8")
export_root.mkdir(parents=True)
(export_root / "old-public.txt").write_text("stale", encoding="utf-8")
config_path = _config_path(tmp_path)
config_path.parent.mkdir(parents=True, exist_ok=True)
config_path.write_text(
f"""
mappings:
fullbody:
- "0-FULLBODY"
augmentations:
mirrored_extra: false
training:
enabled: true
simpletuner:
model:
pretrained_model_name_or_path: lodestones/Chroma1-HD
pretrained_transformer_model_name_or_path: {local_model_path}
trainer:
hub_model_id: ladybug-felkin-v5.1
subsets:
fullbody:
probability: 1.0
publishing:
huggingface:
pretty_name: Ladybug Felkin
version: v5.1
adult_content: true
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path), "-y"])
assert result.exit_code == 0
assert preserved_output.exists()
assert (export_root / "old-public.txt").read_text(encoding="utf-8") == "stale"
assert (export_root / "README.md").exists()
assert (export_root / "metadata.jsonl").exists()
assert (export_root / ".hfignore").exists()
assert (export_root / TRAINING_IMAGE_GRID_FILENAME).exists()
assert not (training_root / "simpletuner-config.json").exists()
assert not (export_root / "_TRAINING").exists()
metadata = _read_jsonl(export_root / "metadata.jsonl")
assert metadata[0]["file_name"] == "fullbody/scene__orig.png"
readme = (export_root / "README.md").read_text(encoding="utf-8")
assert readme.index("## Content Notice") < readme.index("
= 5 and line.split("|")[4].strip() for line in subset_rows
)
assert local_model_path not in readme
assert str(tmp_path) not in readme
@pytest.mark.parametrize(
"artifact_name",
["metadata.jsonl", "README.md", ".hfignore", ".gitignore"],
)
def test_dataset_sync_rejects_symlinked_huggingface_artifact(
tmp_path: Path,
artifact_name: str,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png", color="red")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
publishing:
huggingface:
pretty_name: Demo dataset
""",
encoding="utf-8",
)
export_root = _export_root(tmp_path)
export_root.mkdir(parents=True)
outside = tmp_path / f"outside-{artifact_name.removeprefix('.')}"
outside.write_text("keep", encoding="utf-8")
(export_root / artifact_name).symlink_to(outside)
result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert result.exit_code == 1
assert "Hugging Face dataset artifact must not cross a symlink" in result.stderr
assert outside.read_text(encoding="utf-8") == "keep"
def test_dataset_sync_restores_huggingface_bundle_after_activation_failure(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png", color="red")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
publishing:
huggingface:
pretty_name: First name
""",
encoding="utf-8",
)
first_result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert first_result.exit_code == 0, first_result.output
export_root = _export_root(tmp_path)
artifact_names = ("metadata.jsonl", "README.md", ".hfignore", ".gitignore")
original_bytes = {
name: (export_root / name).read_bytes() for name in artifact_names
}
config_path.write_text(
config_path.read_text(encoding="utf-8").replace("First name", "Second name"),
encoding="utf-8",
)
original_replace = Path.replace
def fail_hfignore_activation(path: Path, target: Path) -> Path:
if path.parent.name == "staged" and target.name == ".hfignore":
raise OSError("injected Hugging Face bundle failure")
return original_replace(path, target)
monkeypatch.setattr(Path, "replace", fail_hfignore_activation)
failed_result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert failed_result.exit_code == 1
assert "injected Hugging Face bundle failure" in failed_result.stderr
assert {
name: (export_root / name).read_bytes() for name in artifact_names
} == original_bytes
assert not list(export_root.glob(".kneiff-hf-artifacts-*"))
def test_dataset_sync_cli_refreshes_readme_without_rewriting_images(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png", color="red")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
augmentations:
mirrored_extra: false
publishing:
huggingface:
pretty_name: Ladybug Felkin
""",
encoding="utf-8",
)
first_result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
export_root = _export_root(tmp_path)
image_path = export_root / "fullbody" / "scene__orig.png"
readme_path = export_root / "README.md"
image_mtime = image_path.stat().st_mtime_ns
readme_path.write_text("stale readme", encoding="utf-8")
def fail_grid(*_args: object, **_kwargs: object) -> None:
"""Reject grid writes when exported images are unchanged."""
raise AssertionError("sync should keep the existing image grid")
monkeypatch.setattr(cli_dataset, "write_training_image_grid", fail_grid)
second_result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert first_result.exit_code == 0
assert second_result.exit_code == 0
assert image_path.stat().st_mtime_ns == image_mtime
readme = readme_path.read_text(encoding="utf-8")
assert "stale readme" not in readme
assert "Ladybug Felkin" in readme
assert "0 images written, 1 skipped" in second_result.stdout
def test_dataset_sync_rejects_unchanged_symlinked_training_grid(
tmp_path: Path,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png", color="red")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
""",
encoding="utf-8",
)
first_result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert first_result.exit_code == 0, first_result.output
grid_path = _export_root(tmp_path) / TRAINING_IMAGE_GRID_FILENAME
grid_path.unlink()
external_grid = tmp_path / "external-grid.jpg"
external_grid.write_bytes(b"external grid")
grid_path.symlink_to(external_grid)
second_result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert second_result.exit_code == 1
assert "Training image grid must not be a symlink" in second_result.stderr
assert external_grid.read_bytes() == b"external grid"
def test_dataset_sync_cli_reports_training_sampling_columns(
tmp_path: Path,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_source_image(tmp_path, "2-HEAD/portrait.png")
_write_project_manifest(
tmp_path,
[
_manifest_row("0-FULLBODY/scene.png"),
_manifest_row("2-HEAD/portrait.png"),
],
)
config_path = _config_path(tmp_path)
config_path.parent.mkdir(parents=True, exist_ok=True)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
details:
- "2-HEAD"
augmentations:
mirrored_extra: false
training:
enabled: true
simpletuner:
trainer:
data_backend_sampling: uniform
subsets:
fullbody:
probability: 2.0
details:
probability: 1.0
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path), "-y"])
assert result.exit_code == 0
assert "Train prob" in result.stdout
assert "Train %" in result.stdout
assert "66.7%" in result.stdout
assert "33.3%" in result.stdout
assert "SimpleTuner configs:" not in result.stdout
assert "Image grids:" in result.stdout
assert (_export_root(tmp_path) / TRAINING_IMAGE_GRID_FILENAME).exists()
def test_dataset_describe_cli_prints_table_without_writing(tmp_path: Path) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_source_image(tmp_path, "2-HEAD/portrait.png")
_write_project_manifest(
tmp_path,
[
_manifest_row("0-FULLBODY/scene.png"),
_manifest_row("2-HEAD/portrait.png"),
],
)
config_path = _config_path(tmp_path)
config_path.parent.mkdir(parents=True, exist_ok=True)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
details:
- "2-HEAD"
augmentations:
mirrored_extra: false
training:
enabled: true
simpletuner:
trainer:
data_backend_sampling: uniform
subsets:
fullbody:
probability: 2.0
details:
probability: 1.0
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "describe", str(config_path)])
assert result.exit_code == 0
assert "Dataset/Subdir" in result.stdout
assert "Train prob" in result.stdout
assert "Train %" in result.stdout
assert "66.7%" in result.stdout
assert "33.3%" in result.stdout
assert "validated" in result.stdout
assert "Dataset sync complete" not in result.stdout
assert not _export_root(tmp_path).exists()
def test_dataset_sync_cli_dry_run_does_not_write_training_image_grid(
tmp_path: Path,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
config_path = _write_project_light_config(tmp_path)
result = invoke_cli(runner, ["dataset", "sync", str(config_path), "--dry-run"])
assert result.exit_code == 0
assert "Image grid:" not in result.stdout
assert "Image resolutions" in result.stdout
assert "Changed images:" not in result.stdout
assert not (_export_root(tmp_path) / TRAINING_IMAGE_GRID_FILENAME).exists()
def test_dataset_plan_cli_prints_resolved_paths(tmp_path: Path) -> None:
config_path = _write_project_light_config(tmp_path)
result = invoke_cli(runner, ["dataset", "plan", str(config_path)])
assert result.exit_code == 0
assert "Dataset plan" in result.stdout
assert "Config: ready" in result.stdout
assert f"Config path: {config_path}" in result.stdout
assert f"Project root: {tmp_path}" in result.stdout
assert f"Source root: {_source_root(tmp_path)}" in result.stdout
assert f"Manifest: {_manifest_path(tmp_path)}" in result.stdout
assert f"Export root: {_export_root(tmp_path)}" in result.stdout
assert "Training workspace:" not in result.stdout
assert "Export subsets: fullbody" in result.stdout
def test_train_prepare_cli_prints_resolved_paths(tmp_path: Path) -> None:
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
genitals:
- "6-DICK"
fullbody:
- "0-FULLBODY"
training:
enabled: true
simpletuner:
curriculum:
phases:
- name: focused_start
start_step: 0
subsets: [genitals]
- name: full_mix
start_step: 200
subsets: all
trainer:
max_train_steps: 1
subsets:
genitals:
probability: 1.0
fullbody:
probability: 1.0
""",
encoding="utf-8",
)
_write_export_file(tmp_path, subset="genitals")
_write_export_file(tmp_path, subset="fullbody")
result = invoke_cli(runner, ["train", "prepare", str(config_path)])
assert result.exit_code == 0
assert "Config: ready" in result.stdout
assert str(config_path) not in result.stdout
assert f"Project root: {tmp_path}" in result.stdout
assert f"Source root: {_source_root(tmp_path)}" in result.stdout
assert "Run: 1" in result.stdout
assert f"Export root: {_export_root(tmp_path)}" in result.stdout
assert f"Training root: {_training_root(tmp_path)}" in result.stdout
assert (_training_root(tmp_path) / "simpletuner-config.json").exists()
def test_train_prepare_cli_reports_simpletuner_resolutions(tmp_path: Path) -> None:
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
training:
enabled: true
simpletuner:
dataset:
crop: false
crop_aspect: preserve
resolution: 1024
resolution_type: pixel_area
trainer:
aspect_bucket_alignment: 64
max_train_steps: 1
subsets:
fullbody:
probability: 1.0
""",
encoding="utf-8",
)
image_path = _export_root(tmp_path) / "fullbody" / "scene.png"
_write_image(image_path, size=(2000, 1000))
image_path.with_suffix(".txt").write_text("caption", encoding="utf-8")
result = invoke_cli(runner, ["train", "prepare", str(config_path)])
assert result.exit_code == 0
assert "SimpleTuner resolutions" in result.stdout
assert "2000x1000" in result.stdout
assert "1472x704 resize" in result.stdout
def test_train_prepare_cli_prefixes_missing_config_errors(tmp_path: Path) -> None:
config_path = _config_path(tmp_path, "missing")
result = invoke_cli(runner, ["train", "prepare", str(config_path)])
assert result.exit_code == 1
assert "[missing | train prepare]" in result.stderr
assert str(config_path) in result.stderr
def test_dataset_grid_cli_writes_existing_training_image_grid(tmp_path: Path) -> None:
export_root = _export_root(tmp_path)
_write_image(export_root / "fullbody" / "scene__orig.png", color="red")
config_path = _write_project_light_config(tmp_path)
result = invoke_cli(runner, ["dataset", "grid", str(config_path)])
grid_path = export_root / TRAINING_IMAGE_GRID_FILENAME
assert result.exit_code == 0
assert f"Wrote {grid_path}" in result.stdout
assert grid_path.exists()
def test_dataset_grid_cli_writes_custom_output(tmp_path: Path) -> None:
export_root = _export_root(tmp_path)
custom_output = tmp_path / "custom-grid.png"
_write_image(export_root / "fullbody" / "scene__orig.png", color="red")
config_path = _write_project_light_config(tmp_path)
result = invoke_cli(
runner,
["dataset", "grid", str(config_path), "--output", str(custom_output)],
)
assert result.exit_code == 0
assert f"Wrote {custom_output}" in result.stdout
assert custom_output.exists()
assert not (export_root / TRAINING_IMAGE_GRID_FILENAME).exists()
def test_dataset_grid_cli_rejects_custom_output_with_multiple_configs(
tmp_path: Path,
) -> None:
alpha_config = _write_project_light_config(tmp_path, name="alpha")
beta_config = _write_project_light_config(tmp_path, name="beta")
result = invoke_cli(
runner,
[
"dataset",
"grid",
str(alpha_config),
str(beta_config),
"--output",
str(tmp_path / "custom-grid.png"),
],
)
assert result.exit_code == 1
assert "[alpha, beta | dataset grid]" in result.stderr
assert "--output cannot be combined with multiple configs" in result.stderr
def test_dataset_grid_cli_excludes_mirrored_variants(tmp_path: Path) -> None:
export_root = _export_root(tmp_path)
custom_output = tmp_path / "custom-grid.png"
_write_image(export_root / "fullbody" / "scene__orig.png", color="red")
_write_image(export_root / "fullbody" / "scene__aug-mirror.png", color="blue")
config_path = _write_project_light_config(tmp_path)
result = invoke_cli(
runner,
["dataset", "grid", str(config_path), "--output", str(custom_output)],
)
assert result.exit_code == 0
with Image.open(custom_output) as rendered:
colors = rendered.convert("RGB").getcolors(maxcolors=100000)
assert colors is not None
color_counts = {color: count for count, color in colors}
assert color_counts.get((255, 0, 0), 0) > 0
assert color_counts.get((0, 0, 255), 0) == 0
def test_dataset_grid_cli_reports_only_mirrored_exports(tmp_path: Path) -> None:
export_root = _export_root(tmp_path)
_write_image(export_root / "fullbody" / "scene__aug-mirror.png", color="blue")
config_path = _write_project_light_config(tmp_path)
result = invoke_cli(runner, ["dataset", "grid", str(config_path)])
assert result.exit_code == 1
assert "No non-mirrored exported images found below" in result.stderr
def test_dataset_grid_cli_allows_empty_generated_sfw_subset(tmp_path: Path) -> None:
export_root = _export_root(tmp_path)
_write_image(export_root / "fullbody" / "scene__orig.png", color="red")
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
export_sfw_subset: true
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "grid", str(config_path)])
assert result.exit_code == 0
assert (export_root / TRAINING_IMAGE_GRID_FILENAME).exists()
def test_dataset_grid_cli_excludes_sfw_unless_configured_for_training(
tmp_path: Path,
) -> None:
export_root = _export_root(tmp_path)
custom_output = tmp_path / "training-grid.png"
_write_image(export_root / "fullbody" / "scene__orig.png", color="red")
_write_image(export_root / "sfw" / "scene__orig.png", color="green")
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
export_sfw_subset: true
training:
enabled: true
simpletuner:
subsets:
fullbody:
probability: 1.0
""",
encoding="utf-8",
)
result = invoke_cli(
runner,
["dataset", "grid", str(config_path), "--output", str(custom_output)],
)
assert result.exit_code == 0
with Image.open(custom_output) as rendered:
colors = rendered.convert("RGB").getcolors(maxcolors=100000)
assert colors is not None
color_counts = {color: count for count, color in colors}
assert color_counts.get((255, 0, 0), 0) > 0
assert color_counts.get((0, 128, 0), 0) == 0
def test_dataset_grid_cli_reports_missing_export_root(tmp_path: Path) -> None:
config_path = _write_project_light_config(tmp_path)
result = invoke_cli(runner, ["dataset", "grid", str(config_path)])
assert result.exit_code == 1
assert "Export root does not exist" in result.stderr
def test_dataset_grid_cli_reports_missing_training_subset(tmp_path: Path) -> None:
export_root = _export_root(tmp_path)
_write_image(export_root / "sfw" / "scene__orig.png", color="green")
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
training:
enabled: true
simpletuner:
subsets:
fullbody:
probability: 1.0
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "grid", str(config_path)])
assert result.exit_code == 1
assert "fullbody" in result.stderr
def test_dataset_readme_cli_rewrites_only_readme_from_existing_json(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
_write_source_image(tmp_path, "0-FULLBODY/scene.png", color="red")
_write_project_manifest(tmp_path, [_manifest_row("0-FULLBODY/scene.png")])
export_root = _export_root(tmp_path)
_write_image(export_root / "fullbody" / "scene__orig.png", color="red")
(export_root / "fullbody" / "scene__orig.txt").write_text(
"Rook_Kaefer full body",
encoding="utf-8",
)
(export_root / "metadata.jsonl").write_text(
json.dumps(
{
"file_name": "fullbody/scene__orig.png",
"text": "Rook_Kaefer full body",
"subset": "fullbody",
"variant": "orig",
},
sort_keys=True,
)
+ "\n",
encoding="utf-8",
)
_write_image(export_root / TRAINING_IMAGE_GRID_FILENAME, color="blue")
training_workspace = _training_root(tmp_path)
training_workspace.mkdir(parents=True)
(training_workspace / "simpletuner-config.json").write_text(
json.dumps(
{
"pretrained_model_name_or_path": "lodestones/Chroma1-HD",
"pretrained_transformer_model_name_or_path": "/tmp/demo-models/model.safetensors",
"output_dir": str(training_workspace / "_simpletuner-output"),
"logging_dir": "logs",
"hub_model_id": "ladybug-felkin-v5.1",
}
),
encoding="utf-8",
)
(training_workspace / "simpletuner-multidatabackend.json").write_text(
json.dumps(
[
{
"id": "fullbody",
"instance_data_dir": str(
training_workspace / "dataset" / "fullbody"
),
"cache_dir_vae": str(
training_workspace / ".simpletuner-cache" / "vae" / "fullbody"
),
"probability": 1.0,
"crop": False,
}
]
),
encoding="utf-8",
)
(training_workspace / "keep.bin").write_text("training", encoding="utf-8")
(export_root / "README.md").write_text("stale", encoding="utf-8")
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
training:
enabled: true
simpletuner:
model:
pretrained_model_name_or_path: lodestones/Chroma1-HD
publishing:
huggingface:
pretty_name: Ladybug Felkin
version: v5.1
adult_content: true
""",
encoding="utf-8",
)
def fail_if_called(*_args, **_kwargs):
raise AssertionError("README-only command called a write workflow")
monkeypatch.setattr(cli_dataset, "export_training_dataset", fail_if_called)
monkeypatch.setattr(cli_dataset, "_sync_manifest_for_plan", fail_if_called)
monkeypatch.setattr(cli_dataset, "write_training_image_grid", fail_if_called)
result = invoke_cli(runner, ["dataset", "readme", str(config_path)])
assert result.exit_code == 0
assert f"Wrote {export_root / 'README.md'}" in result.stdout
assert (training_workspace / "keep.bin").read_text(encoding="utf-8") == "training"
assert not (export_root / "_TRAINING").exists()
readme = (export_root / "README.md").read_text(encoding="utf-8")
assert "stale" not in readme
assert readme.index("## Content Notice") < readme.index("
= 5 and line.split("|")[4].strip() for line in subset_rows
)
assert "pretrained_model_name_or_path: lodestones/Chroma1-HD" in readme
assert "pretrained_transformer_model_name_or_path" not in readme
assert "output_dir" not in readme
assert "logging_dir" not in readme
assert "instance_data_dir" not in readme
assert "/tmp/demo-models" not in readme
assert "TODO: specify before publishing" not in readme
def test_dataset_sync_cli_rejects_duplicate_yaml_keys(tmp_path: Path) -> None:
config_path = _config_path(tmp_path)
config_path.write_text(
"""
mappings:
fullbody:
- "0-FULLBODY"
mappings:
details:
- "2-HEAD"
""",
encoding="utf-8",
)
result = invoke_cli(runner, ["dataset", "sync", str(config_path)])
assert result.exit_code == 1
assert "[ready | dataset sync]" in result.stderr
assert "Duplicate YAML key 'mappings'" in result.stderr