fix(organize): stop keyword-dump tags from becoming folder names (#1119)

CivitAI tags are normally short single-concept labels, but some uploaders
pack their entire keyword list into one tag. The model in #1119 carries
"lora, character, rosie, irish, ... face" as a single 181-character tag.
Priority resolution matches aliases by exact equality, so that tag matched
nothing and resolve_priority_tag_for_model fell back to tags[0] -- the blob.
With the default "{base_model}/{first_tag}" template the model was filed
under "Krea 2/<181-character blob>/", and the full path plus the
".civitai.info" sidecar and the preview images next to it ran into the
Windows MAX_PATH limit.

Tags also bypassed sanitization on the way into a path: both
calculate_relative_path_for_model and DownloadManager._calculate_relative_path
sanitized model_name and version_name but interpolated {first_tag} verbatim,
so a tag containing "/" or ":" silently produced nested or illegal folders.

Two changes:

- The fallback skips tags that cannot serve as a folder name.
  is_usable_path_tag rejects comma-separated keyword dumps and tags longer
  than MAX_PATH_TAG_LENGTH; the resolver returns "" when nothing usable is
  left, which callers already render as "no tags". Whole-tag priority
  matching is untouched, so existing priority configurations behave the
  same.
- sanitize_folder_name gains an optional max_length, and every tag-derived
  segment now goes through it. Tags are capped at MAX_PATH_TAG_LENGTH, model
  and version names at MAX_FOLDER_NAME_LENGTH, and rendered filename stems at
  MAX_FILENAME_STEM_LENGTH.

For the reported model the folder becomes "Krea 2/base model" instead of the
blob, and the full path drops from 235 to 64 characters.

Existing libraries are not migrated up front: a path is only recomputed on
download, on an auto-organize run or when a filename template is applied, and
values already inside the caps are left byte-identical. Models previously
filed under a keyword-dump folder move on the next auto-organize run.
This commit is contained in:
Will Miao
2026-09-23 13:16:26 +08:00
parent 521531111a
commit 0ada32d0c7
9 changed files with 318 additions and 16 deletions
+14 -2
View File
@@ -25,6 +25,8 @@ from ..utils.models import (
)
from ..utils.constants import (
CARD_PREVIEW_WIDTH,
MAX_FOLDER_NAME_LENGTH,
MAX_PATH_TAG_LENGTH,
MODEL_WEIGHT_FILE_TYPES,
SUPPORTED_DOWNLOAD_SKIP_BASE_MODELS,
VALID_LORA_TYPES,
@@ -2327,16 +2329,26 @@ class DownloadManager:
if not first_tag:
first_tag = "no tags" # Default if no tags available
# Tags come straight from CivitAI, so sanitize the value before it
# becomes a path segment and cap its length (#1119).
first_tag = sanitize_folder_name(first_tag, max_length=MAX_PATH_TAG_LENGTH)
# Format the template with available data
formatted_path = path_template
formatted_path = formatted_path.replace("{base_model}", mapped_base_model)
formatted_path = formatted_path.replace("{first_tag}", first_tag)
formatted_path = formatted_path.replace("{author}", author)
formatted_path = formatted_path.replace(
"{model_name}", sanitize_folder_name(model_info.get("name", ""))
"{model_name}",
sanitize_folder_name(
model_info.get("name", ""), max_length=MAX_FOLDER_NAME_LENGTH
),
)
formatted_path = formatted_path.replace(
"{version_name}", sanitize_folder_name(version_info.get("name", ""))
"{version_name}",
sanitize_folder_name(
version_info.get("name", ""), max_length=MAX_FOLDER_NAME_LENGTH
),
)
if model_type == "embedding":
+6 -2
View File
@@ -47,6 +47,7 @@ from ..utils.settings_paths import (
from ..utils.tag_priorities import (
PriorityTagEntry,
collect_canonical_tags,
is_usable_path_tag,
parse_priority_tag_string,
resolve_priority_tag,
)
@@ -1569,9 +1570,12 @@ class SettingsManager:
if resolved:
return resolved
# Fall back to the first tag that is usable as a folder name. The raw
# tag list can contain keyword dumps that would become unusable folders
# and break path length limits, so skip those (#1119).
for tag in tags:
if isinstance(tag, str) and tag:
return tag
if is_usable_path_tag(tag):
return tag.strip()
return ""
def get_priority_tag_suggestions(self) -> Dict[str, List[str]]:
+13
View File
@@ -293,6 +293,19 @@ DEFAULT_DOWNLOAD_PATH_TEMPLATES: Dict[str, str] = {
"other": "",
}
# Length guards for template placeholders that end up in file and folder names.
# Windows enforces MAX_PATH (260 characters) on the full path and 255 on a
# single path component, and a model folder also has to leave room for the
# model file, its ".civitai.info"/".json" sidecars and preview images. Values
# stay well below those limits so the surrounding files still fit.
#
# Tags get a much tighter budget than other names: some CivitAI uploaders dump
# their whole keyword list into a single tag (see issue #1119), and such a tag
# is only useful as a folder name after truncation.
MAX_FOLDER_NAME_LENGTH = 100
MAX_PATH_TAG_LENGTH = 50
MAX_FILENAME_STEM_LENGTH = 150
# baseModel values from CivitAI that should be treated as diffusion models (unet)
# These model types are incorrectly labeled as "checkpoint" by CivitAI but are actually diffusion models
DIFFUSION_MODEL_BASE_MODELS = frozenset(
+26
View File
@@ -5,6 +5,8 @@ from __future__ import annotations
from dataclasses import dataclass
from typing import Dict, Iterable, List, Optional, Sequence, Set
from .constants import MAX_PATH_TAG_LENGTH
@dataclass(frozen=True)
class PriorityTagEntry:
@@ -102,3 +104,27 @@ def collect_canonical_tags(entries: Iterable[PriorityTagEntry]) -> List[str]:
"""Return the ordered list of canonical tags from the parsed entries."""
return [entry.canonical for entry in entries]
def is_usable_path_tag(tag: object) -> bool:
"""Return True when a tag is a sane single-concept folder-name candidate.
CivitAI tags are normally short labels ("character", "anime"), but some
uploaders dump their whole keyword list into a single tag, e.g.
``"lora, character, rosie, irish, ... face"``. Using such a tag as a folder
name produces unwieldy and path-length-breaking directories (#1119), so
tag-derived path segments only accept single-concept tags.
"""
if not isinstance(tag, str):
return False
candidate = tag.strip()
if not candidate:
return False
# Commas mean the tag is a keyword dump rather than one concept.
if "," in candidate:
return False
return len(candidate) <= MAX_PATH_TAG_LENGTH
+45 -11
View File
@@ -6,6 +6,11 @@ from typing import Any, Dict, List, Optional
from ..services.service_registry import ServiceRegistry
from ..config import config
from ..services.settings_manager import get_settings_manager
from .constants import (
MAX_FILENAME_STEM_LENGTH,
MAX_FOLDER_NAME_LENGTH,
MAX_PATH_TAG_LENGTH,
)
import asyncio
logger = logging.getLogger(__name__)
@@ -417,12 +422,17 @@ def fuzzy_match(text: str, pattern: str, threshold: float = 0.85) -> bool:
return True
def sanitize_folder_name(name: str, replacement: str = "_") -> str:
def sanitize_folder_name(
name: str, replacement: str = "_", max_length: Optional[int] = None
) -> str:
"""Sanitize a folder name by removing or replacing invalid characters.
Args:
name: The original folder name.
replacement: The character to use when replacing invalid characters.
max_length: Optional maximum length for the resulting name. Longer
names are truncated (and re-trimmed) so that a single untrusted
value cannot blow past filesystem path limits.
Returns:
A sanitized folder name safe to use across common filesystems.
@@ -449,6 +459,15 @@ def sanitize_folder_name(name: str, replacement: str = "_") -> str:
# If no replacement, just strip spaces and dots from right, spaces from left
sanitized = sanitized.rstrip(" .").lstrip(" ")
if max_length is not None and max_length > 0 and len(sanitized) > max_length:
sanitized = sanitized[:max_length]
# Re-trim separators and spaces exposed by the cut so the truncated
# name stays filesystem-safe.
if replacement:
sanitized = sanitized.rstrip(" ." + replacement).lstrip(" " + replacement)
else:
sanitized = sanitized.rstrip(" .").lstrip(" ")
if not sanitized:
return "unnamed"
@@ -575,12 +594,20 @@ def calculate_relative_path_for_model(
if not first_tag:
first_tag = "no tags" # Default if no tags available
# Tags are user-generated on CivitAI, so sanitize the value before it
# becomes a path segment and cap its length (#1119).
first_tag = sanitize_folder_name(first_tag, max_length=MAX_PATH_TAG_LENGTH)
# Format the template with available data
model_name = sanitize_folder_name(model_data.get("model_name", ""))
model_name = sanitize_folder_name(
model_data.get("model_name", ""), max_length=MAX_FOLDER_NAME_LENGTH
)
version_name = ""
if isinstance(civitai_data, dict):
version_name = sanitize_folder_name(civitai_data.get("name") or "")
version_name = sanitize_folder_name(
civitai_data.get("name") or "", max_length=MAX_FOLDER_NAME_LENGTH
)
formatted_path = path_template
formatted_path = formatted_path.replace("{base_model}", mapped_base_model)
@@ -667,20 +694,22 @@ def calculate_filename_for_model(
else:
original_name = os.path.splitext(str(model_data.get("file_name", "")))[0]
def _sanitize_value(value: Any) -> str:
def _sanitize_value(value: Any, max_length: Optional[int] = None) -> str:
# sanitize_folder_name falls back to "unnamed" for empty input; for
# templates an empty value must stay empty so segments collapse.
text = str(value) if value else ""
return sanitize_folder_name(text) if text else ""
if not text:
return ""
return sanitize_folder_name(text, max_length=max_length)
replacements = {
"{model_name}": _sanitize_value(model_name),
"{version_name}": _sanitize_value(version_name),
"{base_model}": _sanitize_value(mapped_base_model),
"{author}": _sanitize_value(author),
"{first_tag}": _sanitize_value(first_tag),
"{model_name}": _sanitize_value(model_name, MAX_FILENAME_STEM_LENGTH),
"{version_name}": _sanitize_value(version_name, MAX_FILENAME_STEM_LENGTH),
"{base_model}": _sanitize_value(mapped_base_model, MAX_FILENAME_STEM_LENGTH),
"{author}": _sanitize_value(author, MAX_FILENAME_STEM_LENGTH),
"{first_tag}": _sanitize_value(first_tag, MAX_PATH_TAG_LENGTH),
"{hash_short}": hash_short,
"{original_name}": _sanitize_value(original_name),
"{original_name}": _sanitize_value(original_name, MAX_FILENAME_STEM_LENGTH),
}
result = template
@@ -699,6 +728,11 @@ def calculate_filename_for_model(
# A stem must not start or end with separators, spaces or dots.
result = result.strip("-_. ")
# A template can concatenate several values, so cap the rendered stem as
# well and re-trim the cut.
if len(result) > MAX_FILENAME_STEM_LENGTH:
result = result[:MAX_FILENAME_STEM_LENGTH].strip("-_. ")
return result