feat(download): support ModelScope repositories in the URL downloader

ModelScope became a linkable source, but downloading from it was impossible:
the URL picker only recognised huggingface.co, the file listing hit a
huggingface-only endpoint, the resolve URL was hardcoded, and the default
path template always wrote into a `huggingface/` directory.

Move the download knowledge into the providers so the handlers stay generic:

- `ModelSource` gains `list_files()`, `file_download_url()`,
  `default_revision` and `default_subdir`. `HuggingFaceSource` keeps the Hub
  tree API (`/api/models/{id}/tree/{rev}`, LFS-aware sizes, `main`).
  `ModelScopeSource` uses `/api/v1/models/{id}/repo/files?Revision=master`
  — which reports real byte sizes for LFS files, so no HEAD probe is needed,
  and which only accepts `master` (an HF-imported repo still 404s on `main`)
  — and downloads through `/models/{id}/resolve/{rev}/{path}`. That URL
  redirects to a CDN target carrying a time-limited `auth_key`, so it is
  rebuilt on every request and never cached, which is also what keeps
  resumable Range requests working.
- `hf_handlers.py`/`HfHandler` become `model_source_handlers.py`/
  `ModelSourceHandler` with `list_model_source_files` and
  `download_model_source`. New routes `/api/lm/model-source-files` and
  `/api/lm/download-model-source`; the old `/api/lm/hf-repo-files` and
  `/api/lm/download-hf-model` paths stay as aliases, and a payload without
  `platform` still means Hugging Face, so existing callers are unaffected.
- A downloaded sidecar now records `source_platform` + `source_url` (with the
  `hf_url` alias only for Hugging Face) instead of always writing `hf_url`,
  and `use_default_paths` files ModelScope downloads under
  `modelscope/<owner>/<repo>`. The now-unused shared HF aiohttp session and
  its shutdown hook are gone; providers open short-lived sessions.
- Frontend: `detectUrlType` returns the platform-neutral
  `model-source-repo` / `model-source-file` plus an explicit `platform`, the
  DownloadManager's `hf*` state and methods are renamed to `source*`, every
  `source === 'huggingface'` check becomes `isExternalModelSource()`, and
  batch groups are keyed by `platform:repo` so the same `owner/name` on two
  sites renders as two groups. A bare `owner/name` still means Hugging Face.
- `is_valid_source_id()` centralises repo-id validation (exactly
  `owner/name`, no traversal, no leading dot). This also fixes the old HF
  download check that rejected any dot in the name, i.e. legitimate repos
  such as `black-forest-labs/FLUX.1-dev`.

Verified against the live APIs: the example repo lists 8 weight files with
correct sizes, and a ranged GET of the built resolve URL returns 206 after
following the redirect to the CDN. Backend 2853 passed; frontend 1143 JS +
91 Vue passed. The nine locales carry the refreshed download copy in the
next commit.
This commit is contained in:
Will Miao
2026-09-14 07:42:51 +08:00
parent b9bf006998
commit 38d4c59b4c
22 changed files with 1953 additions and 667 deletions
-7
View File
@@ -472,12 +472,5 @@ class LoraManager:
scanner.cancel_task()
logger.debug("LoRA Manager: Cancelled %s", name)
# Close shared aiohttp sessions to avoid "Unclosed client session" warnings
try:
from py.routes.handlers.hf_handlers import close_hf_api_session
await close_hf_api_session()
except Exception as exc:
logger.debug("Error closing HF API session: %s", exc)
except Exception as e:
logger.error(f"Error during cleanup: {e}", exc_info=True)
+10 -7
View File
@@ -55,7 +55,7 @@ from ...utils.constants import (
VALID_LORA_TYPES,
VALID_OTHER_CIVITAI_TYPES,
)
from .hf_handlers import HfHandler
from .model_source_handlers import ModelSourceHandler
from .agent_handlers import AgentHandler
from .download_routing_handlers import DownloadRoutingHandler
from .model_handlers import ModelCivitaiHandler
@@ -4001,7 +4001,7 @@ class MiscHandlerSet:
doctor: DoctorHandler,
example_workflows: ExampleWorkflowsHandler,
base_model: BaseModelHandlerSet,
hf_handler: Any = None,
model_source_handler: Any = None,
agent_handler: Any = None,
download_routing: Any = None,
) -> None:
@@ -4022,7 +4022,7 @@ class MiscHandlerSet:
self.doctor = doctor
self.example_workflows = example_workflows
self.base_model = base_model
self.hf_handler = hf_handler
self.model_source_handler = model_source_handler
self.agent_handler = agent_handler
self.download_routing = download_routing
@@ -4076,10 +4076,13 @@ class MiscHandlerSet:
"get_example_workflows": self.example_workflows.get_example_workflows,
"get_example_workflow": self.example_workflows.get_example_workflow,
# Hugging Face handlers
"get_hf_repo_files": self.hf_handler.get_hf_repo_files,
"download_hf_model": self.hf_handler.download_hf_model,
"set_hf_url": self.hf_handler.set_hf_url,
"get_model_sources": self.hf_handler.get_model_sources,
# External model sources (Hugging Face / ModelScope)
"list_model_source_files": self.model_source_handler.list_model_source_files,
"download_model_source": self.model_source_handler.download_model_source,
"get_hf_repo_files": self.model_source_handler.list_model_source_files,
"download_hf_model": self.model_source_handler.download_model_source,
"set_hf_url": self.model_source_handler.set_hf_url,
"get_model_sources": self.model_source_handler.get_model_sources,
# Agent skill handlers
"get_agent_skills": self.agent_handler.get_agent_skills,
"execute_agent_skill": self.agent_handler.execute_agent_skill,
@@ -1,8 +1,13 @@
"""Handlers for Hugging Face model listing and download.
"""Handlers for external model sources: linking, file listing and downloads.
Minimal MVP implementation uses direct HTTP to the HF API for file
listing and the project's existing aiohttp-based Downloader for
downloading. No huggingface_hub dependency required.
Covers every site registered in :mod:`py.services.model_sources`. The module
was Hugging Face only (``hf_handlers.py`` / ``HfHandler``) until ModelScope
downloads were added; the per-site differences now live in the providers, so
this file has no platform branches beyond the capability lookups.
The historical route paths (``/api/lm/set-hf-url``, ``/api/lm/hf-repo-files``,
``/api/lm/download-hf-model``) are still registered as aliases of the generic
handlers, so existing callers keep working.
"""
from __future__ import annotations
@@ -10,10 +15,8 @@ from __future__ import annotations
import json
import logging
import os
import re
from typing import Any
import aiohttp
from aiohttp import web
from ...config import config
@@ -23,14 +26,17 @@ from ...services.downloader import (
)
from ...services.aria2_downloader import Aria2Downloader
from ...services.model_sources import (
ModelSourceError,
SourceRef,
detect_source,
get_download_source,
is_valid_source_id,
list_sources,
normalize_metadata_source,
)
from ...services.settings_manager import get_settings_manager
from ...services.service_registry import ServiceRegistry
from ...services.websocket_manager import ws_manager
from ...utils.constants import MODEL_FILE_EXTENSIONS
from ...utils.metadata_manager import MetadataManager
from ...utils.models import LoraMetadata, CheckpointMetadata, EmbeddingMetadata
@@ -39,28 +45,6 @@ logger = logging.getLogger(__name__)
_DEFAULT_MODEL_CLASS = LoraMetadata
_DEFAULT_SCANNER_GETTER = "get_lora_scanner"
# Shared aiohttp session for HF API calls (created on first use)
_hf_api_session: aiohttp.ClientSession | None = None
async def _get_hf_api_session() -> aiohttp.ClientSession:
"""Get or create the shared aiohttp session for HF API calls."""
global _hf_api_session # needed because we reassign the module-level name
if _hf_api_session is None or _hf_api_session.closed:
_hf_api_session = aiohttp.ClientSession(
headers={"User-Agent": "ComfyUI-LoRA-Manager/1.0"},
timeout=aiohttp.ClientTimeout(total=30),
)
return _hf_api_session
async def close_hf_api_session() -> None:
"""Close the shared HF API session, if it was ever created."""
global _hf_api_session
if _hf_api_session is not None and not _hf_api_session.closed:
await _hf_api_session.close()
_hf_api_session = None
def _infer_model_type(model_root: str) -> tuple[Any, str]:
"""Determine model class and scanner by matching ``model_root`` against the
@@ -101,18 +85,19 @@ def _infer_model_type(model_root: str) -> tuple[Any, str]:
return _DEFAULT_MODEL_CLASS, _DEFAULT_SCANNER_GETTER
async def _save_hf_metadata(dest_path: str, repo: str, model_root: str) -> None:
async def _save_source_metadata(
dest_path: str, ref: SourceRef, model_root: str
) -> None:
"""Create a proper .metadata.json and add the model to the scanner cache.
Uses ``MetadataManager.create_default_metadata()`` which computes the
SHA256 hash, extracts safetensors header metadata (base_model), and
produces a fully-populated ``LoraMetadata`` (or ``CheckpointMetadata`` /
``EmbeddingMetadata``) object. We then overlay HF-specific fields and
register the model in the in-memory scanner cache so it appears
``EmbeddingMetadata``) object. We then overlay the external-source fields
and register the model in the in-memory scanner cache so it appears
immediately without a full filesystem walk.
"""
try:
hf_url = f"https://huggingface.co/{repo}"
model_class, scanner_getter_name = _infer_model_type(model_root)
# 1. Create proper metadata (computes SHA256, reads safetensors headers)
@@ -123,15 +108,21 @@ async def _save_hf_metadata(dest_path: str, repo: str, model_root: str) -> None:
logger.warning("create_default_metadata returned None for %s", dest_path)
return
# 2. Overlay HF-specific fields
metadata._unknown_fields["hf_url"] = hf_url
metadata._unknown_fields["source_url"] = hf_url
metadata._unknown_fields["source_platform"] = "huggingface"
metadata.from_civitai = False # HF models are not from CivitAI
# 2. Overlay the external-source fields (`hf_url` is written by
# normalisation for Hugging Face only)
fields = metadata._unknown_fields
fields["source_url"] = ref.url
fields["source_platform"] = ref.platform
if ref.platform == "huggingface":
fields["hf_url"] = ref.url
metadata.from_civitai = False # externally-sourced models are not from CivitAI
# 3. Save metadata atomically
await MetadataManager.save_metadata(dest_path, metadata)
logger.info("Saved HF metadata (with hf_url) for %s", dest_path)
logger.info(
"Saved %s metadata (source=%s) for %s",
ref.platform, ref.url, dest_path,
)
# 4. Determine relative folder path for cache
# model_root is an absolute path; dest_path is under it
@@ -145,13 +136,12 @@ async def _save_hf_metadata(dest_path: str, repo: str, model_root: str) -> None:
if scanner_getter is not None:
scanner = await scanner_getter()
if scanner is not None:
metadata_dict = metadata.to_dict()
metadata_dict["hf_url"] = hf_url
metadata_dict = normalize_metadata_source(metadata.to_dict())
await scanner.add_model_to_cache(metadata_dict, folder)
logger.info("Added %s to scanner cache (folder=%s)", dest_path, folder)
except Exception as exc:
logger.warning("Failed to save HF metadata for %s: %s", dest_path, exc)
logger.warning("Failed to save source metadata for %s: %s", dest_path, exc)
def _find_matching_root(dest_dir: str) -> str | None:
@@ -193,14 +183,23 @@ async def _add_to_scanner_cache(dest_path: str, metadata: dict[str, Any]) -> Non
await scanner.update_single_model_cache(dest_path, dest_path, metadata)
class HfHandler:
"""Handle Hugging Face model browsing and download."""
def _unsupported_platform_error(platform: str) -> web.Response:
supported = ", ".join(source.label for source in list_sources() if source.supports_download)
return web.json_response(
{"error": f"'{platform}' does not support downloads. Supported: {supported}"},
status=400,
)
class ModelSourceHandler:
"""Handle external model browsing, linking and downloads."""
async def get_model_sources(self, request: web.Request) -> web.Response:
"""List the external model sites the UI can link a model to.
Used by the "Link Model" dialog to validate URLs client-side and to
explain which sites support AI metadata enrichment.
Used by the "Link Model" dialog to validate URLs client-side, to
explain which sites support AI metadata enrichment, and to pick the
right download endpoint/revision.
"""
return web.json_response([
@@ -209,6 +208,7 @@ class HfHandler:
"label": source.label,
"supports_enrichment": source.supports_enrichment,
"supports_download": source.supports_download,
"default_revision": source.default_revision,
"example_url": source.canonical_url(
"user/repo" if source.platform != "tensorart" else "827823520299086029"
),
@@ -219,10 +219,12 @@ class HfHandler:
async def set_hf_url(self, request: web.Request) -> web.Response:
"""Link a model file to its page on an external model site.
Accepts ``source_url`` (preferred) or the legacy ``hf_url`` /
``url`` payload key. Hugging Face, ModelScope, and TensorArt URLs
are recognised; the platform is stored alongside the canonical URL.
TensorArt models can be linked and browsed, but not AI-enriched.
Accepts ``source_url`` (preferred) or the legacy ``hf_url`` / ``url``
payload key. Every registered site is recognised and the platform is
stored alongside the canonical URL. TensorArt models can be linked and
browsed, but not AI-enriched.
The route path keeps its historical ``set-hf-url`` name.
"""
try:
@@ -337,74 +339,60 @@ class HfHandler:
status=500,
)
async def get_hf_repo_files(self, request: web.Request) -> web.Response:
"""List model-weight files from a HF repo with real file sizes.
async def list_model_source_files(self, request: web.Request) -> web.Response:
"""List the downloadable weight files of an external repository.
Uses the HF tree API endpoint which returns accurate file sizes
(including LFS-tracked files), unlike the model info endpoint.
Query params: ``platform``, ``repo`` (``owner/name``), ``revision``
(optional; each site has its own default branch).
Returns a JSON array of ``{"filename", "size"}``, largest first
the same shape the Hugging Face endpoint has always returned.
"""
repo = request.query.get("repo", "").strip()
if not repo or "/" not in repo:
platform = (request.query.get("platform") or "").strip()
repo = (request.query.get("repo") or "").strip()
revision = (request.query.get("revision") or "").strip()
source = get_download_source(platform)
if source is None:
return _unsupported_platform_error(platform)
if not is_valid_source_id(repo):
return web.json_response(
{"error": "Missing or invalid 'repo' parameter (expected user/repo)"},
{"error": "Missing or invalid 'repo' parameter (expected owner/name)"},
status=400,
)
url = f"https://huggingface.co/api/models/{repo}/tree/main"
try:
session = await _get_hf_api_session()
async with session.get(url) as resp:
if resp.status == 404:
return web.json_response(
{"error": f"Repo '{repo}' not found"}, status=404
)
if resp.status != 200:
text = await resp.text()
return web.json_response(
{"error": f"HF API error {resp.status}: {text[:200]}"},
status=resp.status,
)
tree: list[dict[str, Any]] = await resp.json()
files = await source.list_files(repo, revision)
except ModelSourceError as exc:
return web.json_response({"error": str(exc)}, status=exc.status)
except Exception as exc:
logger.error("Failed to fetch HF repo files: %s", exc)
logger.error("Failed to list %s files in %s: %s", platform, repo, exc)
return web.json_response({"error": str(exc)}, status=502)
files: list[dict[str, Any]] = []
for entry in tree:
path: str = entry.get("path", "")
ext = os.path.splitext(path)[1].lower()
if ext not in MODEL_FILE_EXTENSIONS:
continue
size = entry.get("size", 0) or 0
if size == 0 and "lfs" in entry:
size = entry["lfs"].get("size", 0) or 0
files.append({
"filename": path,
"size": size,
})
files.sort(key=lambda f: f["size"], reverse=True)
return web.json_response(files)
async def download_hf_model(self, request: web.Request) -> web.Response:
"""Download a single file from Hugging Face into the model directory.
async def download_model_source(self, request: web.Request) -> web.Response:
"""Download a single file from an external repository.
POST JSON body::
{
"repo": "dx8152/Flux2-Klein-9B-Consistency",
"filename": "Flux2-Klein-9B-consistency-V2.safetensors",
"revision": "main",
"platform": "modelscope",
"repo": "owner/name",
"filename": "subdir/model.safetensors",
"revision": "master",
"model_root": "loras",
"relative_path": "",
"use_default_paths": false,
"download_id": "optional-batch-id"
}
``platform`` defaults to ``huggingface`` when omitted, which keeps the
legacy ``/api/lm/download-hf-model`` payload working unchanged.
If ``download_id`` is provided, real-time progress (bytes, speed,
percentage) is broadcast via the WebSocket progress system, matching
the CivitAI download experience.
percentage) is broadcast via the WebSocket progress system.
Respects the ``download_backend`` setting (``aria2`` or ``default``).
"""
@@ -413,30 +401,33 @@ class HfHandler:
except json.JSONDecodeError:
return web.json_response({"error": "Invalid JSON"}, status=400)
platform = (payload.get("platform") or "huggingface").strip()
repo = (payload.get("repo") or "").strip()
filename = (payload.get("filename") or "").strip()
revision = (payload.get("revision") or "main").strip()
revision = (payload.get("revision") or "").strip()
model_root = (payload.get("model_root") or "").strip()
relative_path = (payload.get("relative_path") or "").strip()
use_default_paths = bool(payload.get("use_default_paths", False))
download_id: str | None = payload.get("download_id")
logger.info(
"download_hf_model: repo=%s file=%s root=%s download_id=%s",
repo, filename, model_root, download_id,
"download_model_source: platform=%s repo=%s file=%s root=%s download_id=%s",
platform, repo, filename, model_root, download_id,
)
source = get_download_source(platform)
if source is None:
return _unsupported_platform_error(platform)
if not repo or not filename:
return web.json_response(
{"error": "Missing required fields: 'repo' and 'filename'"}, status=400
)
# Validate repo format — must be user/repo_name
if repo.count("/") != 1 or not re.match(r"^[a-zA-Z0-9_.-]+/[a-zA-Z0-9_.-]+$", repo):
return web.json_response({"error": f"Invalid repo format: {repo}"}, status=400)
author, repo_name = repo.split("/", 1)
if ".." in (author, repo_name) or "." in (author, repo_name):
# `owner/name` only; the components become path segments below.
if not is_valid_source_id(repo):
return web.json_response({"error": f"Invalid repo format: {repo}"}, status=400)
owner, repo_name = repo.split("/", 1)
# Validate filename — must not contain path traversal
if ".." in filename:
@@ -455,21 +446,21 @@ class HfHandler:
# unnecessary when the frontend sends the path from its own dropdown
# (populated from scanner roots). Using the "business path" directly
# keeps dest_path consistent with scanner roots so that later folder
# derivation (in _save_hf_metadata) works correctly.
# derivation (in _save_source_metadata) works correctly.
if os.path.isabs(model_root):
base_dir = os.path.normpath(model_root)
else:
base_dir = os.path.normpath(os.path.join(os.getcwd(), "models", model_root))
if use_default_paths:
target_dir = os.path.join(base_dir, "huggingface", author, repo_name)
target_dir = os.path.join(base_dir, source.default_subdir, owner, repo_name)
elif relative_path:
target_dir = os.path.join(base_dir, relative_path)
else:
target_dir = base_dir
# Strip HF repo subdirectory — "diffusion_models/xxx.safetensors"
# is an HF repo convention, not meaningful for local storage.
# Strip the repository sub-directory — "diffusion_models/xxx.safetensors"
# is a repository convention, not meaningful for local storage.
file_base = os.path.basename(filename)
os.makedirs(target_dir, exist_ok=True)
@@ -477,16 +468,18 @@ class HfHandler:
# Check if already exists (simple skip)
if os.path.exists(dest_path) and os.path.getsize(dest_path) > 0:
logger.info("download_hf_model: file already exists, skipping — %s", dest_path)
logger.info("download_model_source: file already exists, skipping — %s", dest_path)
return web.json_response({
"success": True,
"message": f"File already exists: {dest_path}",
"path": dest_path,
})
# Build HF resolve URL
resolve_url = (
f"https://huggingface.co/{repo}/resolve/{revision}/{filename}"
# Built per request: sites that redirect to a CDN hand out a
# time-limited token in the redirect, so the URL must never be cached.
resolve_url = source.file_download_url(repo, filename, revision)
ref = SourceRef(
platform=source.platform, source_id=repo, url=source.canonical_url(repo)
)
# Set up progress callback if download_id is provided
@@ -528,28 +521,27 @@ class HfHandler:
if download_backend == "aria2":
aria2 = await Aria2Downloader.get_instance()
aid = download_id or f"hf_{repo}_{filename}"
aid = download_id or f"{source.platform}_{repo}_{filename}"
try:
hf_success, hf_result = await aria2.download_file(
ok, result = await aria2.download_file(
url=resolve_url,
save_path=dest_path,
download_id=aid,
progress_callback=progress_callback,
)
if hf_success:
await _save_hf_metadata(dest_path, repo, model_root)
if ok:
await _save_source_metadata(dest_path, ref, model_root)
return web.json_response({
"success": True,
"message": f"Downloaded to {dest_path}",
"path": dest_path,
})
else:
return web.json_response(
{"success": False, "error": hf_result or "aria2 download failed"},
status=500,
)
return web.json_response(
{"success": False, "error": result or "aria2 download failed"},
status=500,
)
except Exception as exc:
logger.error("HF download (aria2) failed: %s", exc)
logger.error("%s download (aria2) failed: %s", platform, exc)
return web.json_response(
{"success": False, "error": str(exc)}, status=500
)
@@ -565,19 +557,18 @@ class HfHandler:
progress_callback=progress_callback,
)
if success:
await _save_hf_metadata(dest_path, repo, model_root)
await _save_source_metadata(dest_path, ref, model_root)
return web.json_response({
"success": True,
"message": f"Downloaded to {result}",
"path": result,
})
else:
return web.json_response(
{"success": False, "error": result or "Download failed"},
status=500,
)
return web.json_response(
{"success": False, "error": result or "Download failed"},
status=500,
)
except Exception as exc:
logger.error("HF download failed: %s", exc)
logger.error("%s download failed: %s", platform, exc)
return web.json_response(
{"success": False, "error": str(exc)}, status=500
)
+8 -1
View File
@@ -99,7 +99,11 @@ MISC_ROUTE_DEFINITIONS: tuple[RouteDefinition, ...] = (
RouteDefinition(
"GET", "/api/lm/delete-model-version", "delete_model_version"
),
# Hugging Face model endpoints
# External model source endpoints (Hugging Face / ModelScope).
# The hf-* paths are the historical names, kept as aliases.
RouteDefinition(
"GET", "/api/lm/model-source-files", "list_model_source_files"
),
RouteDefinition(
"GET", "/api/lm/hf-repo-files", "get_hf_repo_files"
),
@@ -107,6 +111,9 @@ MISC_ROUTE_DEFINITIONS: tuple[RouteDefinition, ...] = (
RouteDefinition(
"POST", "/api/lm/download/routing", "get_download_routing"
),
RouteDefinition(
"POST", "/api/lm/download-model-source", "download_model_source"
),
RouteDefinition(
"POST", "/api/lm/download-hf-model", "download_hf_model"
),
+3 -3
View File
@@ -39,7 +39,7 @@ from .handlers.misc_handlers import (
build_service_registry_adapter,
)
from .handlers.base_model_handlers import BaseModelHandlerSet
from .handlers.hf_handlers import HfHandler
from .handlers.model_source_handlers import ModelSourceHandler
from .handlers.agent_handlers import AgentHandler
from .handlers.download_routing_handlers import DownloadRoutingHandler
from .misc_route_registrar import MiscRouteRegistrar
@@ -139,7 +139,7 @@ class MiscRoutes:
doctor = DoctorHandler(settings_service=self._settings)
example_workflows = ExampleWorkflowsHandler()
base_model = BaseModelHandlerSet()
hf_handler = HfHandler()
model_source_handler = ModelSourceHandler()
agent_handler = AgentHandler()
download_routing = DownloadRoutingHandler()
@@ -161,7 +161,7 @@ class MiscRoutes:
doctor=doctor,
example_workflows=example_workflows,
base_model=base_model,
hf_handler=hf_handler,
model_source_handler=model_source_handler,
agent_handler=agent_handler,
download_routing=download_routing,
)
+12
View File
@@ -12,10 +12,14 @@ from .base import (
GROUP_PREFIXES,
HTTP_TIMEOUT,
ModelSource,
ModelSourceError,
SourceRef,
USER_AGENT,
clean_source_url,
fetch_json,
fetch_text,
filter_weight_files,
is_valid_source_id,
)
from .huggingface import HuggingFaceSource
from .modelscope import ModelScopeSource
@@ -24,6 +28,8 @@ from .registry import (
SOURCE_PLATFORM_FIELD,
SOURCE_URL_FIELD,
detect_source,
downloadable_sources,
get_download_source,
get_source,
get_source_platform,
has_external_source,
@@ -40,6 +46,7 @@ __all__ = [
"HTTP_TIMEOUT",
"LEGACY_HF_URL_FIELD",
"ModelSource",
"ModelSourceError",
"HuggingFaceSource",
"ModelScopeSource",
"SOURCE_PLATFORM_FIELD",
@@ -49,10 +56,15 @@ __all__ = [
"USER_AGENT",
"clean_source_url",
"detect_source",
"downloadable_sources",
"fetch_json",
"fetch_text",
"filter_weight_files",
"get_download_source",
"get_source",
"get_source_platform",
"has_external_source",
"is_valid_source_id",
"list_sources",
"normalize_metadata_source",
"resolve_source_ref",
+128 -1
View File
@@ -20,12 +20,15 @@ HTTP handlers never need site-specific branching.
from __future__ import annotations
import logging
import os
import re
from dataclasses import dataclass
from typing import Any, Optional
from typing import Any, Iterable, Optional
import aiohttp
from ...utils.constants import MODEL_FILE_EXTENSIONS
logger = logging.getLogger(__name__)
#: Shared HTTP timeout for model-card fetches.
@@ -58,6 +61,36 @@ class SourceRef:
"""Canonical URL of the model page."""
class ModelSourceError(Exception):
"""Raised when a model source cannot satisfy a request.
Carries the HTTP status the API handler should answer with, so the
handlers stay free of per-site error mapping.
"""
def __init__(self, message: str, status: int = 502) -> None:
super().__init__(message)
self.status = status
#: Repository ids are always exactly ``owner/name``. Components may contain
#: dots (``black-forest-labs/FLUX.1-dev``) but must not be empty, ``.`` / ``..``,
#: or start with a dot - the id is used as a path segment on disk.
_SOURCE_ID_COMPONENT = re.compile(r"^[A-Za-z0-9_][A-Za-z0-9_.\-]*$")
def is_valid_source_id(source_id: str) -> bool:
"""Return ``True`` when *source_id* is a safe ``owner/name`` repository id."""
if not source_id or not isinstance(source_id, str) or source_id.count("/") != 1:
return False
owner, name = source_id.split("/", 1)
return all(
part and part not in (".", "..") and _SOURCE_ID_COMPONENT.match(part)
for part in (owner, name)
)
async def fetch_text(url: str, *, timeout: int = HTTP_TIMEOUT) -> str:
"""Fetch *url* and return its body as text, or ``""`` on any failure.
@@ -80,6 +113,34 @@ async def fetch_text(url: str, *, timeout: int = HTTP_TIMEOUT) -> str:
return ""
async def fetch_json(
url: str, *, timeout: int = HTTP_TIMEOUT
) -> tuple[int, Any]:
"""Fetch *url* and return ``(status, parsed_body)``.
Unlike :func:`fetch_text` this reports the status, because callers such as
the file-listing endpoints need to distinguish "repo not found" (404) from
a transport failure. ``parsed_body`` is ``None`` when the response is not
JSON or the request failed outright (status ``0``).
"""
try:
async with aiohttp.ClientSession(
headers={"User-Agent": USER_AGENT},
timeout=aiohttp.ClientTimeout(total=timeout),
) as session:
async with session.get(url) as resp:
if resp.status != 200:
return resp.status, None
try:
return resp.status, await resp.json(content_type=None)
except Exception:
return resp.status, None
except Exception as exc: # pragma: no cover - network dependent
logger.debug("Failed to fetch %s: %s", url, exc)
return 0, None
class ModelSource:
"""Description and I/O for one external model hosting site."""
@@ -95,6 +156,12 @@ class ModelSource:
#: Whether models can be downloaded directly from this site.
supports_download: bool = False
#: Branch used when the caller does not pass an explicit revision.
default_revision: str = ""
#: Sub-directory the "use default paths" template places downloads in.
default_subdir: str = ""
#: Lenient pattern used to recognise URLs already stored in metadata.
#: Captures the site-specific source id in group ``id``.
url_pattern: re.Pattern[str] | None = None
@@ -163,6 +230,45 @@ class ModelSource:
return ""
# ------------------------------------------------------------------
# Download support
# ------------------------------------------------------------------
async def list_files(
self, source_id: str, revision: str = ""
) -> list[dict[str, Any]]:
"""List downloadable weight files in *source_id*.
Returns ``[{"filename": <repo-relative path>, "size": <bytes>}]``,
largest first, filtered to :data:`MODEL_FILE_EXTENSIONS`. Sites
without download support return an empty list.
Raises :class:`ModelSourceError` when the repository cannot be read,
so the handler can surface "not found" separately from a transport
failure.
"""
return []
def file_download_url(
self, source_id: str, filename: str, revision: str = ""
) -> str:
"""Return the direct (redirecting) download URL for one file."""
raise ModelSourceError(
f"{self.label or self.platform} does not support downloads", status=400
)
def resolve_revision(self, revision: str = "") -> str:
"""Return *revision*, falling back to this site's default branch."""
return revision or self.default_revision
def page_url_for_file(self, source_id: str, filename: str) -> str:
"""Return the human-facing page for *filename* inside *source_id*."""
return self.canonical_url(source_id)
def __repr__(self) -> str: # pragma: no cover - debugging aid
return f"<ModelSource {self.platform}>"
@@ -175,12 +281,33 @@ def clean_source_url(url: Any) -> str:
return url.strip()
def filter_weight_files(entries: Iterable[tuple[str, int]]) -> list[dict[str, Any]]:
"""Keep model-weight files from ``(path, size)`` pairs, largest first.
Every site lists a lot more than weights (READMEs, configs, tokenizers,
…); the download picker only ever wants the files ComfyUI can load, which
is exactly :data:`MODEL_FILE_EXTENSIONS`.
"""
files = [
{"filename": path, "size": int(size or 0)}
for path, size in entries
if path and os.path.splitext(path)[1].lower() in MODEL_FILE_EXTENSIONS
]
files.sort(key=lambda entry: entry["size"], reverse=True)
return files
__all__ = [
"GROUP_PREFIXES",
"HTTP_TIMEOUT",
"ModelSource",
"ModelSourceError",
"SourceRef",
"USER_AGENT",
"clean_source_url",
"fetch_json",
"fetch_text",
"filter_weight_files",
"is_valid_source_id",
]
+59 -2
View File
@@ -2,9 +2,18 @@
from __future__ import annotations
import logging
import re
from .base import ModelSource, fetch_text
from .base import (
ModelSource,
ModelSourceError,
fetch_json,
fetch_text,
filter_weight_files,
)
logger = logging.getLogger(__name__)
#: Lenient — used to normalise URLs already stored in metadata; tolerates
#: sub-paths such as ``/resolve/main/model.safetensors``.
@@ -25,6 +34,8 @@ class HuggingFaceSource(ModelSource):
label = "Hugging Face"
supports_enrichment = True
supports_download = True
default_revision = "main"
default_subdir = "huggingface"
url_pattern = _URL_PATTERN
strict_url_pattern = _STRICT_URL_PATTERN
@@ -32,7 +43,7 @@ class HuggingFaceSource(ModelSource):
return f"https://huggingface.co/{source_id}"
def asset_base_url(self, source_id: str, revision: str = "") -> str:
return f"https://huggingface.co/{source_id}/resolve/{revision or 'main'}"
return f"https://huggingface.co/{source_id}/resolve/{self.resolve_revision(revision)}"
async def fetch_model_card(self, source_id: str) -> str:
"""Fetch ``README.md`` from Hugging Face (tries ``main``, then ``master``)."""
@@ -45,5 +56,51 @@ class HuggingFaceSource(ModelSource):
return text
return ""
async def list_files(
self, source_id: str, revision: str = ""
) -> list[dict]:
"""List weight files via the Hub tree API.
The tree endpoint (rather than the model-info endpoint) is used
because it reports accurate sizes for LFS-tracked files.
"""
revision = self.resolve_revision(revision)
status, payload = await fetch_json(
f"https://huggingface.co/api/models/{source_id}/tree/{revision}"
)
if status == 404:
raise ModelSourceError(f"Repository '{source_id}' not found", status=404)
if status != 200 or not isinstance(payload, list):
raise ModelSourceError(
f"Hugging Face API error while listing '{source_id}' (HTTP {status})"
)
entries = []
for entry in payload:
if not isinstance(entry, dict):
continue
path = entry.get("path", "")
size = entry.get("size", 0) or 0
if not size and isinstance(entry.get("lfs"), dict):
size = entry["lfs"].get("size", 0) or 0
entries.append((path, size))
return filter_weight_files(entries)
def file_download_url(
self, source_id: str, filename: str, revision: str = ""
) -> str:
return (
f"https://huggingface.co/{source_id}/resolve/"
f"{self.resolve_revision(revision)}/{filename}"
)
def page_url_for_file(self, source_id: str, filename: str) -> str:
return (
f"https://huggingface.co/{source_id}/blob/{self.default_revision}/{filename}"
)
__all__ = ["HuggingFaceSource"]
+73 -5
View File
@@ -2,20 +2,38 @@
ModelScope exposes the same "model card as README.md" convention as
Hugging Face, including a YAML frontmatter block that often carries
``base_model:`` and ``trigger_words:``. Two public endpoints are used,
neither of which requires an API key for public models:
``base_model:`` and ``trigger_words:``. Three public endpoints are used,
none of which requires an API key for public models:
* ``/models/{owner}/{name}/resolve/{revision}/README.md`` — raw model card
* ``/api/v1/models/{owner}/{name}/repo?Revision=..&FilePath=README.md`` —
the same content through the API, used as a fallback when the resolve
URL is unavailable.
* ``/api/v1/models/{owner}/{name}/repo/files?Revision=..`` — the file
listing backing the download picker. It reports real sizes for LFS
files (not the pointer size), so no extra HEAD request is needed.
Downloads go through ``/models/{owner}/{name}/resolve/{revision}/{path}``,
which redirects to a CDN URL carrying a time-limited ``auth_key``.
Requesting the resolve URL fresh on every attempt (which the shared
downloader does, including for resumable Range requests) keeps that key
valid; the CDN URL must never be cached.
"""
from __future__ import annotations
import logging
import re
from .base import ModelSource, fetch_text
from .base import (
ModelSource,
ModelSourceError,
fetch_json,
fetch_text,
filter_weight_files,
)
logger = logging.getLogger(__name__)
_URL_PATTERN = re.compile(
r"https?://(?:www\.)?modelscope\.(?:cn|com)/models/(?P<id>[^/?#\s]+/[^/?#\s]+)"
@@ -41,7 +59,9 @@ class ModelScopeSource(ModelSource):
platform = "modelscope"
label = "ModelScope"
supports_enrichment = True
supports_download = False
supports_download = True
default_revision = "master"
default_subdir = "modelscope"
url_pattern = _URL_PATTERN
strict_url_pattern = _STRICT_URL_PATTERN
@@ -49,7 +69,10 @@ class ModelScopeSource(ModelSource):
return f"https://modelscope.cn/models/{source_id}"
def asset_base_url(self, source_id: str, revision: str = "") -> str:
return f"https://modelscope.cn/models/{source_id}/resolve/{revision or 'master'}"
return (
f"https://modelscope.cn/models/{source_id}/resolve/"
f"{self.resolve_revision(revision)}"
)
async def fetch_model_card(self, source_id: str) -> str:
"""Fetch the model card, preferring the raw resolve URL."""
@@ -72,5 +95,50 @@ class ModelScopeSource(ModelSource):
return text
return ""
async def list_files(
self, source_id: str, revision: str = ""
) -> list[dict]:
"""List weight files via the repo files API.
``master`` is the only branch name the API accepts — even repos
imported from Hugging Face are addressed as ``master`` (``main``
returns 404) — so no fallback probing is done here.
"""
revision = self.resolve_revision(revision)
status, payload = await fetch_json(
"https://modelscope.cn/api/v1/models/"
f"{source_id}/repo/files?Revision={revision}"
)
if status == 404:
raise ModelSourceError(f"Repository '{source_id}' not found", status=404)
if status != 200 or not isinstance(payload, dict):
raise ModelSourceError(
f"ModelScope API error while listing '{source_id}' (HTTP {status})"
)
entries = []
for entry in (payload.get("Data") or {}).get("Files") or []:
if not isinstance(entry, dict) or entry.get("Type") != "blob":
continue
entries.append((entry.get("Path", ""), entry.get("Size", 0) or 0))
return filter_weight_files(entries)
def file_download_url(
self, source_id: str, filename: str, revision: str = ""
) -> str:
return (
f"https://modelscope.cn/models/{source_id}/resolve/"
f"{self.resolve_revision(revision)}/{filename}"
)
def page_url_for_file(self, source_id: str, filename: str) -> str:
return (
f"https://modelscope.cn/models/{source_id}/file/view/"
f"{self.default_revision}/{filename}"
)
__all__ = ["ModelScopeSource"]
+17
View File
@@ -56,6 +56,21 @@ def source_label(platform: Optional[str], default: str = "") -> str:
return source.label if source else default
def downloadable_sources() -> list[ModelSource]:
"""Return the sources whose repositories can be downloaded directly."""
return [source for source in _SOURCES if source.supports_download]
def get_download_source(platform: Optional[str]) -> Optional[ModelSource]:
"""Return the source for *platform*, but only when it supports downloads."""
source = get_source(platform)
if source is None or not source.supports_download:
return None
return source
def detect_source(url: Optional[str], *, strict: bool = False) -> Optional[SourceRef]:
"""Return the :class:`SourceRef` for *url*, or ``None`` if unsupported."""
@@ -197,6 +212,8 @@ __all__ = [
"SOURCE_PLATFORM_FIELD",
"SOURCE_URL_FIELD",
"detect_source",
"downloadable_sources",
"get_download_source",
"get_source",
"get_source_platform",
"has_external_source",