perf(modelscope): fetch a repository's model card once per run

A collection repository publishes many model files under a single source id,
but enrichment re-read the README and the model-detail payload for every one
of them: eight checkpoints meant sixteen HTTP requests, each detail payload
being 10-22 KB of JSON.

Add `ModelSourceCache`, created by `execute_skill()` for the duration of a
run and passed to the provider through a new optional `cache` argument on
`fetch_model_card_context()`. The agent caches the README (repository-wide
and provider-agnostic), and ModelScope caches its detail payload under a
provider-namespaced key.

Only successful reads are memoised, so a transient failure is still retried
for the next file, and the per-file selection is redone from the cached
payload so a checkpoint never inherits a sibling's example images. Nothing
is retained across runs — a model card can change at any time — and download
URLs are not routed through the cache.

Measured over the eight checkpoints of one ModelScope repository: 16
requests before, 2 after.

To keep the two concerns separable, `_build_card_context()` now turns a
detail payload into a `ModelCardContext` as a pure function.
This commit is contained in:
Will Miao
2026-09-14 21:24:57 +08:00
parent f0ee30fc68
commit e9e9ee20c6
7 changed files with 247 additions and 26 deletions
+26 -4
View File
@@ -28,6 +28,7 @@ from ...config import config
from ..llm_service import LLMService
from ..model_sources import (
ModelCardContext,
ModelSourceCache,
get_source,
resolve_source_ref,
source_label,
@@ -259,6 +260,11 @@ class AgentService:
llm = await self._ensure_llm()
llm_configured = llm.is_configured() if skill.llm_required else True
# A collection repository holds many model files under one source id;
# this memo keeps the README and the repository metadata from being
# re-fetched once per file. It lives for this run only.
source_cache = ModelSourceCache()
for model_path in model_paths:
model_filename = os.path.basename(model_path)
logger.info(
@@ -288,7 +294,7 @@ class AgentService:
# or not an LLM is available: a user without a key still gets
# the author summary, the example images and the tags.
source_vars, source_context = await self._load_source_card(
model_path, metadata,
model_path, metadata, cache=source_cache,
)
resolved_base_model = ""
if skill_name == "enrich_hf_metadata" and not (
@@ -423,13 +429,22 @@ class AgentService:
return "\n".join(f"- {m}" for m in models)
async def _load_source_card(
self, model_path: str, metadata: Dict[str, Any]
self,
model_path: str,
metadata: Dict[str, Any],
*,
cache: Optional[ModelSourceCache] = None,
) -> tuple[Dict[str, Any], ModelCardContext]:
"""Fetch the model card and site-published extras for one model.
Runs for every source-backed enrichment regardless of LLM
availability, because everything it returns is deterministic data that
should be applied even without a configured provider.
*cache* is the per-run memo created by :meth:`execute_skill`. The
README is repository-wide, so it is fetched once per source id; only
successful reads are memoised, leaving a transient failure to be
retried for the next file.
"""
variables: Dict[str, Any] = {
@@ -450,11 +465,18 @@ class AgentService:
raw_basename = os.path.splitext(os.path.basename(model_path))[0]
variables["asset_base_url"] = source.asset_base_url(ref.source_id)
readme = await source.fetch_model_card(ref.source_id)
cache_key = f"{ref.platform}:{ref.source_id}"
readme = cache.readmes.get(cache_key) if cache is not None else None
if readme is None:
readme = await source.fetch_model_card(ref.source_id)
if cache is not None and readme:
cache.readmes[cache_key] = readme
# Sites such as ModelScope keep part of the model card outside the
# README (author summary, curated tags, per-file example images).
card_context = await source.fetch_model_card_context(
ref.source_id, os.path.basename(model_path)
ref.source_id, os.path.basename(model_path), cache=cache,
)
variables["source_description"] = card_context.description
variables["source_base_model"] = card_context.base_model