perf(rename): make bulk filename-template apply O(n) instead of O(n^2)

Applying a filename template to a large library re-did O(library) work for
every renamed file: a full natsort resort plus whole-table SQLite rewrite and
download-history resync after each rename, and a full scan plus resort of the
entire recipe collection per renamed LoRA. On a 20k-model library with 300k
recipes on a HDD this pushed "Apply to Library" into multi-day runs.

- ModelScanner.defer_cache_persist(): bulk loops update the in-memory entry
  and indexes only; resort + persist + download-history sync run once at
  context exit, forced even on cancellation/error since files are already
  renamed on disk. Single-rename callers keep immediate per-call behavior.
- RecipeScanner.build_lora_hash_index(): one-shot hash -> recipes index so
  per-file lookups are O(1); update_lora_filename_by_hash gains hash_index /
  defer_maintenance params, with a single finalize_bulk_filename_updates()
  resort at the end of a bulk session.
- ModelLifecycleService.bulk_rename_session() / BulkRenameContext wire the
  deferred path through rename_model (hash index built lazily on first
  recipe-touching rename).
- Blocking os.rename sequence offloaded via asyncio.to_thread so one file's
  HDD I/O no longer stalls the event loop (no cross-file parallelism).
- Skip logic, per-batch WebSocket progress, cancellation, and result
  counters unchanged.
This commit is contained in:
Will Miao
2026-10-03 14:31:10 +08:00
parent 515469054c
commit 896ce5eddb
8 changed files with 1080 additions and 63 deletions
+109
View File
@@ -1903,6 +1903,115 @@ async def test_update_lora_filename_by_hash_updates_affected_recipes(
assert cached1["loras"][0]["file_name"] == new_name
@pytest.mark.asyncio
async def test_update_lora_filename_by_hash_bulk_mode_skips_scan_and_defers_resort(
tmp_path: Path, recipe_scanner, monkeypatch: pytest.MonkeyPatch
):
"""Bulk rename path: prebuilt hash index, no per-call cache walk/resort."""
scanner, _ = recipe_scanner
recipes_dir = Path(config.loras_roots[0]) / "recipes"
recipes_dir.mkdir(parents=True, exist_ok=True)
recipe1_id = "recipe-bulk-1"
recipe1_path = recipes_dir / f"{recipe1_id}.recipe.json"
recipe1_data = {
"id": recipe1_id,
"file_path": str(tmp_path / "bulk1.png"),
"title": "Bulk 1",
"modified": 0.0,
"created_date": 0.0,
"loras": [{"file_name": "old_name", "hash": "hash1"}],
}
recipe1_path.write_text(json.dumps(recipe1_data))
await scanner.add_recipe(dict(recipe1_data))
# Build the index once (O(recipes)), as a bulk rename session does.
hash_index = await scanner.build_lora_hash_index()
assert "hash1" in hash_index
# Spies: the bulk call must not walk the recipe cache again, and must not
# schedule a resort per call.
get_cached_calls = 0
original_get_cached = scanner.get_cached_data
async def counting_get_cached_data(*args, **kwargs):
nonlocal get_cached_calls
get_cached_calls += 1
return await original_get_cached(*args, **kwargs)
monkeypatch.setattr(scanner, "get_cached_data", counting_get_cached_data)
resort_calls = 0
original_schedule = scanner._schedule_resort
def counting_schedule_resort(**kwargs):
nonlocal resort_calls
resort_calls += 1
original_schedule(**kwargs)
monkeypatch.setattr(scanner, "_schedule_resort", counting_schedule_resort)
file_count, cache_count = await scanner.update_lora_filename_by_hash(
"HASH1", "new_name", hash_index=hash_index, defer_maintenance=True
)
assert (file_count, cache_count) == (1, 1)
assert get_cached_calls == 0
assert resort_calls == 0
# The per-match recipe JSON rewrite must still happen.
persisted1 = json.loads(recipe1_path.read_text())
assert persisted1["loras"][0]["file_name"] == "new_name"
cached1 = next(r for r in hash_index["hash1"] if r["id"] == recipe1_id)
assert cached1["loras"][0]["file_name"] == "new_name"
# Deferred maintenance runs exactly once at finalize.
await scanner.finalize_bulk_filename_updates()
await asyncio.sleep(0)
assert resort_calls == 1
@pytest.mark.asyncio
async def test_update_lora_filename_by_hash_with_index_still_resorts_when_not_deferred(
tmp_path: Path, recipe_scanner, monkeypatch: pytest.MonkeyPatch
):
"""hash_index without defer_maintenance: O(1) lookup, immediate resort."""
scanner, _ = recipe_scanner
recipes_dir = Path(config.loras_roots[0]) / "recipes"
recipes_dir.mkdir(parents=True, exist_ok=True)
recipe1_id = "recipe-idx-1"
recipe1_path = recipes_dir / f"{recipe1_id}.recipe.json"
recipe1_data = {
"id": recipe1_id,
"file_path": str(tmp_path / "idx1.png"),
"title": "Index 1",
"modified": 0.0,
"created_date": 0.0,
"loras": [{"file_name": "old_name", "hash": "hash1"}],
}
recipe1_path.write_text(json.dumps(recipe1_data))
await scanner.add_recipe(dict(recipe1_data))
hash_index = await scanner.build_lora_hash_index()
resort_calls = 0
original_schedule = scanner._schedule_resort
def counting_schedule_resort(**kwargs):
nonlocal resort_calls
resort_calls += 1
original_schedule(**kwargs)
monkeypatch.setattr(scanner, "_schedule_resort", counting_schedule_resort)
file_count, cache_count = await scanner.update_lora_filename_by_hash(
"hash1", "new_name", hash_index=hash_index
)
assert (file_count, cache_count) == (1, 1)
assert resort_calls == 1
@pytest.mark.asyncio
async def test_get_paginated_data_filters_by_favorite(recipe_scanner):
scanner, _ = recipe_scanner