From 32a937913ba784dbbd48aca7648bab90bf2fd4b9 Mon Sep 17 00:00:00 2001 From: Andreea Luca Date: Fri, 12 Jun 2026 16:05:49 +0300 Subject: [PATCH 1/5] chore: add doc image cleanup tooling and monthly workflow Add scripts/rename_images.py: scans .gitbook/assets, proposes descriptive caption-derived names for generically-named images, reports usage/captions, and (with --apply) renames + rewrites references and deletes unreferenced orphans. Includes a live re-verify before any delete, an angle-bracket link parser, a caption-relevance check, --dry-run, and --sections scoping. Add .github/workflows/clean-up-images.yml: monthly + manual run that opens a PR for review (never pushes to main), with optional per-section scoping and an orphan-deletion toggle. Ignore the generated mapping CSV / report artifacts. Co-Authored-By: Claude Opus 4.8 (1M context) --- .github/workflows/clean-up-images.yml | 101 ++++ .gitignore | 4 + scripts/rename_images.py | 725 ++++++++++++++++++++++++++ 3 files changed, 830 insertions(+) create mode 100644 .github/workflows/clean-up-images.yml create mode 100644 scripts/rename_images.py diff --git a/.github/workflows/clean-up-images.yml b/.github/workflows/clean-up-images.yml new file mode 100644 index 000000000000..ae92e0b2553a --- /dev/null +++ b/.github/workflows/clean-up-images.yml @@ -0,0 +1,101 @@ +name: Clean up doc images + +# Renames generically-named images (1.png, "Screenshot ….png", …) to +# descriptive, caption-derived names — and deletes images that nothing +# references — then opens a PR for human review. Never pushes to main directly. +# +# Triggers: +# - workflow_dispatch: run on demand (use this for the first big cleanup) +# - schedule: once a month, to keep newly-added images tidy +on: + workflow_dispatch: + inputs: + sections: + description: "Limit to these top-level dirs (comma-separated, e.g. developer-tools). Blank = all (minus excluded)." + type: string + default: "" + exclude_sections: + description: "Top-level dirs to skip (comma-separated). Defaults to 'docs', which is managed separately." + type: string + default: "docs" + delete_orphans: + description: "Delete unreferenced (orphan) images" + type: boolean + default: true + schedule: + - cron: "0 6 1 * *" # 06:00 UTC on the 1st of each month + +permissions: + contents: write + pull-requests: write + +jobs: + clean-up-images: + name: clean-up-images + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + + - name: Scan + build mapping/report + env: + SECTIONS: ${{ github.event.inputs.sections }} + EXCLUDE: ${{ github.event.inputs.exclude_sections }} + run: | + # `docs` is excluded by default (managed separately). On scheduled runs + # github.event.inputs is empty, so fall back to the default here too. + EXCLUDE="${EXCLUDE:-docs}" + python3 scripts/rename_images.py \ + ${SECTIONS:+--sections "$SECTIONS"} \ + ${EXCLUDE:+--exclude-sections "$EXCLUDE"} | tee /tmp/scan.log + + - name: Apply (rename + delete orphans) + id: apply + run: | + # scheduled runs always delete orphans; manual runs honor the toggle + DELETE_FLAG="--delete-orphans" + if [ "${{ github.event_name }}" = "workflow_dispatch" ] && \ + [ "${{ github.event.inputs.delete_orphans }}" = "false" ]; then + DELETE_FLAG="" + fi + python3 scripts/rename_images.py --apply $DELETE_FLAG | tee /tmp/apply.log + echo 'SUMMARY<> "$GITHUB_OUTPUT" + tail -n 1 /tmp/apply.log >> "$GITHUB_OUTPUT" + echo 'EOF' >> "$GITHUB_OUTPUT" + + - name: Upload review report + if: always() + uses: actions/upload-artifact@v4 + with: + name: image-rename-report + path: | + image-rename-report.md + image-rename-mapping.csv + if-no-files-found: ignore + + - uses: peter-evans/create-pull-request@v7 + with: + token: ${{ secrets.GITHUB_TOKEN }} + commit-message: "chore: rename generic doc images and remove orphans" + title: "chore: clean up doc image names and remove orphan images" + body: | + Automated image cleanup from `scripts/rename_images.py`. + + **${{ steps.apply.outputs.SUMMARY }}** + + What this PR does: + - Renames generically-named images (`1.png`, `Screenshot ….png`, `unnamed.png`, …) to descriptive, **caption-derived** names, and rewrites every `.gitbook/assets/` reference in the docs to match. + - Deletes images that nothing in the repo references (orphans). + - The `docs` section is **excluded by default** (managed separately). Override via the `exclude_sections` input on a manual run. + + **Please review before merging:** + - Spot-check a few renames against the rendered pages — names come from each image's caption/alt text. + - Confirm the deleted images really are unused (they were unreferenced at scan time and re-verified at apply time, but a brand-new image added in the same window could look like an orphan). + - The full report (every rename, every deletion, caption sources, and any low-relevance names) is attached to this workflow run as the **image-rename-report** artifact. + branch: chore/image-cleanup + base: main + sign-commits: true + delete-branch: true diff --git a/.gitignore b/.gitignore index 52986e539494..97795ced1811 100644 --- a/.gitignore +++ b/.gitignore @@ -3,3 +3,7 @@ .idea/ error-catalog/ .vscode/settings.json + +# Generated by scripts/rename_images.py (working artifacts, not committed) +image-rename-mapping.csv +image-rename-report.md diff --git a/scripts/rename_images.py b/scripts/rename_images.py new file mode 100644 index 000000000000..72d9e44537d6 --- /dev/null +++ b/scripts/rename_images.py @@ -0,0 +1,725 @@ +#!/usr/bin/env python3 +""" +rename_images.py — find generically-named images in the user-docs repo, work out +what each one depicts (from its captions/alt text and surrounding docs), and +suggest a descriptive, reusable name for it. + +By default the script only *reports*: it writes + + image-rename-report.md human-readable review of every flagged image + image-rename-mapping.csv old_path,old_name,suggested_name,uses,locations + +Review/edit the CSV, then re-run with --apply to actually rename the files and +rewrite every reference to them in the Markdown docs. + + python3 scripts/rename_images.py # report only (safe) + python3 scripts/rename_images.py --all # report ALL images, not just generic ones + python3 scripts/rename_images.py --vision # use Claude vision to name context-less images + python3 scripts/rename_images.py --apply # rename using image-rename-mapping.csv + python3 scripts/rename_images.py --apply --csv my-mapping.csv + +The base script has no third-party dependencies (Python 3.8+). The optional +--vision step additionally requires `pip install anthropic` and an +ANTHROPIC_API_KEY in the environment; it is imported lazily so the rest of the +script runs without it. +""" +from __future__ import annotations + +import argparse +import base64 +import csv +import json +import os +import re +import sys +from collections import Counter, defaultdict +from concurrent.futures import ThreadPoolExecutor, as_completed +from pathlib import Path + +# --------------------------------------------------------------------------- +# Configuration +# --------------------------------------------------------------------------- + +REPO_ROOT = Path(__file__).resolve().parent.parent + +IMAGE_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".svg", ".webp", ".bmp"} + +REPORT_MD = REPO_ROOT / "image-rename-report.md" +MAPPING_CSV = REPO_ROOT / "image-rename-mapping.csv" + +# A filename (stem only, no extension) is considered "generic / non-descriptive" +# if it matches any of these patterns. +GENERIC_PATTERNS = [ + re.compile(r"^\d+$"), # 1, 23, 100 + re.compile(r"^\d+[ _-]*\(\d+\)"), # "1 (1)", "10 (2)" + re.compile(r"\(\d+\)\s*\(\d+\)"), # "1 (1) (1)" + re.compile(r"^image[ _-]?\d*$", re.I), # image, image1, image_2 + re.compile(r"^screenshot", re.I), # Screenshot 2025-04-24 at ... + re.compile(r"^screen[ _-]?shot", re.I), + re.compile(r"^unnamed", re.I), # unnamed, unnamed (1) + re.compile(r"^download", re.I), # download, download (3) + re.compile(r"^paste", re.I), # pasted images + re.compile(r"^untitled", re.I), + re.compile(r"^[0-9a-f]{8,}$", re.I), # hex / uuid-ish blobs + re.compile(r"^img[ _-]?\d+$", re.I), # IMG_1234 + re.compile(r"^capture", re.I), +] + +# Words that carry no descriptive value when building a slug from a caption. +STOPWORDS = { + "the", "a", "an", "of", "in", "on", "to", "for", "and", "or", "with", "at", + "by", "is", "are", "this", "that", "your", "you", "from", "as", "it", "its", + "page", "screen", "screenshot", "image", "view", "showing", "shows", "shown", + "displaying", "displays", "example", "snyk", "here", "above", "below", "see", + "click", "clicking", "select", "selecting", "where", "which", "when", "into", + "can", "will", "be", "new", "all", +} + +MAX_SLUG_WORDS = 6 + +# Vision: only raster formats Claude can actually look at (SVG is XML, PDF isn't +# an image block). Maps file extension -> API media_type. +VISION_MEDIA_TYPES = { + ".png": "image/png", + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".gif": "image/gif", + ".webp": "image/webp", +} +VISION_MODEL = "claude-opus-4-8" +VISION_MAX_BYTES = 5 * 1024 * 1024 # API per-image limit + +# --------------------------------------------------------------------------- +# Reference + context extraction +# --------------------------------------------------------------------------- + +#
...
blocks (alt + optional figcaption) +RE_FIGURE = re.compile(r"", re.DOTALL | re.IGNORECASE) +RE_IMG_SRC = re.compile(r']*?\bsrc="([^"]+)"', re.IGNORECASE) +RE_IMG_ALT = re.compile(r']*?\balt="([^"]*)"', re.IGNORECASE) +RE_FIGCAPTION = re.compile(r"]*>(.*?)", re.DOTALL | re.IGNORECASE) +# markdown image: ![alt](path) OR GitBook's angle-bracket form ![alt]() +RE_MD_IMG = re.compile(r"!\[([^\]]*)\]\((<[^>]*>|[^)]*)\)") +# any .gitbook/assets/ reference inside a non-markdown text file (yaml, html, …) +RE_ASSET_TOKEN = re.compile( + r"\.gitbook/assets/(<[^>]+>|[^\s\"'`)>\],]+)", re.IGNORECASE) +RE_HEADING = re.compile(r"^#{1,6}\s+(.*)$", re.MULTILINE) +RE_TAGS = re.compile(r"<[^>]+>") + +ASSET_RE = re.compile(r"\.gitbook/assets/", re.IGNORECASE) + + +def section_for(md_path: Path) -> Path: + """Return the top-level section dir (the one whose .gitbook/assets this md uses).""" + rel = md_path.relative_to(REPO_ROOT) + return REPO_ROOT / rel.parts[0] + + +def asset_basename(src: str) -> str | None: + """Given an or markdown target, return the asset basename, or None.""" + if not ASSET_RE.search(src): + return None + # strip query string / anchor, then take the part after assets/ + src = src.split("?", 1)[0].split("#", 1)[0] + after = re.split(r"\.gitbook/assets/", src, flags=re.IGNORECASE)[-1] + # GitBook wraps space/paren paths in angle brackets: ![alt](<...assets/x (1).png>) + name = after.strip().strip("<>\"' ") + return name or None + + +def clean_text(html: str) -> str: + return re.sub(r"\s+", " ", RE_TAGS.sub(" ", html)).strip() + + +def nearest_heading(text: str, pos: int) -> str: + last = "" + for m in RE_HEADING.finditer(text): + if m.start() > pos: + break + last = m.group(1).strip() + return clean_text(last) + + +def extract_references(md_path: Path): + """Yield (section, basename, context, md_path) for every image ref in a file.""" + section = section_for(md_path) + try: + text = md_path.read_text(encoding="utf-8") + except (UnicodeDecodeError, OSError): + return + + seen_spans = [] + + # 1.
blocks: richest context (figcaption beats alt) + for fig in RE_FIGURE.finditer(text): + block = fig.group(0) + src_m = RE_IMG_SRC.search(block) + if not src_m: + continue + name = asset_basename(src_m.group(1)) + if not name: + continue + cap = RE_FIGCAPTION.search(block) + alt = RE_IMG_ALT.search(block) + context = "" + if cap and clean_text(cap.group(1)): + context = clean_text(cap.group(1)) + elif alt and alt.group(1).strip(): + context = clean_text(alt.group(1)) + else: + context = nearest_heading(text, fig.start()) + seen_spans.append(fig.span()) + yield section, name, context, md_path + + # 2. bare not inside a figure we already handled + for img in RE_IMG_SRC.finditer(text): + if any(s <= img.start() < e for s, e in seen_spans): + continue + name = asset_basename(img.group(1)) + if not name: + continue + alt_m = RE_IMG_ALT.search(text[img.start():img.start() + 400]) + context = clean_text(alt_m.group(1)) if alt_m and alt_m.group(1).strip() else "" + if not context: + context = nearest_heading(text, img.start()) + yield section, name, context, md_path + + # 3. markdown images + for md in RE_MD_IMG.finditer(text): + alt, target = md.group(1), md.group(2) + name = asset_basename(target) + if not name: + continue + context = clean_text(alt) if alt.strip() else nearest_heading(text, md.start()) + yield section, name, context, md_path + + +# --------------------------------------------------------------------------- +# Naming +# --------------------------------------------------------------------------- + +def is_generic(stem: str) -> bool: + return any(p.search(stem) for p in GENERIC_PATTERNS) + + +def slugify(text: str, max_words: int = MAX_SLUG_WORDS) -> str: + words = re.findall(r"[A-Za-z0-9]+", text.lower()) + kept = [w for w in words if w not in STOPWORDS] + if not kept: # everything was a stopword; fall back + kept = words + return "-".join(kept[:max_words]) + + +def suggest_name(contexts: list[str], ext: str, fallback: str) -> str: + """Build a descriptive kebab-case name from the most common context string.""" + texts = [c for c in contexts if c] + if texts: + # most common non-empty caption wins; it best describes the depiction + dominant = Counter(texts).most_common(1)[0][0] + slug = slugify(dominant) + else: + slug = slugify(fallback) + if not slug: + slug = "image" + return f"{slug}{ext.lower()}" + + +def caption_relevance(name: str, contexts: list[str]): + """How well a proposed name reflects the image's captions. + + Returns (score, shared_words). score is the fraction of the name's + meaningful words that appear in the captions (1.0 = fully caption-derived). + Returns (None, []) when there is no caption to judge against (orphans / + vision-named images) — relevancy only applies to referenced images. + """ + caption_words = set() + for c in contexts: + if c: + caption_words |= {w for w in re.findall(r"[A-Za-z0-9]+", c.lower()) + if w not in STOPWORDS} + if not caption_words: + return None, [] + stem = os.path.splitext(name)[0] + # ignore pure-digit tokens (e.g. the "-2" dedupe suffix) — not meaningful + name_words = [w for w in stem.split("-") if w and not w.isdigit()] + if not name_words: + return 0.0, [] + shared = [w for w in name_words if w in caption_words] + return len(shared) / len(name_words), shared + + +def dedupe(suggestions: dict) -> None: + """Ensure suggested names are unique within a section (mutates in place).""" + by_section_name: dict = defaultdict(list) + for rec in suggestions.values(): + by_section_name[(rec["section"], rec["suggested"])].append(rec) + for (_section, _name), recs in by_section_name.items(): + if len(recs) <= 1: + continue + for i, rec in enumerate(recs[1:], start=2): + stem, ext = os.path.splitext(rec["suggested"]) + rec["suggested"] = f"{stem}-{i}{ext}" + + +# --------------------------------------------------------------------------- +# Scan +# --------------------------------------------------------------------------- + +# Non-markdown text files that could also reference assets (configs, nav, html). +AUX_EXTS = {".yaml", ".yml", ".json", ".html", ".htm", ".txt", ".toml"} +# Directories we never treat as a reference source (self, generated output, vendored). +SKIP_DIRS = {".git", "node_modules", "scripts"} + + +def _in_section(path: Path, sections) -> bool: + if sections is None: + return True + rel = path.relative_to(REPO_ROOT) + return rel.parts and rel.parts[0] in sections + + +def find_md_files(sections=None): + for p in REPO_ROOT.rglob("*.md"): + if SKIP_DIRS & set(p.parts): + continue + if _in_section(p, sections): + yield p + + +def find_aux_files(sections=None): + """Non-.md text files (yaml/json/html/…) that might reference assets.""" + for p in REPO_ROOT.rglob("*"): + if not p.is_file() or p.suffix.lower() not in AUX_EXTS: + continue + if SKIP_DIRS & set(p.parts) or ".gitbook" in p.parts: + continue # skip assets dir itself and vendored/self dirs + if _in_section(p, sections): + yield p + + +def find_image_files(sections=None): + for assets in REPO_ROOT.rglob(".gitbook/assets"): + if not assets.is_dir() or (SKIP_DIRS & set(assets.parts)): + continue + if not _in_section(assets, sections): + continue + for f in assets.iterdir(): + if f.is_file() and f.suffix.lower() in IMAGE_EXTS: + yield f + + +def all_sections() -> set: + """Top-level section dirs that own a .gitbook/assets folder.""" + out = set() + for assets in REPO_ROOT.rglob(".gitbook/assets"): + if assets.is_dir() and not (SKIP_DIRS & set(assets.parts)): + out.add(assets.relative_to(REPO_ROOT).parts[0]) + return out + + +def collect_references(sections=None) -> dict: + """(section, basename) -> list of (context, source_path), across md + aux files.""" + refs: dict = defaultdict(list) + for md in find_md_files(sections): + for section, name, context, src in extract_references(md): + refs[(section, name)].append((context, src)) + for aux in find_aux_files(sections): + try: + text = aux.read_text(encoding="utf-8") + except (UnicodeDecodeError, OSError): + continue + section = section_for(aux) + for m in RE_ASSET_TOKEN.finditer(text): + name = m.group(1).strip("<>\"' ") + if name: + refs[(section, name)].append(("", aux)) + return refs + + +def scan(include_all: bool, sections=None): + refs = collect_references(sections) + + records = {} + for img in find_image_files(sections): + section = section_for(img) + stem = img.stem + if not include_all and not is_generic(stem): + continue + key = (section, img.name) + usages = refs.get(key, []) + contexts = [c for c, _ in usages] + locations = sorted({str(p.relative_to(REPO_ROOT)) for _, p in usages}) + page_fallback = " ".join( + re.findall(r"[A-Za-z0-9]+", img.relative_to(section).as_posix()) + ) + records[str(img.relative_to(REPO_ROOT))] = { + "section": str(section.relative_to(REPO_ROOT)), + "path": img, + "old_name": img.name, + "suggested": suggest_name(contexts, img.suffix, page_fallback), + "uses": len(usages), + "locations": locations, + "contexts": contexts, + "vision_description": None, + } + return records + + +# --------------------------------------------------------------------------- +# Vision naming (optional) — name images that have no in-repo caption context +# --------------------------------------------------------------------------- + +VISION_PROMPT = ( + "You are naming a screenshot/image from the Snyk product documentation so it " + "can be reused across pages. Look at the image and return a concise, " + "descriptive file name that captures what it depicts (the screen, feature, or " + "action shown) rather than incidental details. Use lowercase kebab-case, 2-6 " + "words, no file extension, no the/a/of filler, no the word 'snyk' or " + "'screenshot'. Also return a one-sentence description of what the image shows." +) + +VISION_SCHEMA = { + "type": "object", + "properties": { + "name": {"type": "string", "description": "kebab-case file name, no extension"}, + "description": {"type": "string", "description": "one sentence on what is shown"}, + }, + "required": ["name", "description"], + "additionalProperties": False, +} + + +def _needs_vision(rec: dict) -> bool: + """True when an image has no usable in-repo caption/alt context to name from.""" + return not any(c for c in rec["contexts"]) + + +def _name_one_with_vision(client, rec: dict, model: str) -> None: + """Call Claude vision for a single record; mutate suggested/vision_description.""" + img: Path = rec["path"] + ext = img.suffix.lower() + media_type = VISION_MEDIA_TYPES.get(ext) + if media_type is None: + return # SVG/PDF/etc. — nothing to look at + if img.stat().st_size > VISION_MAX_BYTES: + return + data = base64.standard_b64encode(img.read_bytes()).decode("utf-8") + + response = client.messages.create( + model=model, + max_tokens=300, + output_config={"format": {"type": "json_schema", "schema": VISION_SCHEMA}}, + messages=[{ + "role": "user", + "content": [ + {"type": "image", "source": {"type": "base64", "media_type": media_type, "data": data}}, + {"type": "text", "text": VISION_PROMPT}, + ], + }], + ) + if response.stop_reason == "refusal": + return + text = next((b.text for b in response.content if b.type == "text"), None) + if not text: + return + parsed = json.loads(text) + slug = slugify(parsed.get("name", "")) + if slug: + rec["suggested"] = f"{slug}{ext}" + rec["vision_description"] = parsed.get("description") or None + + +def apply_vision(records: dict, model: str, workers: int) -> None: + """Fill suggestions for context-less images by looking at them with Claude.""" + try: + import anthropic # lazy — only needed with --vision + except ImportError: + sys.exit("--vision needs the Anthropic SDK: pip install anthropic") + if not (os.environ.get("ANTHROPIC_API_KEY") or os.environ.get("ANTHROPIC_AUTH_TOKEN")): + sys.exit("--vision needs ANTHROPIC_API_KEY (or ANTHROPIC_AUTH_TOKEN) in the environment.") + + client = anthropic.Anthropic() + targets = [r for r in records.values() + if _needs_vision(r) and r["path"].suffix.lower() in VISION_MEDIA_TYPES] + skipped = sum(1 for r in records.values() + if _needs_vision(r) and r["path"].suffix.lower() not in VISION_MEDIA_TYPES) + if skipped: + print(f" vision: skipping {skipped} non-raster image(s) (SVG/PDF — can't be analyzed)") + print(f" vision: analyzing {len(targets)} context-less image(s) with {model}...") + + done = errors = 0 + with ThreadPoolExecutor(max_workers=workers) as pool: + futures = {pool.submit(_name_one_with_vision, client, r, model): r for r in targets} + for fut in as_completed(futures): + done += 1 + try: + fut.result() + except Exception as e: # keep going; leave the heuristic name in place + errors += 1 + print(f" ! {futures[fut]['old_name']}: {e}") + if done % 25 == 0: + print(f" {done}/{len(targets)} analyzed") + print(f" vision: done ({done} analyzed, {errors} failed and left with fallback names)") + + +# --------------------------------------------------------------------------- +# Output +# --------------------------------------------------------------------------- + +# Below this caption-overlap fraction, a referenced image's proposed name is +# flagged for manual review (it doesn't clearly reflect its captions). +RELEVANCE_THRESHOLD = 0.5 + + +def write_report(records: dict, report_path: Path = REPORT_MD): + rows = sorted(records.values(), key=lambda r: (-r["uses"], r["section"], r["old_name"])) + # caption-relevance for every referenced image (None for orphans/vision) + for r in rows: + score, shared = caption_relevance(r["suggested"], r["contexts"]) + r["relevance"], r["relevance_shared"] = score, shared + low = [r for r in rows + if r["relevance"] is not None and r["relevance"] < RELEVANCE_THRESHOLD] + + with report_path.open("w", encoding="utf-8") as f: + f.write("# Image rename report\n\n") + f.write(f"Flagged **{len(rows)}** image(s) for renaming.\n\n") + unused = [r for r in rows if r["uses"] == 0] + f.write(f"- Referenced at least once: **{len(rows) - len(unused)}**\n") + f.write(f"- Never referenced (orphans): **{len(unused)}**\n") + f.write(f"- Names with weak caption relevance (review these): **{len(low)}**\n\n") + f.write("Review the suggested names, edit `image-rename-mapping.csv`, " + "then run `python3 scripts/rename_images.py --apply`.\n\n") + + if low: + f.write("## ⚠ Proposed names to double-check (low caption overlap)\n\n") + f.write("These names don't clearly draw from the image's captions — " + "edit `proposed_name` in the CSV if they're off.\n\n") + for r in low: + caps = "; ".join(dict.fromkeys(c for c in r["contexts"] if c)) or "—" + f.write(f"- `{r['old_name']}` → `{r['suggested']}` " + f"(relevance {r['relevance']:.0%}) — captions: \"{caps}\"\n") + f.write("\n") + + for r in rows: + f.write(f"## `{r['old_name']}` → `{r['suggested']}`\n\n") + f.write(f"- **Section:** {r['section']}\n") + f.write(f"- **Used:** {r['uses']} time(s)\n") + if r["relevance"] is not None: + flag = " ⚠ review" if r["relevance"] < RELEVANCE_THRESHOLD else "" + f.write(f"- **Caption relevance:** {r['relevance']:.0%}{flag}\n") + if r["locations"]: + f.write("- **Appears in:**\n") + for loc in r["locations"]: + f.write(f" - {loc}\n") + else: + f.write("- **Appears in:** _(not referenced — safe to rename or remove)_\n") + ctx = [c for c in r["contexts"] if c] + if ctx: + f.write("- **Depicts (from captions / alt text):**\n") + for c in dict.fromkeys(ctx): # unique, order-preserving + f.write(f" - \"{c}\"\n") + elif r.get("vision_description"): + f.write(f"- **Depicts (from image analysis):** \"{r['vision_description']}\"\n") + f.write("\n") + print(f" report -> {report_path}") + return low + + +def write_csv(records: dict, csv_path: Path = MAPPING_CSV): + # n_pages = distinct docs that reference the image (not raw occurrence count) + rows = sorted(records.values(), + key=lambda r: (-len(r["locations"]), r["section"], r["old_name"])) + with csv_path.open("w", encoding="utf-8", newline="") as f: + w = csv.writer(f) + w.writerow([ + "repo_location", "current_name", "proposed_name", + "used_on_n_pages", "source_captions", + ]) + for r in rows: + captions = list(dict.fromkeys(c for c in r["contexts"] if c)) # unique, ordered + if not captions and r.get("vision_description"): + captions = [f"(image analysis) {r['vision_description']}"] + w.writerow([ + str(r["path"].relative_to(REPO_ROOT)), + r["old_name"], + r["suggested"], + len(r["locations"]), + " | ".join(captions), + ]) + print(f" mapping -> {csv_path}") + + +# --------------------------------------------------------------------------- +# Apply +# --------------------------------------------------------------------------- + +def apply_mapping(csv_path: Path, delete_orphans: bool, dry_run: bool = False): + if not csv_path.exists(): + sys.exit(f"Mapping CSV not found: {csv_path}. Run without --apply first to generate it.") + + # Live re-scan: never trust a possibly-stale CSV for a destructive op. + live_refs = collect_references() + + renames = [] # (old_path, new_path, section, old_name, new_name) + deletes = [] # old_path + with csv_path.open(encoding="utf-8", newline="") as f: + for row in csv.DictReader(f): + old_path = REPO_ROOT / row["repo_location"] + new_name = (row.get("proposed_name") or "").strip() + old_name = row["current_name"].strip() + try: + n_pages = int(row.get("used_on_n_pages") or 0) + except ValueError: + n_pages = 0 + if not old_path.exists(): + print(f" skip (missing file): {row['repo_location']}") + continue + + section = str(section_for(old_path).relative_to(REPO_ROOT)) + live_count = len(live_refs.get((section, old_name), [])) + + if n_pages == 0: + if not delete_orphans: + continue # orphan, but deletion not requested → leave it + if live_count > 0: # SAFETY: CSV said orphan, live scan disagrees + print(f" KEEP (now referenced, not deleting): {row['repo_location']}") + continue + deletes.append(old_path) + continue + + # Referenced image → rename. Guard against a stale CSV claiming refs + # that no longer exist (we'd rewrite nothing and orphan the new name). + if not new_name or new_name == old_name: + continue + new_path = old_path.with_name(new_name) + if new_path.exists(): + print(f" skip (rename target exists): {new_path.relative_to(REPO_ROOT)}") + continue + renames.append((old_path, new_path, section, old_name, new_name)) + + if not renames and not deletes: + print("Nothing to apply.") + return + + if dry_run: + print(f"\nDRY RUN — no files changed. Would rename {len(renames)}, delete {len(deletes)}.") + for old_path, new_path, _s, _o, _n in renames[:40]: + print(f" RENAME {old_path.relative_to(REPO_ROOT)} -> {new_path.name}") + if len(renames) > 40: + print(f" ... and {len(renames) - 40} more renames") + for old_path in deletes[:40]: + print(f" DELETE {old_path.relative_to(REPO_ROOT)}") + if len(deletes) > 40: + print(f" ... and {len(deletes) - 40} more deletions") + return + + # 1. Rewrite references first (so we never leave dangling links), then move. + # IMPORTANT: assets are section-scoped. The same generic basename (e.g. 1.png) + # can exist in several sections with DIFFERENT new names, so rewriting must be + # scoped to each file's own section — a global by-basename rewrite corrupts + # references in sibling sections (and parallel trees like docs/ vs the rest). + renames_by_section: dict = defaultdict(dict) + for _op, _np, section, old, new in renames: + renames_by_section[section][old] = new + + changed_files = 0 + if renames_by_section: + for src in (*find_md_files(), *find_aux_files()): + section = str(section_for(src).relative_to(REPO_ROOT)) + section_map = renames_by_section.get(section) + if not section_map: + continue + try: + text = src.read_text(encoding="utf-8") + except (UnicodeDecodeError, OSError): + continue + new_text = text + for old, new in section_map.items(): + # match only inside an assets/ path, and require a real boundary + # after the name so "1.png" can't match inside "image (1).png" etc. + new_text = re.sub( + r"(\.gitbook/assets/)" + re.escape(old) + r"(?=[)>\"'\s\],]|$)", + lambda m, new=new: m.group(1) + new, + new_text, + ) + if new_text != text: + src.write_text(new_text, encoding="utf-8") + changed_files += 1 + + for old_path, new_path, _s, _o, _n in renames: + os.rename(old_path, new_path) + + # 2. Delete confirmed-unreferenced orphans. + for old_path in deletes: + old_path.unlink() + + print(f"Renamed {len(renames)} file(s); updated references in {changed_files} doc(s); " + f"deleted {len(deletes)} orphan(s).") + + +# --------------------------------------------------------------------------- +# Main +# --------------------------------------------------------------------------- + +def main(): + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--apply", action="store_true", + help="rename files and rewrite references using the mapping CSV") + ap.add_argument("--dry-run", action="store_true", + help="with --apply, print the rename/delete plan without changing anything") + ap.add_argument("--all", action="store_true", + help="include every image, not only generically-named ones") + ap.add_argument("--delete-orphans", action="store_true", + help="on --apply, delete generically-named images that nothing references " + "(re-verified live at apply time)") + ap.add_argument("--sections", default=None, + help="comma-separated top-level dirs to limit the scan to (e.g. for a dry run)") + ap.add_argument("--exclude-sections", default=None, + help="comma-separated top-level dirs to skip (applied after --sections)") + ap.add_argument("--vision", action="store_true", + help="use Claude vision to name images that have no caption/alt context " + "(needs `pip install anthropic` + ANTHROPIC_API_KEY)") + ap.add_argument("--vision-model", default=VISION_MODEL, + help=f"model for --vision (default: {VISION_MODEL})") + ap.add_argument("--vision-workers", type=int, default=8, + help="concurrent vision requests (default: 8)") + ap.add_argument("--csv", default=str(MAPPING_CSV), + help="mapping CSV path (default: image-rename-mapping.csv)") + args = ap.parse_args() + + if args.apply: + apply_mapping(Path(args.csv), delete_orphans=args.delete_orphans, dry_run=args.dry_run) + return + + sections = None + if args.sections: + sections = {s.strip() for s in args.sections.split(",") if s.strip()} + if args.exclude_sections: + excluded = {s.strip() for s in args.exclude_sections.split(",") if s.strip()} + base = sections if sections is not None else all_sections() + sections = base - excluded + print(f"Excluding sections: {', '.join(sorted(excluded))}") + if sections is not None: + print(f"Scanning sections: {', '.join(sorted(sections)) or '(none)'}") + + print("Scanning repository for images...") + records = scan(include_all=args.all, sections=sections) + n_orphans = sum(1 for r in records.values() if not r["locations"]) + print(f"Flagged {len(records)} image(s) ({n_orphans} orphan(s) with 0 references).") + if args.vision: + apply_vision(records, args.vision_model, args.vision_workers) + dedupe(records) + csv_path = Path(args.csv) + # keep the default report name, but for a custom --csv put the report beside it + report_path = REPORT_MD if csv_path == MAPPING_CSV else csv_path.with_name(csv_path.stem + "-report.md") + low = write_report(records, report_path) + write_csv(records, csv_path) + if low: + print(f"{len(low)} proposed name(s) have weak caption relevance — " + f"see the '⚠ Proposed names to double-check' section of the report.") + print("\nReview the report/CSV, then run:" + "\n python3 scripts/rename_images.py --apply --delete-orphans") + + +if __name__ == "__main__": + main() From 6e004ddf510215f1d0215af2099969c76defcff6 Mon Sep 17 00:00:00 2001 From: Andreea Luca Date: Mon, 15 Jun 2026 14:58:14 +0300 Subject: [PATCH 2/5] docs: add scripts/README.md explaining the image cleanup tool Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/README.md | 103 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 103 insertions(+) create mode 100644 scripts/README.md diff --git a/scripts/README.md b/scripts/README.md new file mode 100644 index 000000000000..066b36e9b0c1 --- /dev/null +++ b/scripts/README.md @@ -0,0 +1,103 @@ +# Image cleanup — `rename_images.py` + +A maintenance tool for the images under each section's `.gitbook/assets/` folder. It gives screenshots meaningful, reusable names and removes ones that nothing uses — safely, with a review step. + +## The problem it solves + +GitBook drops uploaded images into `.gitbook/assets/` with whatever name they arrived with. Over time that produces: + +- **Non-descriptive names** — `1.png`, `2 (2).png`, `Screenshot 2025-04-24 at 14.52.22.png`, `unnamed.png`, `image (105).png`. You can't tell what an image is without opening it, and you can't reuse it across pages with confidence. +- **Orphans** — images that no page references anymore (left behind when a page was edited or deleted). They bloat the repo and make the assets folder hard to navigate. +- **Risky manual renames** — renaming an asset by hand means hunting down every `.gitbook/assets/…` reference in the Markdown and updating each one, or you silently break an image. + +This script fixes all three: it proposes a descriptive name **derived from each image's own caption/alt text**, rewrites every reference for you, and deletes the orphans — and it never does any of that without a review step. + +## What it does + +1. **Scans** every section's `.gitbook/assets/` for generically-named images. +2. For each one, finds where it's used and reads the **caption / alt text** around each use. +3. **Proposes a descriptive kebab-case name** from the most common caption (e.g. a figure captioned *"Group Settings: SSO"* → `group-settings-sso.png`). +4. Writes a **review report** and a **mapping CSV** you can edit. +5. On `--apply`: renames the files, **rewrites every reference** to them, and (with `--delete-orphans`) deletes unreferenced images. + +### Safety features + +- **Dry run** (`--dry-run`) prints the full rename/delete plan without changing anything. +- **Live re-verification** — before deleting an "orphan", it re-scans the repo at apply time and refuses to delete anything that turns out to be referenced. +- **Section-scoped reference rewriting** — the same generic name (`1.png`) can exist in several sections with different meanings; rewrites are scoped per section so a reference is never pointed at the wrong section's file. +- **Robust reference parsing** — handles GitBook's angle-bracket link form `![alt](<…/assets/image (1).png>)` and references in non-Markdown files (`.yaml`, `.json`, `.html`). +- **Caption-relevance check** — flags any proposed name that doesn't actually reflect its captions, so you can correct it before applying. + +## Requirements + +- Python 3.8+ (standard library only). +- Optional: `pip install anthropic` + `ANTHROPIC_API_KEY` — only needed for `--vision` (see below). + +## Usage + +Run from the repo root. + +```bash +# 1. Report only (safe — writes the report + CSV, changes nothing) +python3 scripts/rename_images.py + +# 2. Review the outputs +# image-rename-report.md — per-image: proposed name, usage, captions, relevance +# image-rename-mapping.csv — edit proposed_name here if you disagree with any + +# 3. See exactly what applying would do (no changes) +python3 scripts/rename_images.py --apply --delete-orphans --dry-run + +# 4. Apply: rename + rewrite references + delete orphans +python3 scripts/rename_images.py --apply --delete-orphans +``` + +### Useful flags + +| Flag | What it does | +|---|---| +| *(none)* | Report mode: writes `image-rename-mapping.csv` + `image-rename-report.md`. Changes nothing. | +| `--apply` | Execute the mapping CSV: rename files and rewrite references. | +| `--delete-orphans` | With `--apply`, also delete images that nothing references (re-verified live). | +| `--dry-run` | With `--apply`, print the plan and change nothing. | +| `--sections a,b` | Limit to these top-level section dirs (e.g. for a smaller, reviewable run). | +| `--exclude-sections a,b` | Skip these sections (applied after `--sections`). | +| `--all` | Include every image, not just generically-named ones. | +| `--csv PATH` | Use a custom mapping CSV path (keeps scoped runs from clobbering the canonical one). | +| `--vision` | Name context-less images (orphans / no caption) by having Claude look at them. Needs `anthropic` + `ANTHROPIC_API_KEY`. | + +### Mapping CSV columns + +| Column | Meaning | +|---|---| +| `repo_location` | Path to the image file. | +| `current_name` | Its current filename. | +| `proposed_name` | Suggested descriptive name (**edit this** to override). | +| `used_on_n_pages` | Number of distinct pages that reference it (`0` = orphan). | +| `source_captions` | The captions/alt text the name was derived from. | + +To **keep** an image the report wants to rename or delete, edit its `proposed_name` (or delete its row) before `--apply`. + +## Automated cleanup (GitHub Action) + +[`.github/workflows/clean-up-images.yml`](../.github/workflows/clean-up-images.yml) runs this monthly and on manual dispatch, and **opens a PR for review** — it never pushes to `main`. + +- **Excludes `docs` by default** (that section is managed separately). Override with the `exclude_sections` input on a manual run. +- Manual runs also accept a `sections` input (limit scope) and a `delete_orphans` toggle. +- The review report is attached to each run as an artifact. + +To run the one-time cleanup: trigger the workflow via **Actions → Clean up doc images → Run workflow**. + +## How this was built (summary) + +The tool was developed iteratively, hardening it against problems found along the way: + +1. **Catalog first** — scan the assets, derive names from captions, and emit a human-readable report + CSV (no changes yet). +2. **Add apply** — rename files and rewrite all references; default to a safe report-only mode. +3. **Add orphan deletion** — with a live re-verification so a stale CSV can never cause a wrong delete. +4. **Fix reference parsing** — GitBook's angle-bracket links were being mis-parsed, which had flagged real images as false orphans; fixed, and broadened to non-Markdown files. +5. **Fix section scoping** — a full-repo dry run revealed that rewriting references globally by filename broke links across the repo's parallel section trees; rewriting is now scoped per section. +6. **Add relevancy + dry run + section filters** — verify names match captions, preview the plan, and scope runs to a few folders. +7. **Wrap in a review-gated workflow** — monthly + manual, opening a PR rather than committing to `main`, with `docs` excluded by default. + +Throughout, every destructive run was preceded by a dry run and a dangling-reference check (rename → confirm zero broken references → commit). From c1b6fcb723cd79811a5c0b4530e396324371e77e Mon Sep 17 00:00:00 2001 From: Andreea Luca Date: Mon, 15 Jun 2026 14:59:11 +0300 Subject: [PATCH 3/5] docs: link maintenance scripts from the root README Co-Authored-By: Claude Opus 4.8 (1M context) --- README.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/README.md b/README.md index 594bebc4ded0..8da696327144 100644 --- a/README.md +++ b/README.md @@ -23,6 +23,10 @@ Before working on larger contributions, [contact support](https://support.snyk.i If you want to add an integration to Snyk, [apply to become a Snyk partner](https://partners.snyk.io/English/register_email.aspx). +### Maintenance scripts + +The [`scripts/`](scripts/) directory holds repository maintenance tooling. See [scripts/README.md](scripts/README.md) for the image cleanup tool, which gives `.gitbook/assets/` images descriptive, caption-derived names and removes unreferenced ones (run monthly, or on demand, as a reviewable PR). + ## Security For any security issues or concerns, go to [SECURITY.md](SECURITY.md). From 41f4f547c378051699cca4826a500082ce1294d8 Mon Sep 17 00:00:00 2001 From: Andreea Luca Date: Tue, 16 Jun 2026 14:09:41 +0300 Subject: [PATCH 4/5] =?UTF-8?q?refactor:=20make=20mapping=20CSV=20concise?= =?UTF-8?q?=20=E2=80=94=20one=20source=20caption=20per=20row?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/README.md | 2 +- scripts/rename_images.py | 14 ++++++++++---- 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/scripts/README.md b/scripts/README.md index 066b36e9b0c1..e6f9a852111a 100644 --- a/scripts/README.md +++ b/scripts/README.md @@ -74,7 +74,7 @@ python3 scripts/rename_images.py --apply --delete-orphans | `current_name` | Its current filename. | | `proposed_name` | Suggested descriptive name (**edit this** to override). | | `used_on_n_pages` | Number of distinct pages that reference it (`0` = orphan). | -| `source_captions` | The captions/alt text the name was derived from. | +| `source_captions` | The caption/alt text the name was derived from (the dominant one, for readability). | To **keep** an image the report wants to rename or delete, edit its `proposed_name` (or delete its row) before `--apply`. diff --git a/scripts/rename_images.py b/scripts/rename_images.py index 72d9e44537d6..dfcbe8fa7a69 100644 --- a/scripts/rename_images.py +++ b/scripts/rename_images.py @@ -536,15 +536,21 @@ def write_csv(records: dict, csv_path: Path = MAPPING_CSV): "used_on_n_pages", "source_captions", ]) for r in rows: - captions = list(dict.fromkeys(c for c in r["contexts"] if c)) # unique, ordered - if not captions and r.get("vision_description"): - captions = [f"(image analysis) {r['vision_description']}"] + # Keep it concise: show the single caption the name was derived from + # (the most common one), not every caption the image ever had. + present = [c for c in r["contexts"] if c] + if present: + source_caption = Counter(present).most_common(1)[0][0] + elif r.get("vision_description"): + source_caption = f"(image analysis) {r['vision_description']}" + else: + source_caption = "" w.writerow([ str(r["path"].relative_to(REPO_ROOT)), r["old_name"], r["suggested"], len(r["locations"]), - " | ".join(captions), + source_caption, ]) print(f" mapping -> {csv_path}") From a722f69d0e3d9c9b22f3f6f81ed21aadc0338ac3 Mon Sep 17 00:00:00 2001 From: Andreea Luca Date: Tue, 16 Jun 2026 14:17:02 +0300 Subject: [PATCH 5/5] feat: flag likely-misspelled words in proposed names (report review aid) Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/README.md | 1 + scripts/rename_images.py | 102 +++++++++++++++++++++++++++++++++++++-- 2 files changed, 100 insertions(+), 3 deletions(-) diff --git a/scripts/README.md b/scripts/README.md index e6f9a852111a..429493df97e0 100644 --- a/scripts/README.md +++ b/scripts/README.md @@ -27,6 +27,7 @@ This script fixes all three: it proposes a descriptive name **derived from each - **Section-scoped reference rewriting** — the same generic name (`1.png`) can exist in several sections with different meanings; rewrites are scoped per section so a reference is never pointed at the wrong section's file. - **Robust reference parsing** — handles GitBook's angle-bracket link form `![alt](<…/assets/image (1).png>)` and references in non-Markdown files (`.yaml`, `.json`, `.html`). - **Caption-relevance check** — flags any proposed name that doesn't actually reflect its captions, so you can correct it before applying. +- **Spell-sanity check** — flags proposed names containing a likely-misspelled word (typos are often carried over from the caption, e.g. `attribue`). Heuristic: uses the system word list plus a product/tech allow-list, and skips words that appear capitalized in the caption (product names). Reported for review; never auto-corrected. ## Requirements diff --git a/scripts/rename_images.py b/scripts/rename_images.py index dfcbe8fa7a69..d4a16342517b 100644 --- a/scripts/rename_images.py +++ b/scripts/rename_images.py @@ -248,6 +248,85 @@ def caption_relevance(name: str, contexts: list[str]): return len(shared) / len(name_words), shared +# --- spell-sanity (best-effort) ------------------------------------------------- +# Names mirror their captions, so a typo in a caption ("attribue") becomes a typo +# in the file name. This flags likely-misspelled words for manual review. It's a +# heuristic: it uses the system word list (if present) plus a small allow-list of +# product/tech terms, and tolerates common plural/verb inflections. +_DICT_PATHS = ("/usr/share/dict/words", "/usr/share/dict/web2") +_dictionary_cache = None + +SPELL_ALLOW = { + "snyk", "sast", "sca", "iac", "sbom", "cli", "api", "apis", "sso", "saml", + "oauth", "idp", "mfa", "scm", "repo", "repos", "github", "gitlab", "bitbucket", + "jira", "jenkins", "npm", "pypi", "maven", "gradle", "nuget", "dotnet", "golang", + "javascript", "typescript", "dropdown", "config", "configs", "url", "urls", + "devops", "webhook", "webhooks", "jetbrains", "vscode", "cwe", "cwes", "cve", + "cves", "readme", "json", "yaml", "sdk", "sdks", "gcr", "ecr", "acr", "aws", + "gcp", "azure", "ide", "ides", "cicd", "toml", "nexus", "artifactory", + "kubernetes", "docker", "onboarding", "analytics", "dashboard", "auth", "login", + "logout", "backend", "frontend", "runtime", "namespace", "unignore", + "reprioritize", "prioritization", "integrations", "checkbox", "tooltip", + # common modern words missing from the archaic system word list + "email", "emails", "download", "downloads", "downloading", "upload", "uploads", + "uploading", "workspace", "workspaces", "plugin", "plugins", "app", "apps", + "mapping", "mappings", "dataset", "datasets", "signup", "walkthrough", + "changelog", "dashboards", "filename", "filenames", "username", "usernames", + "timestamp", "metadata", "subgroup", "subgroups", "tooltips", "wildcard", + "workflow", "workflows", "ranking", "rankings", "gitbook", "filtering", +} + + +def _load_dictionary() -> set: + global _dictionary_cache + if _dictionary_cache is None: + _dictionary_cache = set() + for p in _DICT_PATHS: + fp = Path(p) + if fp.exists(): + try: + _dictionary_cache = { + w.strip().lower() + for w in fp.read_text(encoding="utf-8", errors="ignore").splitlines() + } + break + except OSError: + pass + return _dictionary_cache + + +def _known(word: str, words: set) -> bool: + if word in words or word in SPELL_ALLOW: + return True + for suf, repl in (("s", ""), ("es", ""), ("ies", "y"), ("ing", ""), + ("ing", "e"), ("ed", ""), ("ed", "e"), ("d", "")): + if word.endswith(suf): + stem = word[:-len(suf)] + repl + if stem in words or stem in SPELL_ALLOW: + return True + return False + + +def suspect_words(name: str, contexts: list[str] = ()) -> list: + """Words in a proposed name that look misspelled (empty if no dictionary). + + Words that appear Capitalized in the captions are treated as proper nouns / + product names (Okta, Google, CircleCI) and skipped — that's where most false + positives come from, since the system word list is an archaic dictionary. + """ + words = _load_dictionary() + if not words: + return [] + proper = set() + for c in contexts: + if c: + proper |= {m.lower() for m in re.findall(r"[A-Z][A-Za-z]+", c)} + stem = os.path.splitext(name)[0] + return [t for t in stem.split("-") + if len(t) >= 4 and t.isalpha() + and t not in proper and not _known(t, words)] + + def dedupe(suggestions: dict) -> None: """Ensure suggested names are unique within a section (mutates in place).""" by_section_name: dict = defaultdict(list) @@ -477,8 +556,11 @@ def write_report(records: dict, report_path: Path = REPORT_MD): for r in rows: score, shared = caption_relevance(r["suggested"], r["contexts"]) r["relevance"], r["relevance_shared"] = score, shared + # only check names that will actually be applied (orphans get deleted) + r["suspect_words"] = suspect_words(r["suggested"], r["contexts"]) if r["locations"] else [] low = [r for r in rows if r["relevance"] is not None and r["relevance"] < RELEVANCE_THRESHOLD] + misspelled = [r for r in rows if r["suspect_words"]] with report_path.open("w", encoding="utf-8") as f: f.write("# Image rename report\n\n") @@ -486,10 +568,19 @@ def write_report(records: dict, report_path: Path = REPORT_MD): unused = [r for r in rows if r["uses"] == 0] f.write(f"- Referenced at least once: **{len(rows) - len(unused)}**\n") f.write(f"- Never referenced (orphans): **{len(unused)}**\n") - f.write(f"- Names with weak caption relevance (review these): **{len(low)}**\n\n") + f.write(f"- Names with weak caption relevance (review these): **{len(low)}**\n") + f.write(f"- Names with a possibly-misspelled word (review these): **{len(misspelled)}**\n\n") f.write("Review the suggested names, edit `image-rename-mapping.csv`, " "then run `python3 scripts/rename_images.py --apply`.\n\n") + if misspelled: + f.write("## ⚠ Proposed names with a possible misspelling\n\n") + f.write("Heuristic spell-check (often a typo carried over from the caption) — " + "fix `proposed_name` in the CSV if needed.\n\n") + for r in misspelled: + f.write(f"- `{r['suggested']}` — suspect: {', '.join(r['suspect_words'])}\n") + f.write("\n") + if low: f.write("## ⚠ Proposed names to double-check (low caption overlap)\n\n") f.write("These names don't clearly draw from the image's captions — " @@ -507,6 +598,8 @@ def write_report(records: dict, report_path: Path = REPORT_MD): if r["relevance"] is not None: flag = " ⚠ review" if r["relevance"] < RELEVANCE_THRESHOLD else "" f.write(f"- **Caption relevance:** {r['relevance']:.0%}{flag}\n") + if r["suspect_words"]: + f.write(f"- **Possible misspelling:** {', '.join(r['suspect_words'])} ⚠ review\n") if r["locations"]: f.write("- **Appears in:**\n") for loc in r["locations"]: @@ -522,7 +615,7 @@ def write_report(records: dict, report_path: Path = REPORT_MD): f.write(f"- **Depicts (from image analysis):** \"{r['vision_description']}\"\n") f.write("\n") print(f" report -> {report_path}") - return low + return low, misspelled def write_csv(records: dict, csv_path: Path = MAPPING_CSV): @@ -718,11 +811,14 @@ def main(): csv_path = Path(args.csv) # keep the default report name, but for a custom --csv put the report beside it report_path = REPORT_MD if csv_path == MAPPING_CSV else csv_path.with_name(csv_path.stem + "-report.md") - low = write_report(records, report_path) + low, misspelled = write_report(records, report_path) write_csv(records, csv_path) if low: print(f"{len(low)} proposed name(s) have weak caption relevance — " f"see the '⚠ Proposed names to double-check' section of the report.") + if misspelled: + print(f"{len(misspelled)} proposed name(s) have a possible misspelling — " + f"see the '⚠ Proposed names with a possible misspelling' section of the report.") print("\nReview the report/CSV, then run:" "\n python3 scripts/rename_images.py --apply --delete-orphans")