diff --git a/scripts/save_visual_assets.py b/scripts/save_visual_assets.py index 05ceec0..f5e20f8 100755 --- a/scripts/save_visual_assets.py +++ b/scripts/save_visual_assets.py @@ -2,6 +2,8 @@ import argparse import json import os +import re +import shutil import urllib.error import urllib.request from pathlib import Path @@ -27,6 +29,21 @@ DEFAULT_USER_AGENT = ( "Chrome/123.0.0.0 Safari/537.36" ) +CONTROLLER_PARENT_IDS = { + "413763", + "413813", + "433306", + "433326", + "433356", + "433376", + "433416", + "433434", + "433470", + "433484", +} +MACHINE_PARENT_IDS = {"413772", "413808"} +FRAME_PARENT_IDS = {"413776", "413804"} + def is_visual_url(url: str) -> bool: path = urlparse(url).path.lower() @@ -43,6 +60,22 @@ def walk_json(value): yield from walk_json(item) +def extract_media_metadata(obj: dict, source_path: str) -> dict: + details = obj.get("media_details") if isinstance(obj.get("media_details"), dict) else {} + title = obj.get("title") if isinstance(obj.get("title"), dict) else {} + return { + "media_id": str(obj.get("id") or ""), + "parent_id": str(obj.get("post") or ""), + "title": title.get("rendered") or obj.get("slug") or "", + "slug": obj.get("slug") or "", + "mime": obj.get("mime_type") or "", + "width": details.get("width") or "", + "height": details.get("height") or "", + "filesize": details.get("filesize") or "", + "source_json": source_path, + } + + def collect_from_json(path: Path): if path.name in {"media-spec-search.json", "media-manual-search.json", "media-video.json"}: return [] @@ -53,14 +86,47 @@ def collect_from_json(path: Path): return [] hits = [] - for obj in walk_json(data): - if not isinstance(obj, dict): - continue + + def append_hit(obj: dict, metadata: dict) -> None: url = obj.get("source_url") mime = obj.get("mime_type", "") if isinstance(url, str) and url.startswith("http"): if mime.startswith("image/") or mime.startswith("video/") or is_visual_url(url): - hits.append((url, mime, str(path))) + hits.append((url, mime, str(path), metadata)) + + def collect(value): + if isinstance(value, list): + for item in value: + collect(item) + return + if not isinstance(value, dict): + return + + url = value.get("source_url") + mime = value.get("mime_type", "") + is_media = isinstance(url, str) and url.startswith("http") and ( + mime.startswith("image/") or mime.startswith("video/") or is_visual_url(url) + ) + if is_media: + metadata = extract_media_metadata(value, str(path)) + append_hit(value, metadata) + details = value.get("media_details") if isinstance(value.get("media_details"), dict) else {} + sizes = details.get("sizes") if isinstance(details.get("sizes"), dict) else {} + for size in sizes.values(): + if not isinstance(size, dict): + continue + size_metadata = dict(metadata) + size_metadata["mime"] = size.get("mime_type") or metadata.get("mime", "") + size_metadata["width"] = size.get("width") or "" + size_metadata["height"] = size.get("height") or "" + size_metadata["filesize"] = size.get("filesize") or "" + append_hit(size, size_metadata) + return + + for child in value.values(): + collect(child) + + collect(data) return hits @@ -92,10 +158,132 @@ def collect_from_seed_reports(base_dir: Path, current_run_dir: Path): if not isinstance(url, str) or not url.startswith("http") or not is_visual_url(url): continue mime = parts[mime_index] if mime_index is not None and len(parts) > mime_index else "" - hits.append((url, mime, f"seed:{run_dir.name}:{report_name}")) + hits.append( + ( + url, + mime, + f"seed:{run_dir.name}:{report_name}", + { + "media_id": "", + "parent_id": "", + "title": Path(urlparse(url).path).stem, + "slug": Path(urlparse(url).path).stem, + "mime": mime, + "width": "", + "height": "", + "filesize": "", + "source_json": f"seed:{run_dir.name}:{report_name}", + }, + ) + ) return hits +def slugify(value: str) -> str: + slug = re.sub(r"[^a-z0-9]+", "-", (value or "").lower()).strip("-") + return slug or "asset" + + +def has_size_suffix(url: str) -> bool: + stem = Path(urlparse(url).path).stem + return bool(re.search(r"-\d+x\d+$", stem)) + + +def dimension_int(value) -> int: + try: + return int(value) + except (TypeError, ValueError): + return 0 + + +def is_thumbnail_size(width, height, url: str) -> bool: + w = dimension_int(width) + h = dimension_int(height) + path = urlparse(url).path.lower() + if re.search(r"-(100x100|150x150)(?=\.[^.]+$)", path): + return True + return bool(w and h and w <= 200 and h <= 200) + + +def infer_hardware(metadata: dict, url: str, sources) -> str: + parent_id = str(metadata.get("parent_id") or "") + source_ids = set() + for source in sources or []: + source_ids.update(re.findall(r"media-parent-(?:section|product)-(\d+)", source)) + all_ids = {parent_id, *source_ids} + if all_ids & CONTROLLER_PARENT_IDS: + return "steam-controller" + if all_ids & MACHINE_PARENT_IDS: + return "steam-machine" + if all_ids & FRAME_PARENT_IDS: + return "steam-frame" + + haystack = " ".join( + [ + metadata.get("title", ""), + metadata.get("slug", ""), + url, + " ".join(sorted(sources)) if sources else "", + ] + ).lower() + if "steam-machine" in haystack or "steam machine" in haystack: + return "steam-machine" + if "steam-frame" in haystack or "steam frame" in haystack: + return "steam-frame" + if "steam-controller" in haystack or "steam controller" in haystack or "controller" in haystack: + return "steam-controller" + return "unknown" + + +def classify_library_category(metadata: dict, url: str) -> str: + mime = (metadata.get("mime") or "").lower() + title = (metadata.get("title") or metadata.get("slug") or "").lower() + extension = expected_family(url) + if mime.startswith("video/") or extension in {"mp4", "webm"}: + return "videos" + if extension == "svg" or "logo" in title: + return "logos" + if is_thumbnail_size(metadata.get("width"), metadata.get("height"), url): + return "thumbnails" + if has_size_suffix(url): + return "variants" + if "videoframe" in Path(urlparse(url).path).stem.lower() or "video frame" in title: + return "poster-frames" + if mime.startswith("image/") or extension in {"png", "jpeg", "gif", "webp", "avif"}: + return "images" + return "unknown" + + +def build_library_filename(metadata: dict, url: str, hardware: str) -> str: + media_id = metadata.get("media_id") or "no-id" + label = slugify(metadata.get("title") or metadata.get("slug") or Path(urlparse(url).path).stem) + width = dimension_int(metadata.get("width")) + height = dimension_int(metadata.get("height")) + dimensions = f"_{width}x{height}" if width and height else "" + suffix = Path(urlparse(url).path).suffix.lower() + return f"{hardware}_{media_id}_{label}{dimensions}{suffix}" + + +def resolve_unique_path(path: Path, used_paths: set[Path]) -> Path: + candidate = path + index = 2 + while candidate in used_paths or candidate.exists(): + candidate = path.with_name(f"{path.stem}-{index}{path.suffix}") + index += 1 + used_paths.add(candidate) + return candidate + + +def copy_to_library(library_dir: Path, raw_path: Path, metadata: dict, url: str, sources, used_paths: set[Path]) -> tuple[str, Path]: + hardware = infer_hardware(metadata, url, sources) + category = classify_library_category(metadata, url) + filename = build_library_filename(metadata, url, hardware) + target = resolve_unique_path(library_dir / category / filename, used_paths) + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(raw_path, target) + return category, target + + def destination_for(base_dir: Path, url: str) -> Path: parsed = urlparse(url) rel = (parsed.netloc + parsed.path).lstrip("/") @@ -211,9 +399,15 @@ def main(): base_dir = Path(args.base_dir) if args.base_dir else run_dir.parent api_dir = run_dir / "api" asset_dir = run_dir / "assets" / "discovered" + library_dir = run_dir / "assets" / "library" + manifest_dir = run_dir / "assets" / "manifests" report_dir = run_dir / "reports" report_dir.mkdir(parents=True, exist_ok=True) asset_dir.mkdir(parents=True, exist_ok=True) + if library_dir.exists(): + shutil.rmtree(library_dir) + library_dir.mkdir(parents=True, exist_ok=True) + manifest_dir.mkdir(parents=True, exist_ok=True) hits = [] for path in (api_dir / "komodo").rglob("*.json"): @@ -221,15 +415,19 @@ def main(): hits.extend(collect_from_seed_reports(base_dir, run_dir)) unique = {} - for url, mime, source in hits: - unique.setdefault(url, {"mime": mime, "sources": set()}) + for url, mime, source, metadata in hits: + unique.setdefault(url, {"mime": mime, "sources": set(), "metadata": metadata}) if mime and not unique[url]["mime"]: unique[url]["mime"] = mime unique[url]["sources"].add(source) + if metadata.get("media_id") and not unique[url]["metadata"].get("media_id"): + unique[url]["metadata"] = metadata discovered_rows = [] retrieved_rows = [] error_rows = [] + library_rows = [] + used_library_paths = set() for url in sorted(unique): discovered_rows.append( "\t".join( @@ -287,6 +485,35 @@ def main(): } if result["ok"]: + metadata = unique[url]["metadata"] + category, library_path = copy_to_library( + library_dir, + dest, + metadata, + url, + unique[url]["sources"], + used_library_paths, + ) + library_rows.append( + { + "original_url": url, + "raw_path": str(dest), + "library_path": str(library_path), + "hardware": infer_hardware(metadata, url, unique[url]["sources"]), + "category": category, + "media_id": metadata.get("media_id", ""), + "parent_id": metadata.get("parent_id", ""), + "title": metadata.get("title", ""), + "slug": metadata.get("slug", ""), + "mime_type": unique[url]["mime"], + "detected_format": result["detected"], + "width": metadata.get("width", ""), + "height": metadata.get("height", ""), + "filesize": metadata.get("filesize", ""), + "source_json": metadata.get("source_json", ""), + "sources": ",".join(sorted(unique[url]["sources"])), + } + ) retrieved_rows.append( "\t".join( [ @@ -335,6 +562,36 @@ def main(): "\n".join(retrieved_rows) + ("\n" if retrieved_rows else ""), encoding="utf-8", ) + library_rows.sort(key=lambda row: (row["category"], row["hardware"], row["library_path"])) + (manifest_dir / "asset-library.json").write_text( + json.dumps(library_rows, indent=2, ensure_ascii=True) + ("\n" if library_rows else ""), + encoding="utf-8", + ) + manifest_fields = [ + "original_url", + "raw_path", + "library_path", + "hardware", + "category", + "media_id", + "parent_id", + "title", + "slug", + "mime_type", + "detected_format", + "width", + "height", + "filesize", + "source_json", + "sources", + ] + tsv_lines = ["\t".join(manifest_fields)] + for row in library_rows: + tsv_lines.append("\t".join(str(row.get(field, "")) for field in manifest_fields)) + (manifest_dir / "asset-library.tsv").write_text( + "\n".join(tsv_lines) + "\n", + encoding="utf-8", + ) if __name__ == "__main__": diff --git a/tests/test_save_visual_assets.py b/tests/test_save_visual_assets.py index 8850082..f23ec16 100644 --- a/tests/test_save_visual_assets.py +++ b/tests/test_save_visual_assets.py @@ -41,7 +41,196 @@ class FakeResponse: return self._data +PNG_BYTES = b"\x89PNG\r\n\x1a\nfake" +AVIF_BYTES = b"\x00\x00\x00\x18ftypavif\x00\x00\x00\x00mif1avif" +MP4_BYTES = b"\x00\x00\x00\x18ftypmp42\x00\x00\x00\x00isommp42" +SVG_BYTES = b"" + + +def media_item( + media_id, + slug, + mime, + parent, + url, + title, + width="", + height="", + filesize=123, +): + details = {"filesize": filesize, "sizes": {}} + if width: + details["width"] = width + if height: + details["height"] = height + return { + "id": media_id, + "slug": slug, + "mime_type": mime, + "post": parent, + "source_url": url, + "title": {"rendered": title}, + "media_details": details, + } + + class SaveVisualAssetsTests(unittest.TestCase): + def test_extracts_wordpress_media_metadata(self): + metadata = save_visual_assets.extract_media_metadata( + { + "id": 433313, + "slug": "3-in-1-2", + "mime_type": "video/mp4", + "post": 433326, + "source_url": "https://komodostation.com/wp-content/uploads/2026/04/3-in-1.mp4", + "title": {"rendered": "3 in 1"}, + "media_details": { + "width": 1920, + "height": 850, + "filesize": 12242748, + "sizes": {}, + }, + }, + "/run/api/komodo/media-parent-section-433326.json", + ) + + self.assertEqual("433313", metadata["media_id"]) + self.assertEqual("433326", metadata["parent_id"]) + self.assertEqual("3 in 1", metadata["title"]) + self.assertEqual("3-in-1-2", metadata["slug"]) + self.assertEqual("video/mp4", metadata["mime"]) + self.assertEqual(1920, metadata["width"]) + self.assertEqual(850, metadata["height"]) + self.assertEqual(12242748, metadata["filesize"]) + self.assertEqual( + "/run/api/komodo/media-parent-section-433326.json", + metadata["source_json"], + ) + + def test_nested_wordpress_sizes_inherit_parent_metadata(self): + with TemporaryDirectory() as tmp: + path = Path(tmp) / "media.json" + path.write_text( + json.dumps( + [ + media_item( + 433477, + "spec-image", + "image/avif", + 433484, + "https://komodostation.com/wp-content/uploads/2026/04/spec-image.avif", + "spec image", + 2400, + 1158, + ) + ] + ), + encoding="utf-8", + ) + data = json.loads(path.read_text(encoding="utf-8")) + data[0]["media_details"]["sizes"] = { + "large": { + "file": "spec-image-1024x494.avif", + "width": 1024, + "height": 494, + "filesize": 456, + "mime_type": "image/avif", + "source_url": "https://komodostation.com/wp-content/uploads/2026/04/spec-image-1024x494.avif", + } + } + path.write_text(json.dumps(data), encoding="utf-8") + + hits = save_visual_assets.collect_from_json(path) + variant = next(hit for hit in hits if hit[0].endswith("-1024x494.avif")) + + self.assertEqual("image/avif", variant[1]) + self.assertEqual(str(path), variant[2]) + self.assertEqual("433477", variant[3]["media_id"]) + self.assertEqual("433484", variant[3]["parent_id"]) + self.assertEqual("spec image", variant[3]["title"]) + self.assertEqual(1024, variant[3]["width"]) + self.assertEqual(494, variant[3]["height"]) + + def test_classifies_assets_for_media_type_library(self): + cases = [ + ( + {"mime": "video/mp4", "title": "3 in 1", "width": 1920, "height": 850}, + "https://komodostation.com/wp-content/uploads/2026/04/3-in-1.mp4", + "videos", + ), + ( + {"mime": "image/svg+xml", "title": "controllerLogo", "width": "", "height": ""}, + "https://komodostation.com/wp-content/uploads/2025/11/controllerLogo.svg", + "logos", + ), + ( + {"mime": "image/png", "title": "videoframe_789", "width": 1920, "height": 1080}, + "https://komodostation.com/wp-content/uploads/2026/04/videoframe_789.png", + "poster-frames", + ), + ( + {"mime": "image/png", "title": "videoframe_789", "width": "", "height": ""}, + "https://komodostation.com/wp-content/uploads/2026/04/videoframe_789-150x150.png", + "thumbnails", + ), + ( + {"mime": "image/png", "title": "videoframe_789", "width": "", "height": ""}, + "https://komodostation.com/wp-content/uploads/2026/04/videoframe_789-1024x576.png", + "variants", + ), + ( + {"mime": "image/avif", "title": "spec image", "width": 150, "height": 150}, + "https://komodostation.com/wp-content/uploads/2026/04/d760b59ee00769c931d0fb73f498ab74-150x150.avif", + "thumbnails", + ), + ( + {"mime": "image/avif", "title": "spec image", "width": 2048, "height": 988}, + "https://komodostation.com/wp-content/uploads/2026/04/d760b59ee00769c931d0fb73f498ab74-2048x988.avif", + "variants", + ), + ( + {"mime": "image/avif", "title": "spec image", "width": 2048, "height": 988}, + "https://komodostation.com/wp-content/uploads/2026/04/d760b59ee00769c931d0fb73f498ab74.avif", + "images", + ), + ] + + for metadata, url, expected in cases: + with self.subTest(url=url): + self.assertEqual( + expected, + save_visual_assets.classify_library_category(metadata, url), + ) + + def test_builds_readable_library_filename(self): + metadata = { + "media_id": "433313", + "title": "3 in 1", + "width": 1920, + "height": 850, + } + + self.assertEqual( + "steam-controller_433313_3-in-1_1920x850.mp4", + save_visual_assets.build_library_filename( + metadata, + "https://komodostation.com/wp-content/uploads/2026/04/3-in-1.mp4", + "steam-controller", + ), + ) + + def test_infers_hardware_from_source_section_without_treating_videoframe_as_frame(self): + self.assertEqual( + "steam-controller", + save_visual_assets.infer_hardware( + {"title": "videoframe_789", "slug": "videoframe_789", "parent_id": ""}, + "https://komodostation.com/wp-content/uploads/2026/04/videoframe_789-1024x576.png", + { + "/Users/sean/steam_hardware_watch/2026-04-26/api/komodo/media-parent-section-433306.json" + }, + ), + ) + def test_downloads_fetchable_komodo_avif_assets(self): with TemporaryDirectory() as tmp: run_dir = Path(tmp) / "2026-04-26" @@ -90,6 +279,121 @@ class SaveVisualAssetsTests(unittest.TestCase): self.assertEqual(avif_bytes, saved.read_bytes()) self.assertEqual(1, urlopen.call_count) + def test_generates_media_type_library_and_manifests(self): + with TemporaryDirectory() as tmp: + run_dir = Path(tmp) / "2026-04-26" + api_dir = run_dir / "api" / "komodo" + api_dir.mkdir(parents=True) + items = [ + media_item( + 433313, + "3-in-1-2", + "video/mp4", + 433326, + "https://komodostation.com/wp-content/uploads/2026/04/3-in-1.mp4", + "3 in 1", + 1920, + 850, + ), + media_item( + 413728, + "controllerlogo", + "image/svg+xml", + 413763, + "https://komodostation.com/wp-content/uploads/2025/11/controllerLogo.svg", + "controller logo", + ), + media_item( + 433298, + "videoframe_789-2", + "image/png", + 433306, + "https://komodostation.com/wp-content/uploads/2026/04/videoframe_789.png", + "videoframe_789", + 1920, + 1080, + ), + media_item( + 433299, + "videoframe_789-150x150", + "image/png", + 433306, + "https://komodostation.com/wp-content/uploads/2026/04/videoframe_789-150x150.png", + "videoframe_789", + 150, + 150, + ), + media_item( + 433477, + "spec-image", + "image/avif", + 433484, + "https://komodostation.com/wp-content/uploads/2026/04/spec-image-2048x988.avif", + "spec image", + 2048, + 988, + ), + ] + (api_dir / "media-parent-section-433326.json").write_text( + json.dumps(items), + encoding="utf-8", + ) + + def fake_urlopen(request, timeout=15): + url = request.full_url + if url.endswith(".mp4"): + return FakeResponse("video/mp4", MP4_BYTES) + if url.endswith(".svg"): + return FakeResponse("image/svg+xml", SVG_BYTES) + if url.endswith(".avif"): + return FakeResponse("image/avif", AVIF_BYTES) + return FakeResponse("image/png", PNG_BYTES) + + with patch.object( + sys, + "argv", + [ + "save_visual_assets.py", + "--run-dir", + str(run_dir), + "--base-dir", + str(Path(tmp)), + ], + ), patch("urllib.request.urlopen", side_effect=fake_urlopen): + save_visual_assets.main() + + expected_paths = [ + run_dir / "assets/library/videos/steam-controller_433313_3-in-1_1920x850.mp4", + run_dir / "assets/library/logos/steam-controller_413728_controller-logo.svg", + run_dir / "assets/library/poster-frames/steam-controller_433298_videoframe-789_1920x1080.png", + run_dir / "assets/library/thumbnails/steam-controller_433299_videoframe-789_150x150.png", + run_dir / "assets/library/variants/steam-controller_433477_spec-image_2048x988.avif", + ] + for path in expected_paths: + with self.subTest(path=path): + self.assertTrue(path.exists(), path) + + manifest_json = run_dir / "assets/manifests/asset-library.json" + manifest_tsv = run_dir / "assets/manifests/asset-library.tsv" + self.assertTrue(manifest_json.exists()) + self.assertTrue(manifest_tsv.exists()) + rows = json.loads(manifest_json.read_text(encoding="utf-8")) + self.assertEqual(5, len(rows)) + video_row = next(row for row in rows if row["category"] == "videos") + self.assertEqual("steam-controller", video_row["hardware"]) + self.assertEqual("433313", video_row["media_id"]) + self.assertEqual("433326", video_row["parent_id"]) + self.assertEqual("1920", str(video_row["width"])) + self.assertEqual("850", str(video_row["height"])) + self.assertEqual( + "https://komodostation.com/wp-content/uploads/2026/04/3-in-1.mp4", + video_row["original_url"], + ) + self.assertIn("assets/discovered/", video_row["raw_path"]) + self.assertIn("assets/library/videos/", video_row["library_path"]) + self.assertIn("media-parent-section-433326.json", video_row["source_json"]) + self.assertIn("original_url\t", manifest_tsv.read_text(encoding="utf-8")) + if __name__ == "__main__": unittest.main()