197 lines
9 KiB
Python
197 lines
9 KiB
Python
|
|
"""
|
||
|
|
Thumbnail Generation Module for A/B Testing.
|
||
|
|
Generates 3 distinct high-impact visual concepts via NvidiaLLM ComfyUI (z-image-turbo/SDXL)
|
||
|
|
and renders crisp typography overlays for both English and Spanish + clean raw versions.
|
||
|
|
"""
|
||
|
|
|
||
|
|
import os
|
||
|
|
import re
|
||
|
|
import zlib
|
||
|
|
import logging
|
||
|
|
from typing import Dict, Any, List, Optional
|
||
|
|
from PIL import Image, ImageDraw, ImageFont
|
||
|
|
from .remote import RemotePC
|
||
|
|
from .visual_gen import VisualGen, StockMediaFetcher
|
||
|
|
|
||
|
|
logger = logging.getLogger("YTFactory.Thumb")
|
||
|
|
|
||
|
|
THUMB_PROMPT_STYLES = [
|
||
|
|
"hyper-realistic dramatic close-up, intense emotional expression, high-contrast cinematic lighting, 8k wallpaper",
|
||
|
|
"epic wide-angle establishing shot, vast mysterious atmosphere, vibrant golden hour illumination, ultra-detailed",
|
||
|
|
"striking high-contrast stylized illustration, vivid neon backlight, bold cinematic composition, intriguing perspective",
|
||
|
|
]
|
||
|
|
|
||
|
|
class ThumbnailGen:
|
||
|
|
def __init__(self, pcs: Dict[str, RemotePC], cfg: Any):
|
||
|
|
self.pcs = pcs
|
||
|
|
self.cfg = cfg
|
||
|
|
self.visual = VisualGen(pcs, cfg)
|
||
|
|
|
||
|
|
def _render_typography_overlay(self, bg_image_path: str, title: str, out_path: str) -> str:
|
||
|
|
"""Render high-impact YouTube thumbnail typography over background image using Pillow."""
|
||
|
|
try:
|
||
|
|
img = Image.open(bg_image_path).convert("RGBA")
|
||
|
|
img = img.resize((1280, 720), Image.Resampling.LANCZOS)
|
||
|
|
|
||
|
|
# Format punchy 3-5 word headline
|
||
|
|
clean_title = re.sub(r'[^A-Za-z0-9 \-\?\!\,\']', '', title).strip().upper()
|
||
|
|
words = clean_title.split()
|
||
|
|
if len(words) > 6:
|
||
|
|
words = words[:6]
|
||
|
|
|
||
|
|
# Group into 2 lines if needed
|
||
|
|
if len(words) > 3:
|
||
|
|
mid = len(words) // 2
|
||
|
|
lines = [" ".join(words[:mid]), " ".join(words[mid:])]
|
||
|
|
else:
|
||
|
|
lines = [" ".join(words)]
|
||
|
|
|
||
|
|
# Locate bold font
|
||
|
|
font_paths = [
|
||
|
|
"/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf",
|
||
|
|
"/usr/share/fonts/TTF/DejaVuSans-Bold.ttf",
|
||
|
|
"/usr/share/fonts/truetype/liberation/LiberationSans-Bold.ttf",
|
||
|
|
"/usr/share/fonts/truetype/freefont/FreeSansBold.ttf",
|
||
|
|
]
|
||
|
|
font_file = None
|
||
|
|
for fp in font_paths:
|
||
|
|
if os.path.exists(fp):
|
||
|
|
font_file = fp
|
||
|
|
break
|
||
|
|
|
||
|
|
font_size = 58 if len(lines) > 1 else 68
|
||
|
|
font = ImageFont.truetype(font_file, font_size) if font_file else ImageFont.load_default()
|
||
|
|
|
||
|
|
overlay = Image.new("RGBA", img.size, (0, 0, 0, 0))
|
||
|
|
draw = ImageDraw.Draw(overlay)
|
||
|
|
|
||
|
|
# Calculate box dimensions
|
||
|
|
line_height = font_size + 16
|
||
|
|
total_text_height = len(lines) * line_height
|
||
|
|
max_line_width = 0
|
||
|
|
for line in lines:
|
||
|
|
bbox = draw.textbbox((0, 0), line, font=font)
|
||
|
|
max_line_width = max(max_line_width, bbox[2] - bbox[0])
|
||
|
|
|
||
|
|
# Draw semi-transparent dark banner backing
|
||
|
|
banner_padding = 24
|
||
|
|
banner_top = 720 - total_text_height - 60 - banner_padding
|
||
|
|
banner_bottom = 720 - 60 + banner_padding
|
||
|
|
banner_left = max(20, int((1280 - max_line_width) / 2) - banner_padding)
|
||
|
|
banner_right = min(1260, int((1280 + max_line_width) / 2) + banner_padding)
|
||
|
|
|
||
|
|
draw.rounded_rectangle(
|
||
|
|
[banner_left, banner_top, banner_right, banner_bottom],
|
||
|
|
radius=16,
|
||
|
|
fill=(0, 0, 0, 180),
|
||
|
|
outline=(255, 204, 0, 220), # Vibrant yellow-gold accent border
|
||
|
|
width=3
|
||
|
|
)
|
||
|
|
|
||
|
|
# Draw text with dark drop shadow and bright yellow/white fill
|
||
|
|
cur_y = banner_top + banner_padding
|
||
|
|
for i, line in enumerate(lines):
|
||
|
|
bbox = draw.textbbox((0, 0), line, font=font)
|
||
|
|
line_w = bbox[2] - bbox[0]
|
||
|
|
cur_x = int((1280 - line_w) / 2)
|
||
|
|
|
||
|
|
# Shadow
|
||
|
|
draw.text((cur_x + 3, cur_y + 3), line, font=font, fill=(0, 0, 0, 255))
|
||
|
|
# Text fill (Top line yellow, bottom line white)
|
||
|
|
fill_color = (255, 220, 0, 255) if i == 0 else (255, 255, 255, 255)
|
||
|
|
draw.text((cur_x, cur_y), line, font=font, fill=fill_color)
|
||
|
|
cur_y += line_height
|
||
|
|
|
||
|
|
final_img = Image.alpha_composite(img, overlay).convert("RGB")
|
||
|
|
final_img.save(out_path, "PNG", quality=95)
|
||
|
|
return out_path
|
||
|
|
except Exception as e:
|
||
|
|
logger.warning(f"[THUMB] Pillow typography overlay failed: {e}. Copying raw image.")
|
||
|
|
import shutil
|
||
|
|
shutil.copy(bg_image_path, out_path)
|
||
|
|
return out_path
|
||
|
|
|
||
|
|
def generate(self, script: Dict[str, Any], topic: str, job_dir: str) -> Dict[str, List[str]]:
|
||
|
|
"""
|
||
|
|
Generate 3 A/B test thumbnail concepts in English, Spanish, and Clean.
|
||
|
|
Returns: { 'en': [...], 'es': [...], 'clean': [...] }
|
||
|
|
"""
|
||
|
|
n = getattr(self.cfg, "thumbnail_count", 3)
|
||
|
|
logger.info(f"[THUMB] Generating {n} distinct A/B thumbnail variations on NvidiaLLM...")
|
||
|
|
pc = self.pcs.get("nvidiam")
|
||
|
|
if not pc or not pc.check(8188):
|
||
|
|
logger.warning("[THUMB] NvidiaLLM ComfyUI unreachable, skipping AI thumbnails")
|
||
|
|
return {"en": [], "es": [], "clean": []}
|
||
|
|
|
||
|
|
thumbs_dir = os.path.join(job_dir, "thumbs")
|
||
|
|
os.makedirs(thumbs_dir, exist_ok=True)
|
||
|
|
|
||
|
|
titles_en = script.get("titles", [topic])
|
||
|
|
titles_es = script.get("titles_es", [topic])
|
||
|
|
|
||
|
|
results = {"en": [], "es": [], "clean": []}
|
||
|
|
|
||
|
|
for i in range(n):
|
||
|
|
variant_letter = chr(65 + i) # A, B, C
|
||
|
|
style = THUMB_PROMPT_STYLES[i % len(THUMB_PROMPT_STYLES)]
|
||
|
|
prompt = f"YouTube documentary thumbnail for '{topic}', {style}, no text, no logos, 16:9 widescreen"
|
||
|
|
seed = zlib.crc32(prompt.encode()) % (2**31)
|
||
|
|
|
||
|
|
clean_path = os.path.join(thumbs_dir, f"thumb_{variant_letter}_clean.png")
|
||
|
|
en_path = os.path.join(thumbs_dir, f"thumb_{variant_letter}_en.png")
|
||
|
|
es_path = os.path.join(thumbs_dir, f"thumb_{variant_letter}_es.png")
|
||
|
|
|
||
|
|
# 1. Generate Clean Background via z-image-turbo or SDXL
|
||
|
|
generated = False
|
||
|
|
for wf in (self.visual._sdxl_workflow(prompt, seed), self._zimage_workflow(prompt, seed)):
|
||
|
|
try:
|
||
|
|
pid = self.visual._submit(pc.host, wf)
|
||
|
|
outputs = self.visual._wait_result(pc.host, pid, timeout_s=600)
|
||
|
|
self.visual._download_outputs(pc.host, outputs, clean_path)
|
||
|
|
generated = True
|
||
|
|
break
|
||
|
|
except Exception as e:
|
||
|
|
logger.debug(f"[THUMB] Variant {variant_letter} generator attempt failed: {e}")
|
||
|
|
|
||
|
|
if not generated:
|
||
|
|
# Stock image fallback
|
||
|
|
logger.info(f"[THUMB] Variant {variant_letter} falling back to stock image...")
|
||
|
|
self.visual.stock_fetcher.search_and_download(topic, clean_path, prefer_video=False)
|
||
|
|
|
||
|
|
if not os.path.exists(clean_path):
|
||
|
|
logger.error(f"[THUMB] Variant {variant_letter} failed to produce background image.")
|
||
|
|
continue
|
||
|
|
|
||
|
|
results["clean"].append(clean_path)
|
||
|
|
|
||
|
|
# 2. Render English Overlay
|
||
|
|
title_en = titles_en[i % len(titles_en)]
|
||
|
|
self._render_typography_overlay(clean_path, title_en, en_path)
|
||
|
|
results["en"].append(en_path)
|
||
|
|
|
||
|
|
# 3. Render Spanish Overlay
|
||
|
|
title_es = titles_es[i % len(titles_es)]
|
||
|
|
self._render_typography_overlay(clean_path, title_es, es_path)
|
||
|
|
results["es"].append(es_path)
|
||
|
|
|
||
|
|
logger.info(f"[THUMB] Concept {variant_letter} OK (EN, ES, Clean)")
|
||
|
|
|
||
|
|
return results
|
||
|
|
|
||
|
|
def _zimage_workflow(self, prompt: str, seed: int) -> Dict[str, Any]:
|
||
|
|
from .visual_gen import ZIMAGE_UNET, ZIMAGE_CLIP, ZIMAGE_VAE
|
||
|
|
return {
|
||
|
|
"1": {"inputs": {"unet_name": ZIMAGE_UNET, "weight_dtype": "default"}, "class_type": "UNETLoader"},
|
||
|
|
"2": {"inputs": {"clip_name": ZIMAGE_CLIP, "type": "lumina2"}, "class_type": "CLIPLoader"},
|
||
|
|
"3": {"inputs": {"vae_name": ZIMAGE_VAE}, "class_type": "VAELoader"},
|
||
|
|
"4": {"inputs": {"text": prompt, "clip": ["2", 0]}, "class_type": "CLIPTextEncode"},
|
||
|
|
"5": {"inputs": {"text": "", "clip": ["2", 0]}, "class_type": "CLIPTextEncode"},
|
||
|
|
"6": {"inputs": {"width": 1280, "height": 720, "batch_size": 1}, "class_type": "EmptyLatentImage"},
|
||
|
|
"7": {"inputs": {"seed": seed, "steps": 8, "cfg": 1.0, "sampler_name": "euler",
|
||
|
|
"scheduler": "simple", "denoise": 1.0,
|
||
|
|
"model": ["1", 0], "positive": ["4", 0], "negative": ["5", 0],
|
||
|
|
"latent_image": ["6", 0]}, "class_type": "KSampler"},
|
||
|
|
"8": {"inputs": {"samples": ["7", 0], "vae": ["3", 0]}, "class_type": "VAEDecode"},
|
||
|
|
"9": {"inputs": {"filename_prefix": "yt_thumb", "images": ["8", 0]}, "class_type": "SaveImage"},
|
||
|
|
}
|