#!/usr/bin/env python3
"""
job_common.py — job-folder helpers shared by every frontend (Slack bot, HTTP API, ...).

Nothing here is Slack- or HTTP-specific: sanitizing a job name, writing job.md, the
tracking-domain/brand/CTA output guardrail, and zipping the deliverable are the same
regardless of what triggered the job.
"""

import re
import zipfile
from pathlib import Path

# Tracking domains used only for the independent output guardrail (not for cleaning —
# the agent does the cleaning; this just verifies it actually happened).
TRACKING_DOMAINS = [
    r'googletagmanager\.com', r'google-analytics\.com', r'connect\.facebook\.net',
    r'sc-static\.net', r'static\.hotjar\.com', r'clarity\.ms', r'cdn\.segment\.com',
    r'cdn\.amplitude\.com', r'fullstory\.com', r'a\.klaviyo\.com', r'config-security\.com',
]


def sanitize_job_name(raw: str) -> str:
    return re.sub(r"[^\w\-]", "-", raw).lower().strip("-")


def build_job_md(job_name, url, orig_brand, our_brand, orig_product, our_product, cta_url):
    lines = [
        f"# Job: {job_name}", "",
        "## Source",
        f"- **URL:** {url}", "",
    ]
    if orig_brand or our_brand or orig_product or our_product:
        lines += [
            "## Branding",
            f"- **Original brand:** {orig_brand or '(none — do not change brands)'}",
            f"- **Our brand:** {our_brand or '(none)'}",
            f"- **Original product:** {orig_product or '(none — do not change products)'}",
            f"- **Our product:** {our_product or '(none)'}", "",
        ]
    else:
        lines += ["## Branding", "- None requested — do NOT change any brand or product names.", ""]
    if cta_url:
        lines += ["## CTA", f"- **Our CTA URL:** {cta_url}", ""]
    else:
        lines += ["## CTA", "- None requested — do NOT change any links or CTAs.", ""]
    lines += [
        "## Text / asset changes",
        "_Applied from the change-request thread/history. New media is in assets-new/._",
    ]
    return "\n".join(lines) + "\n"


def verify_job_output(job_dir: Path, meta: dict) -> list:
    """Independent guardrail checked against what the pipeline claims — not a substitute for it."""
    checks = []
    idx = job_dir / "index.html"
    if not idx.exists():
        return ["❌ index.html does NOT exist — the clone was not produced"]
    checks.append("✅ index.html exists")
    content = idx.read_text(encoding="utf-8", errors="ignore")

    if any(job_dir.rglob("*.css")) or (job_dir / "assets").exists():
        checks.append("✅ assets present")
    else:
        checks.append("⚠️ no CSS/assets found near the page")

    if any(re.search(d, content, re.IGNORECASE) for d in TRACKING_DOMAINS):
        checks.append("❌ a known tracking domain is STILL present")
    else:
        checks.append("✅ no known tracking domains found")

    ob, nb = meta.get("orig_brand"), meta.get("our_brand")
    if ob and nb:
        checks.append(f"❌ original brand '{ob}' still present"
                      if re.search(re.escape(ob), content, re.IGNORECASE)
                      else f"✅ brand replaced ('{ob}' gone)")
    op, npd = meta.get("orig_product"), meta.get("our_product")
    if op and npd:
        checks.append(f"❌ original product '{op}' still present"
                      if re.search(re.escape(op), content, re.IGNORECASE)
                      else f"✅ product replaced ('{op}' gone)")
    cta = meta.get("cta_url")
    if cta:
        checks.append("✅ CTA URL found in index.html" if cta in content
                      else "❌ CTA URL NOT found in index.html")
    return checks


def make_zip(job_dir: Path) -> Path:
    """Zip the clean deliverable (index.html + assets/) to <JOBS_DIR>/<job>.zip so it can be
    downloaded at SERVER_URL/<job>.zip. Excludes the inbox, logs, and per-job config."""
    zip_path = job_dir.parent / f"{job_dir.name}.zip"
    with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as z:
        idx = job_dir / "index.html"
        if idx.exists():
            z.write(idx, "index.html")
        assets = job_dir / "assets"
        if assets.is_dir():
            for f in assets.rglob("*"):
                if f.is_file():
                    z.write(f, str(f.relative_to(job_dir)).replace("\\", "/"))
    return zip_path
