#!/usr/bin/env python3
"""
changes.py — deterministic structured edits for the initial clone.

The team provides EXACT strings (brand, product, CTA URL) in the form, so there is no judgment
about WHAT to change — only WHERE. The OpenHands agent proved unreliable at these (churns to the
iteration cap without applying them), so we do them deterministically:
  • brand/product → visible text + safe text attributes only (never file paths, URLs, scripts,
    or class names),
  • CTA → primary <a> hrefs (skipping nav/footer/social/legal links).
"""

import re
from pathlib import Path
from urllib.parse import urlparse

_SAFE_ATTRS = ("title", "alt", "placeholder", "aria-label")  # 'content' handled separately (URLs)
_SKIP_LINK = re.compile(
    r'privacy|terms|policy|cookie|facebook|instagram|twitter|x\.com|tiktok|youtube|linkedin|'
    r'whatsapp|mailto:|tel:', re.IGNORECASE)


def _dominant_outbound(html, min_count=3):
    """The most common external <a> href that isn't nav/footer/social/legal — i.e. the CTA URL on
    pages that use a real repeated link. Returns it only if it appears >= min_count times."""
    from collections import Counter
    hrefs = re.findall(r'<a\b[^>]*\bhref\s*=\s*["\']([^"\']+)["\']', html, re.IGNORECASE)
    cand = [h for h in hrefs if h.lower().startswith(("http://", "https://")) and not _SKIP_LINK.search(h)]
    if not cand:
        return None
    url, cnt = Counter(cand).most_common(1)[0]
    return url if cnt >= min_count else None


def _mask(html):
    """Hide <script>/<style> blocks so brand/product edits never touch JS/CSS."""
    blocks = []
    def m(mt):
        blocks.append(mt.group(0))
        return f"\x00{len(blocks) - 1}\x00"
    return re.sub(r'<(script|style)\b[^>]*>.*?</\1>', m, html, flags=re.DOTALL | re.IGNORECASE), blocks


def _unmask(html, blocks):
    return re.sub(r'\x00(\d+)\x00', lambda mt: blocks[int(mt.group(1))], html)


def _replace_textual(html, old, new):
    """Replace old->new (case-insensitive) in visible text + safe text attributes only."""
    masked, blocks = _mask(html)
    pat = re.compile(re.escape(old), re.IGNORECASE)
    n = [0]
    def sub(s):
        out, c = pat.subn(new, s)
        n[0] += c
        return out
    # visible text nodes (between > and <)
    masked = re.sub(r'>([^<]+)<', lambda mt: '>' + sub(mt.group(1)) + '<', masked)
    # Attributes are skipped when the replacement contains HTML — a <script>/tag inside alt/title/
    # content would not execute and would break the attribute. Only plain-text swaps touch attrs.
    if '<' not in new and '>' not in new:
        for attr in _SAFE_ATTRS:
            masked = re.sub(rf'({attr}\s*=\s*")([^"]*)(")',
                            lambda mt: mt.group(1) + sub(mt.group(2)) + mt.group(3),
                            masked, flags=re.IGNORECASE)
            masked = re.sub(rf"({attr}\s*=\s*')([^']*)(')",
                            lambda mt: mt.group(1) + sub(mt.group(2)) + mt.group(3),
                            masked, flags=re.IGNORECASE)
        # meta content="" — replace only when it's text, not a URL/path (protects og:url, og:image)
        def content_sub(mt):
            val = mt.group(2)
            if '://' in val or val.strip().startswith('/'):
                return mt.group(0)
            return mt.group(1) + sub(val) + mt.group(3)
        masked = re.sub(r'(content\s*=\s*")([^"]*)(")', content_sub, masked, flags=re.IGNORECASE)
    return _unmask(masked, blocks), n[0]


def _absolutize(url):
    """A CTA URL with no scheme (e.g. 'doggo.com/click') resolves RELATIVE to the page (giving
    http://<host>/<job>/doggo.com/click). Prepend https:// so it's an absolute link."""
    url = (url or "").strip()
    if url and "://" not in url and not url.startswith(("//", "mailto:", "tel:", "#")):
        url = "https://" + url
    return url


def _replace_cta(html, cta_url, page_slug=""):
    """Make every CTA a DIRECT static link. A CTA is an <a> whose href is a PLACEHOLDER
    (javascript:void(0), '#', empty) OR a self-link to the offer page itself (e.g. view44902.html
    — the page's own slug). Those become href="<cta>" target="_blank". Other real-URL links
    (footer / nav / social) are left untouched. Then neutralize the page's link-rewriting loop so
    no script can re-touch the CTA hrefs."""
    cta_url = _absolutize(cta_url)
    slugs = set()
    if page_slug:
        s = page_slug.lower()
        slugs = {s, f"{s}.html", f"{s}.htm"}
    # Also catch templates whose CTAs are a real external URL repeated on every button (e.g.
    # https://track.../click/1). The dominant outbound href that repeats >=3x is the CTA; the
    # threshold keeps one-off footer/nav links safe.
    dominant = _dominant_outbound(html, min_count=3)
    n = [0]

    def _is_cta(href):
        h = href.strip()
        if h == "" or h == "#" or h.lower().startswith("javascript:"):
            return True
        if dominant and h == dominant:
            return True
        # self-link to the offer page: compare the final path segment to the page's slug
        base = h.split("?")[0].split("#")[0].rstrip("/").split("/")[-1].lower()
        return base in slugs

    def repl(mt):
        tag = mt.group(0)  # full <a ...> opening tag
        hrefm = re.search(r'href\s*=\s*(["\'])(.*?)\1', tag, re.IGNORECASE)
        if not hrefm or not _is_cta(hrefm.group(2)):
            return tag
        n[0] += 1
        tag = tag[:hrefm.start()] + f'href="{cta_url}"' + tag[hrefm.end():]
        if not re.search(r'\btarget\s*=', tag, re.IGNORECASE):
            tag = re.sub(r'^<a\b', '<a target="_blank"', tag, count=1, flags=re.IGNORECASE)
        return tag

    html = re.sub(r'<a\b[^>]*>', repl, html, flags=re.IGNORECASE)

    # Kill the link-rewriting loop: affiliate pages do `var x = document.querySelectorAll('a')`
    # then reset every href to '#' and intercept clicks (Goto). Replacing the document-wide anchor
    # grab with [] makes that loop a no-op, so the static CTA hrefs above stand as plain links and
    # NO script touches them. Element-scoped selectors (e.g. swipeable.querySelectorAll) stay intact.
    html = re.sub(r"document\.querySelectorAll\(\s*(['\"])a\1\s*\)", "[]", html)
    return html, n[0]


def apply(index_path: Path, meta: dict) -> str:
    """Apply brand/product/CTA from the job meta. Returns a short summary string."""
    index_path = Path(index_path)
    html = index_path.read_text(encoding="utf-8", errors="ignore")
    notes = []

    ob, nb = meta.get("orig_brand", ""), meta.get("our_brand", "")
    if ob and nb:
        html, c = _replace_textual(html, ob, nb)
        # Also the affiliate `?brand=<orig>` query param (inert JS string) so NO original-brand
        # reference remains anywhere — safe: it's a query-string value, not a JS identifier.
        html, c2 = re.subn(r'brand=' + re.escape(ob), lambda m: 'brand=' + nb, html, flags=re.IGNORECASE)
        notes.append(f"brand×{c + c2}")
    op, npd = meta.get("orig_product", ""), meta.get("our_product", "")
    if op and npd:
        html, c = _replace_textual(html, op, npd)
        notes.append(f"product×{c}")
    cta = meta.get("cta_url", "")
    if cta:
        slug = ""
        src = meta.get("url", "")
        if src:
            slug = urlparse(src).path.strip("/").split("/")[-1]
        html, c = _replace_cta(html, cta, slug)
        notes.append(f"CTA×{c}")

    index_path.write_text(html, encoding="utf-8")
    return ", ".join(notes) if notes else "none requested"
