#!/usr/bin/env python3
"""
agent.py — minimal in-house tool-calling loop for free-form CHANGE REQUESTS.

Replaces OpenHands, whose state machine repeatedly stuck/looped. Talks straight to the local
vLLM endpoint with OpenAI-style tool calling (proven fast + correct in isolation). The model
edits files in the job folder via tools, then calls done(). Nothing leaves the box.
"""

import os
import re
import json
import shutil
import subprocess
from html import unescape as _html_unescape
from pathlib import Path

from openai import OpenAI

import changes  # reuse deterministic helpers (_absolutize, _replace_textual)
import verify   # headless render/health gate (rung 0)
import planner  # second model: plans the approach + reviews the result against intent (rung 2)

USE_PLANNER = os.environ.get("USE_PLANNER", "1") == "1"

LLM_BASE_URL = os.environ.get("LLM_BASE_URL", "http://127.0.0.1:8000/v1")
LLM_MODEL    = os.environ.get("LLM_MODEL", "devstral").split("/")[-1]  # vLLM served name
LLM_API_KEY  = os.environ.get("LLM_API_KEY", "dummy")

_TOOLS = [
    {"type": "function", "function": {
        "name": "bash",
        "description": "Run a shell command in the job folder (grep/sed are handy for finding and editing).",
        "parameters": {"type": "object", "properties": {"cmd": {"type": "string"}}, "required": ["cmd"]}}},
    {"type": "function", "function": {
        "name": "read_file",
        "description": "Read a file, path relative to the job folder.",
        "parameters": {"type": "object", "properties": {"path": {"type": "string"}}, "required": ["path"]}}},
    {"type": "function", "function": {
        "name": "replace_in_file",
        "description": "Replace every occurrence of `old` with `new` in a file. `old` must match exactly.",
        "parameters": {"type": "object", "properties": {
            "path": {"type": "string"}, "old": {"type": "string"}, "new": {"type": "string"}},
            "required": ["path", "old", "new"]}}},
    {"type": "function", "function": {
        "name": "edit_link",
        "description": "Change the href of the link (<a>) whose VISIBLE TEXT contains `text`. Works on "
                       "minified HTML. Use for requests like 'change the Contact Us link to <url>'. "
                       "Sets target=_blank automatically.",
        "parameters": {"type": "object", "properties": {
            "text": {"type": "string", "description": "visible text of the link, e.g. 'Contact Us'"},
            "url": {"type": "string"}}, "required": ["text", "url"]}}},
    {"type": "function", "function": {
        "name": "replace_text",
        "description": "Replace VISIBLE text `old` with `new` across the page (never touches URLs, "
                       "file paths, scripts, or class names). Use for wording changes.",
        "parameters": {"type": "object", "properties": {
            "old": {"type": "string"}, "new": {"type": "string"}}, "required": ["old", "new"]}}},
    {"type": "function", "function": {
        "name": "set_cta",
        "description": "Change ALL primary CTA links on the page to a new URL. Use for 'change all "
                       "CTA links to <url>'. A URL with no scheme becomes https://.",
        "parameters": {"type": "object", "properties": {"url": {"type": "string"}}, "required": ["url"]}}},
    {"type": "function", "function": {
        "name": "add_script",
        "description": "Insert a <script> (or any HTML snippet) into the page. Use for 'add this "
                       "script to the html'. where='head' (default — runs early, for code that "
                       "defines variables used elsewhere) or 'body_end'.",
        "parameters": {"type": "object", "properties": {
            "code": {"type": "string", "description": "the script/HTML to insert (with or without <script> tags)"},
            "where": {"type": "string", "enum": ["head", "body_end"]}}, "required": ["code"]}}},
    {"type": "function", "function": {
        "name": "set_favicon",
        "description": "Set the page favicon to a newly uploaded image (from assets-new/). Use for "
                       "'change the favicon to this'. Copies the file into assets/ and updates the "
                       "<link rel=icon> href.",
        "parameters": {"type": "object", "properties": {
            "new_file": {"type": "string", "description": "uploaded filename"}}, "required": ["new_file"]}}},
    {"type": "function", "function": {
        "name": "set_logo",
        "description": "Set the header logo to a specific IMAGE FILE — either a newly uploaded file "
                       "(assets-new/) or an existing asset (assets/images/). Use for 'change the "
                       "logo to <filename>' (e.g. aeriocool-logo.png). For a TEXT wordmark, use "
                       "make_logo instead.",
        "parameters": {"type": "object", "properties": {
            "image": {"type": "string", "description": "image filename or path, e.g. aeriocool-logo.png"}},
            "required": ["image"]}}},
    {"type": "function", "function": {
        "name": "make_logo",
        "description": "GENERATE a text wordmark logo (PNG) from given TEXT and set it as the logo. "
                       "Use ONLY when the user gives words to render (e.g. 'make the logo say Aerio'), "
                       "NOT when they give an image filename — use set_logo for a file.",
        "parameters": {"type": "object", "properties": {
            "text": {"type": "string"}}, "required": ["text"]}}},
    {"type": "function", "function": {
        "name": "remove_section",
        "description": "Remove a whole section/block from the page, identified by distinctive text it "
                       "contains. Use for 'remove the section about X' / 'delete the part that says X'. "
                       "Locates the exact block by RENDERING the page and deletes only that element "
                       "subtree — it will NOT remove the wrong section.",
        "parameters": {"type": "object", "properties": {
            "text": {"type": "string", "description": "distinctive visible text inside the section to remove"}},
            "required": ["text"]}}},
    {"type": "function", "function": {
        "name": "set_hero_image",
        "description": "Replace the HERO image (the big image at the top of the page) with an "
                       "uploaded/existing file. Use for 'change the hero image to <file>'. Locates the "
                       "hero by RENDERING the page and finding the largest image at the top — reliable "
                       "even on minified HTML, where guessing the first <img> picks the wrong one.",
        "parameters": {"type": "object", "properties": {
            "new_file": {"type": "string", "description": "uploaded/existing filename, e.g. F1.jpg"}},
            "required": ["new_file"]}}},
    {"type": "function", "function": {
        "name": "swap_media",
        "description": "Replace the image OR video under/after a section heading with a newly "
                       "uploaded file (from assets-new/). Use for 'change the image/video under "
                       "<heading> to this'. Picks images/ or media/ by file type and emits <img> or "
                       "<video> as appropriate. Pass section_text as the PLAIN visible heading text "
                       "exactly as a person reads it — do NOT HTML-encode quotes or punctuation.",
        "parameters": {"type": "object", "properties": {
            "section_text": {"type": "string", "description": "text of the heading the media sits under"},
            "new_file": {"type": "string", "description": "uploaded filename, e.g. messi-1.jpg or stadium-1.mp4"}},
            "required": ["section_text", "new_file"]}}},
    {"type": "function", "function": {
        "name": "done",
        "description": "Call when the requested change is complete.",
        "parameters": {"type": "object", "properties": {"summary": {"type": "string"}}, "required": ["summary"]}}},
]


_SKIP_LINK = re.compile(
    r'privacy|terms|policy|cookie|facebook|instagram|twitter|x\.com|tiktok|youtube|linkedin|'
    r'whatsapp|mailto:|tel:', re.IGNORECASE)


def _set_cta(job_dir, url):
    """Change every primary CTA link to `url`. The current CTA is the most common outbound
    (http/https) <a> href that isn't a nav/footer/social/legal link."""
    from collections import Counter
    url = changes._absolutize(url)
    p = job_dir / "index.html"
    html = p.read_text(encoding="utf-8", errors="ignore")
    hrefs = re.findall(r'<a\b[^>]*\bhref\s*=\s*["\']([^"\']+)["\']', html, re.IGNORECASE)
    cand = [h for h in hrefs if h.lower().startswith(("http://", "https://")) and not _SKIP_LINK.search(h)]
    if not cand:
        return "no CTA links found on the page"
    current = Counter(cand).most_common(1)[0][0]
    n = [0]

    def repl(m):
        if m.group(2) == current:
            n[0] += 1
            return m.group(1) + url + m.group(3)
        return m.group(0)

    html = re.sub(r'(<a\b[^>]*\bhref\s*=\s*["\'])([^"\']+)(["\'])', repl, html, flags=re.IGNORECASE)
    p.write_text(html, encoding="utf-8")
    return f"changed {n[0]} CTA link(s) to {url}"


def _edit_link(html, text, url):
    """Set the href (+ target=_blank) of every <a> whose visible text contains `text`."""
    url = changes._absolutize(url)
    n = 0

    def repl(m):
        nonlocal n
        opening, inner, closing = m.group(1), m.group(2), m.group(3)
        visible = re.sub(r'<[^>]+>', '', inner)
        if text.strip().lower() not in visible.lower():
            return m.group(0)
        if re.search(r'href\s*=', opening, re.IGNORECASE):
            opening = re.sub(r'href\s*=\s*(["\']).*?\1', f'href="{url}"', opening, count=1, flags=re.IGNORECASE)
        else:
            opening = re.sub(r'^<a\b', f'<a href="{url}"', opening, count=1, flags=re.IGNORECASE)
        if not re.search(r'\btarget\s*=', opening, re.IGNORECASE):
            opening = re.sub(r'^<a\b', '<a target="_blank"', opening, count=1, flags=re.IGNORECASE)
        n += 1
        return opening + inner + closing

    html = re.sub(r'(<a\b[^>]*>)(.*?)(</a>)', repl, html, flags=re.DOTALL | re.IGNORECASE)
    return html, n


_FONT_PATHS = [
    "/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf",
    "/usr/share/fonts/truetype/liberation/LiberationSans-Bold.ttf",
    "/usr/share/fonts/truetype/freefont/FreeSansBold.ttf",
]


def _find_logo_src(html):
    """The logo is an <img> mentioning 'logo', else the first <img> on the page (header)."""
    m = re.search(r'<img\b[^>]*\blogo\b[^>]*>', html, re.IGNORECASE) or re.search(r'<img\b[^>]*>', html, re.IGNORECASE)
    if not m:
        return None
    sm = re.search(r'src\s*=\s*["\']([^"\']+)["\']', m.group(0), re.IGNORECASE)
    return sm.group(1) if sm else None


def _set_logo(job_dir, image):
    """Point the header logo <img> at a specific file — an upload (assets-new/) or an existing
    asset (assets/images/ or any existing path under the job dir)."""
    name = image.replace("\\", "/").rsplit("/", 1)[-1]
    inbox = job_dir / "assets-new" / name
    in_assets = job_dir / "assets" / "images" / name
    if inbox.exists():
        in_assets.parent.mkdir(parents=True, exist_ok=True)
        shutil.copy2(str(inbox), str(in_assets))
        rel = f"assets/images/{name}"
    elif in_assets.exists():
        rel = f"assets/images/{name}"
    elif (job_dir / image.lstrip("/")).exists():
        rel = image.lstrip("/")
    else:
        return f"image '{image}' not found in assets-new/ or assets/images/"
    idx = job_dir / "index.html"
    html = idx.read_text(encoding="utf-8", errors="ignore")
    logo_src = _find_logo_src(html)
    if not logo_src:
        return "couldn't find a logo <img> on the page"
    idx.write_text(html.replace(logo_src, rel), encoding="utf-8")
    return f"set the logo image to {rel}"


def _make_logo(job_dir, text, color="#1a1a1a"):
    """Render `text` to a transparent PNG wordmark and point the page's logo <img> at it."""
    try:
        from PIL import Image, ImageDraw, ImageFont
    except Exception:
        return "the image library (Pillow) isn't installed on the server — run pip install pillow"
    idx = job_dir / "index.html"
    html = idx.read_text(encoding="utf-8", errors="ignore")
    logo_src = _find_logo_src(html)
    if not logo_src:
        return "couldn't find a logo <img> on the page"
    font_path = next((p for p in _FONT_PATHS if os.path.exists(p)), None)
    font = ImageFont.truetype(font_path, 72) if font_path else ImageFont.load_default()
    bbox = ImageDraw.Draw(Image.new("RGBA", (10, 10))).textbbox((0, 0), text, font=font)
    pad_x, pad_y = 30, 20
    img = Image.new("RGBA", (bbox[2] - bbox[0] + 2 * pad_x, bbox[3] - bbox[1] + 2 * pad_y), (0, 0, 0, 0))
    rgb = tuple(int(color.lstrip("#")[i:i + 2], 16) for i in (0, 2, 4))
    ImageDraw.Draw(img).text((pad_x - bbox[0], pad_y - bbox[1]), text, font=font, fill=rgb + (255,))
    slug = re.sub(r'[^\w\-]', '', text).lower() or "logo"
    out = job_dir / "assets" / "images" / f"{slug}-logo.png"
    out.parent.mkdir(parents=True, exist_ok=True)
    img.save(str(out))
    rel = f"assets/images/{out.name}"
    idx.write_text(html.replace(logo_src, rel), encoding="utf-8")
    return f"generated a '{text}' wordmark logo and set it as the page logo ({rel})"


def _set_hero_image(job_dir, new_file):
    """Swap the hero image — located via the RENDERED page (largest image at the top), not by
    guessing the first <img> in minified source — for an uploaded/existing file."""
    src = next((p for p in (job_dir / "assets-new" / new_file, job_dir / new_file) if p.exists()), None)
    if not src:
        return f"file '{new_file}' not found in assets-new/"
    hero = verify.hero_image(job_dir)
    if not hero or not hero.get("src"):
        return "couldn't locate a hero image on the rendered page"
    old_src = hero["src"]
    dest_dir = job_dir / "assets" / "images"
    dest_dir.mkdir(parents=True, exist_ok=True)
    shutil.copy2(str(src), str(dest_dir / src.name))
    rel = f"assets/images/{src.name}"
    idx = job_dir / "index.html"
    html = idx.read_text(encoding="utf-8", errors="ignore")
    if old_src not in html:
        return f"located the hero (src={old_src}) but couldn't find that src in index.html to replace"
    idx.write_text(html.replace(old_src, rel), encoding="utf-8")
    return f"set the hero image to {rel} (replaced {old_src}, {hero.get('width')}x{hero.get('height')}px at top)"


def _remove_element_by_id(html, tag, el_id):
    """Remove the balanced <tag ... id="el_id" ...>...</tag> element from html. Returns (html, ok).
    Scans open/close tags of `tag` from the opening tag, deleting through the matching close — so
    only that subtree is removed and the rest of the file is byte-for-byte untouched."""
    eid = re.escape(el_id)
    opening = re.search(
        rf'<{re.escape(tag)}\b[^>]*\bid\s*=\s*(?:"{eid}"|\'{eid}\'|{eid}(?=[\s>]))',
        html, re.IGNORECASE)
    if not opening:
        return html, False
    start = opening.start()
    depth, end = 0, None
    for m in re.finditer(rf'<{re.escape(tag)}\b|</{re.escape(tag)}\s*>', html[start:], re.IGNORECASE):
        if m.group().lower().startswith('</'):
            depth -= 1
            if depth == 0:
                gt = html.find('>', start + m.start())
                end = gt + 1 if gt != -1 else None
                break
        else:
            depth += 1
    if end is None:
        return html, False
    return html[:start] + html[end:], True


def _remove_section(job_dir, text):
    """Remove the on-page section containing `text` — located via the rendered DOM, deleted as a
    balanced element from source so nothing else changes."""
    loc = verify.locate_section(job_dir, text)
    if not loc:
        return f"couldn't find a section containing '{text}' on the page"
    if not loc.get("id"):
        return (f"found a <{loc.get('tag')}> block ('{loc.get('textPreview')}') but it has no id to "
                f"remove precisely — give more distinctive text from the exact section to remove")
    idx = job_dir / "index.html"
    html = idx.read_text(encoding="utf-8", errors="ignore")
    new_html, ok = _remove_element_by_id(html, loc["tag"], loc["id"])
    if not ok:
        return f"located the section (id={loc['id']}) but couldn't remove it cleanly from source"
    idx.write_text(new_html, encoding="utf-8")
    return (f"removed the <{loc['tag']}> section (id={loc['id']}, {loc['width']}x{loc['height']}px, "
            f"text '{loc['textPreview']}')")


def _add_script(job_dir, code, where="head"):
    """Insert a script/HTML snippet into <head> (default) or just before </body>."""
    idx = job_dir / "index.html"
    html = idx.read_text(encoding="utf-8", errors="ignore")
    snippet = code if re.search(r'<\s*script', code, re.IGNORECASE) else f"<script>\n{code}\n</script>"
    if where == "head" and re.search(r'</head\s*>', html, re.IGNORECASE):
        html = re.sub(r'</head\s*>', lambda m: snippet + "\n" + m.group(0), html, count=1, flags=re.IGNORECASE)
    elif re.search(r'</body\s*>', html, re.IGNORECASE):
        html = re.sub(r'</body\s*>', lambda m: snippet + "\n" + m.group(0), html, count=1, flags=re.IGNORECASE)
    else:
        html += snippet
    idx.write_text(html, encoding="utf-8")
    return f"inserted the script into <{'head' if where == 'head' else 'body end'}>"


def _set_favicon(job_dir, new_file):
    """Copy an uploaded image into assets/ and point every <link rel=...icon...> at it."""
    src = next((p for p in (job_dir / "assets-new" / new_file, job_dir / new_file) if p.exists()), None)
    if not src:
        return f"file '{new_file}' not found in assets-new/"
    dest_dir = job_dir / "assets" / "images"
    dest_dir.mkdir(parents=True, exist_ok=True)
    shutil.copy2(str(src), str(dest_dir / src.name))
    rel = f"assets/images/{src.name}"
    idx = job_dir / "index.html"
    html = idx.read_text(encoding="utf-8", errors="ignore")
    n = [0]

    def repl(m):
        tag = m.group(0)
        if not re.search(r'rel\s*=\s*["\'][^"\']*icon', tag, re.IGNORECASE):
            return tag
        n[0] += 1
        if re.search(r'href\s*=', tag, re.IGNORECASE):
            return re.sub(r'href\s*=\s*(["\']).*?\1', f'href="{rel}"', tag, count=1, flags=re.IGNORECASE)
        return re.sub(r'^<link\b', f'<link href="{rel}"', tag, count=1, flags=re.IGNORECASE)

    html = re.sub(r'<link\b[^>]*>', repl, html, flags=re.IGNORECASE)
    if n[0] == 0 and re.search(r'</head>', html, re.IGNORECASE):
        html = re.sub(r'</head>', f'<link rel="icon" href="{rel}"></head>', html, count=1, flags=re.IGNORECASE)
        n[0] = 1
    idx.write_text(html, encoding="utf-8")
    return f"set favicon to {rel} ({n[0]} link tag(s))"


_VIDEO_EXT = {".mp4", ".webm", ".ogg", ".ogv", ".mov", ".m4v"}


def _swap_media(job_dir, section_text, new_file):
    """Copy assets-new/<new_file> into assets/ (images/ or media/ by type) and point the first
    image/video AFTER the section heading at it. A video replacing an <img> is converted to a
    <video controls> (preserving id/class/style/width/height for layout). Heading match is
    punctuation/quote/entity tolerant (works on minified HTML)."""
    src = next((p for p in (job_dir / "assets-new" / new_file, job_dir / new_file) if p.exists()), None)
    if not src:
        return f"file '{new_file}' not found in assets-new/"
    is_video = src.suffix.lower() in _VIDEO_EXT
    folder = "media" if is_video else "images"
    dest_dir = job_dir / "assets" / folder
    dest_dir.mkdir(parents=True, exist_ok=True)
    shutil.copy2(str(src), str(dest_dir / src.name))
    rel = f"assets/{folder}/{src.name}"

    idx_path = job_dir / "index.html"
    html = idx_path.read_text(encoding="utf-8", errors="ignore")
    # Models sometimes HTML-encode the heading (e.g. "Heart" → &quot_0022;Heart&quot_0022;). After
    # unescape that leaves junk tokens like "_0022" (html.unescape matches &quot without the ';').
    # Visible heading text never contains underscores, and bare entity names aren't real words —
    # drop both so we match on the words a person actually reads.
    _ENT = {"quot", "amp", "lt", "gt", "nbsp", "apos", "ndash", "mdash", "rsquo", "lsquo", "rdquo", "ldquo"}
    words = [w for w in re.findall(r'\w+', _html_unescape(section_text))
             if "_" not in w and w.lower() not in _ENT]
    if not words:
        return "section_text was empty"
    # Allow inline HTML tags (e.g. <strong>, <span>) AND punctuation between the heading words —
    # LanderLab/GrapesJS pages wrap parts of a heading in tags, which a \W*-only matcher can't skip.
    heading = re.compile(r'(?:<[^>]+>|\W)*'.join(re.escape(w) for w in words), re.IGNORECASE)
    hm = heading.search(html)
    if not hm:
        return f"could not find the section heading '{section_text}'"
    after = html[hm.end():]
    em = re.search(r'<video\b[^>]*>.*?</video>|<img\b[^>]*>', after, re.IGNORECASE | re.DOTALL)
    if not em:
        return "no image or video found after that section"
    old_el = em.group(0)

    def attrs_of(s):
        out = []
        for a in ("id", "class", "style", "width", "height"):
            m = re.search(rf'\b{a}\s*=\s*(["\'])(.*?)\1', s, re.IGNORECASE)
            if m:
                out.append(f'{a}="{m.group(2)}"')
        return " ".join(out)

    if is_video:
        new_el = f'<video {attrs_of(old_el)} src="{rel}" controls playsinline muted></video>'
    elif old_el[:6].lower() == "<video":
        new_el = f'<img {attrs_of(old_el)} src="{rel}">'
    else:
        new_el = re.sub(r'(src\s*=\s*)(["\']).*?\2', lambda mm: mm.group(1) + f'"{rel}"',
                        old_el, count=1, flags=re.IGNORECASE)

    html = html[:hm.end()] + after.replace(old_el, new_el, 1)
    idx_path.write_text(html, encoding="utf-8")
    kind = "video" if is_video else "image"
    return f"swapped the media under '{section_text}' for a {kind} ({rel})"


_SYSTEM = (
    "You apply ONE change request to an already-cloned landing page in the current job folder "
    "(index.html plus an assets/ folder). index.html is MINIFIED (mostly one long line), so do NOT "
    "rely on line-based grep — prefer the high-level tools below. Make the change, then call "
    "done(summary). Rules:\n"
    "- ALWAYS call a tool; never answer with plain prose.\n"
    "- 'change the X link to <url>'  →  edit_link(text='X', url='<url>').\n"
    "- 'change wording A to B'        →  replace_text(old='A', new='B').\n"
    "- 'change all CTA links to <url>' →  set_cta(url='<url>').\n"
    "- 'change the logo to <image file>' (a FILENAME, e.g. aeriocool-logo.png) →  set_logo(image='<file>').\n"
    "- 'make the logo say <words>' (TEXT to render)  →  make_logo(text='<words>').\n"
    "- 'change the favicon to this'  →  set_favicon(new_file='<uploaded filename>').\n"
    "- 'add this script/snippet to the page' →  add_script(code='<...>', where='head'|'body_end').\n"
    "Uploaded files live in assets-new/ (an INBOX) — NEVER reference assets-new/ in the page. The "
    "swap_media/set_favicon/make_logo tools copy uploads into assets/ for you; if you ever place an "
    "upload by hand, first copy it into assets/images/ and reference assets/images/<name>.\n"
    "- 'change the image/video under <heading> to this'  →  swap_media(section_text='<heading>', "
    "new_file='<uploaded filename>').\n"
    "- 'change the HERO image to this' (the big top image, no heading above it)  →  "
    "set_hero_image(new_file='<uploaded filename>'). Do NOT use swap_media for the hero.\n"
    "- 'remove/delete the section about <X>'  →  remove_section(text='<distinctive text in that section>').\n"
    "- Anything else                  →  read_file / bash / replace_in_file (use `grep -o` to "
    "extract from the minified line, since line numbers are useless here).\n"
    "- Make the SMALLEST edit that satisfies the request; change nothing else. Preserve layout, "
    "CSS, classes, and JS behaviour. Then call done().\n"
    "- If the request does NOT mention the favicon, logo, or an image, do not touch them. Your "
    "done() summary must describe ONLY what this request changed.\n"
    "- You edit the cloned page's OWN files — index.html and its assets/, INCLUDING the page's own "
    "JavaScript (e.g. assets/js/*.js). Those edits are served live immediately; there is NO deploy "
    "or git 'push' step, EVER — writing the file IS the change. Never run git/sudo or try to push.\n"
    "- NEVER delete files or use shell to modify them (rm/mv/sed -i/tee/redirects). Make changes "
    "through the tools (replace_in_file for free-form edits); use bash ONLY to INSPECT "
    "(grep/find/sed -n/cat). You cannot modify the bot's own source — if asked, say so in done().\n"
    "- Your done() summary must be TRUTHFUL: describe only edits you actually made and verified. If "
    "you could not do part of the request, say which part — do not claim success you didn't achieve."
)


# bash is for INSPECTING the page (grep/find/sed -n/cat) only. Block anything that deletes, moves,
# or rewrites files, plus git/sudo/deploy — edits must go through the tools, and the agent must
# never be able to destroy a clone or push code (a run once `rm`'d index.html with no way back).
_BASH_DENY = re.compile(
    r'(?:^|[\s;&|(`])(?:rm|rmdir|mv|dd|mkfs\w*|shutdown|reboot|sudo|git|chmod|chown|kill|killall|truncate|ln|tee)\b'
    r'|\bsed\b[^|;&]*\s-i'              # in-place sed edits
    r'|>\s*[^\s|&;]*\.html',           # redirection that would clobber an html file
    re.IGNORECASE)


def _exec(name, args, job_dir):
    try:
        if name == "bash":
            cmd = args.get("cmd", "")
            if _BASH_DENY.search(cmd):
                return ("blocked: that command can delete/modify files or run git/sudo. Use bash "
                        "ONLY to inspect (grep/find/sed -n/cat); make changes with replace_in_file/"
                        "replace_text/swap_media. You cannot delete index.html, run git, or deploy.")
            r = subprocess.run(cmd, shell=True, cwd=str(job_dir),
                               capture_output=True, text=True, timeout=120)
            return (r.stdout + r.stderr).strip()[:4000] or "(no output)"
        if name == "read_file":
            return (job_dir / args["path"]).read_text(encoding="utf-8", errors="ignore")[:60000]
        if name == "replace_in_file":
            p = job_dir / args["path"]
            t = p.read_text(encoding="utf-8", errors="ignore")
            cnt = t.count(args["old"])
            p.write_text(t.replace(args["old"], args["new"]), encoding="utf-8")
            return f"replaced {cnt} occurrence(s) in {args['path']}"
        if name == "edit_link":
            p = job_dir / "index.html"
            html = p.read_text(encoding="utf-8", errors="ignore")
            html, c = _edit_link(html, args["text"], args["url"])
            p.write_text(html, encoding="utf-8")
            return f"set href on {c} link(s) whose text contains '{args['text']}'" if c \
                else f"no <a> found whose visible text contains '{args['text']}'"
        if name == "replace_text":
            p = job_dir / "index.html"
            html = p.read_text(encoding="utf-8", errors="ignore")
            html, c = changes._replace_textual(html, args["old"], args["new"])
            p.write_text(html, encoding="utf-8")
            return f"replaced {c} occurrence(s) of visible text '{args['old']}'"
        if name == "set_cta":
            return _set_cta(job_dir, args["url"])
        if name == "set_logo":
            return _set_logo(job_dir, args["image"])
        if name == "make_logo":
            return _make_logo(job_dir, args["text"])
        if name == "add_script":
            return _add_script(job_dir, args["code"], args.get("where", "head"))
        if name == "set_favicon":
            return _set_favicon(job_dir, args["new_file"])
        if name == "remove_section":
            return _remove_section(job_dir, args["text"])
        if name == "set_hero_image":
            return _set_hero_image(job_dir, args["new_file"])
        if name == "swap_media":
            return _swap_media(job_dir, args["section_text"], args["new_file"])
        return f"unknown tool {name}"
    except Exception as e:
        return f"error: {e}"


def _baseline(job_dir, on_log):
    """Render the page BEFORE any edit so the gate can tell pre-existing junk from regressions.
    Returns the render() dict, or None if the verifier is unavailable (then we skip gating)."""
    try:
        r = verify.render(job_dir)
        if on_log:
            on_log("🔎 baseline rendered")
        return r
    except Exception as e:
        if on_log:
            on_log(f"⚠️ verifier unavailable — gate disabled ({str(e)[:80]})")
        return None


def _gate(job_dir, baseline, on_log):
    """Differential check after an edit. Returns (ok, regressions[]). Verifier failure never
    blocks the change — we degrade to 'ok' rather than dead-ending on our own tooling."""
    if baseline is None:
        return True, []
    try:
        r = verify.gate(job_dir, baseline=baseline)
    except Exception as e:
        if on_log:
            on_log(f"⚠️ verify skipped: {str(e)[:80]}")
        return True, []
    regs = r.get("regressions", [])
    if on_log:
        on_log("🔎 verify: " + ("clean" if r["ok"] else "regressions → " + "; ".join(regs[:3])))
    return r["ok"], regs


def _change_digest(before, after):
    """Ground-truth summary of what actually changed in index.html, for the intent reviewer — so it
    checks reality, not the coder's self-report. Reports added/removed src|href refs (image/link/
    media swaps are the common verifiable edits); flags 'nothing changed' outright. This is what
    catches 'claimed 3 swaps but only did 1'."""
    from collections import Counter

    def refs(b):
        return re.findall(r'(?:src|href)\s*=\s*["\']([^"\']+)["\']',
                          b.decode("utf-8", "ignore"), re.IGNORECASE)

    cb, ca = Counter(refs(before)), Counter(refs(after))
    added = sorted(set((ca - cb).elements()))
    removed = sorted(set((cb - ca).elements()))
    parts = []
    if added:
        parts.append("added refs: " + ", ".join(added[:12]))
    if removed:
        parts.append("removed refs: " + ", ".join(removed[:12]))
    if not parts:
        if before == after:
            return "NOTHING changed in index.html."
        bt, at = before.decode("utf-8", "ignore"), after.decode("utf-8", "ignore")
        return f"text/markup changed (~{abs(len(at) - len(bt))} char delta), no src/href refs changed"
    return "; ".join(parts)


def _snapshot(job_dir):
    """Back up index.html (+ any root-level .html) before edits so a destructive run can be undone.
    The agent edits index.html almost exclusively — this is the file that must never be lost."""
    snap = {}
    for p in job_dir.glob("*.html"):
        try:
            snap[p.name] = p.read_bytes()
        except Exception:
            pass
    return snap


def _restore_if_lost(job_dir, snap, on_log):
    """If a backed-up html file was deleted or emptied during the run, put the original back."""
    restored = []
    for name, data in snap.items():
        p = job_dir / name
        if not p.exists() or p.stat().st_size == 0:
            p.write_bytes(data)
            restored.append(name)
    if restored and on_log:
        on_log("♻️ restored " + ", ".join(restored) + " from pre-edit backup")
    return restored


def save_version(job_dir):
    """Save the current index.html into .history/ before a change, so the user can `undo`. Keeps the
    last 15 versions; skips if identical to the latest (no churn from no-op runs)."""
    idx = Path(job_dir) / "index.html"
    if not idx.exists():
        return
    hd = Path(job_dir) / ".history"
    hd.mkdir(exist_ok=True)
    cur = idx.read_bytes()
    versions = sorted(hd.glob("index.*.html"))
    if versions and versions[-1].read_bytes() == cur:
        return
    import time
    (hd / f"index.{int(time.time() * 1000)}.html").write_bytes(cur)
    for old in sorted(hd.glob("index.*.html"))[:-15]:
        old.unlink()


def undo(job_dir):
    """Restore the previous saved version of index.html and pop it, so repeated `undo` steps
    further back. Returns a human-facing message."""
    hd = Path(job_dir) / ".history"
    versions = sorted(hd.glob("index.*.html")) if hd.exists() else []
    if not versions:
        return "Nothing to undo — no earlier version is saved for this page yet."
    last = versions[-1]
    (Path(job_dir) / "index.html").write_bytes(last.read_bytes())
    last.unlink()
    return "↩️ Reverted the page to the version before the last change."


def run(job_dir, request, on_log=None, max_iters=25, max_repair=2, uploads=None):
    """Apply a free-form change request, with a safety net: index.html is versioned before the run
    (for user `undo`) and auto-restored if the agent somehow destroys it, so a clone is never lost."""
    job_dir = Path(job_dir)
    save_version(job_dir)   # user-facing undo history
    snap = _snapshot(job_dir)
    result = _run_inner(job_dir, request, on_log, max_iters, max_repair, uploads)
    if _restore_if_lost(job_dir, snap, on_log):
        return ("⚠️ That request led to an unsafe edit, so I reverted the page to its original state "
                "to avoid breaking it. Nothing was changed — please re-try with a more specific request.")
    return result


def _run_inner(job_dir, request, on_log=None, max_iters=25, max_repair=2, uploads=None):
    """Apply a free-form change request to the clone in job_dir. Returns a summary string.

    After the model calls done(), the change is gated against a pre-edit baseline render. If the
    edit introduced a regression (broken local asset / uncaught JS), the model gets the failure
    fed back and a bounded number of repair passes (max_repair). If it still can't clear the gate,
    we return the summary WITH an explicit unresolved-issues note — never a silent dead-end."""
    job_dir = Path(job_dir)
    client = OpenAI(base_url=LLM_BASE_URL, api_key=LLM_API_KEY)
    try:
        before_html = (job_dir / "index.html").read_bytes()  # ground truth for the intent diff
    except Exception:
        before_html = b""

    # Pin the file(s) attached in THIS message — assets-new/ accumulates older uploads, so letting
    # the model pick from the whole inbox makes it grab a stale file (it once used cristiano-1.jpg
    # when cristiano-2.webp was just attached). The bot knows the current attachment; trust it.
    if uploads:
        which = "this file" if len(uploads) == 1 else "these files"
        request = (f"{request}\n\n(The user just attached {', '.join(uploads)} — use {which} "
                   f"for this request, not any older file.)")
    else:
        # No attachment this turn: only surface the inbox when the request refers to one — otherwise
        # its mere presence biases the model into "updating the favicon/logo" on unrelated requests.
        new_assets = sorted(p.name for p in (job_dir / "assets-new").glob("*") if p.is_file())
        if new_assets and re.search(
            r'\b(this|these|that|uploaded|upload|attached|provided|image|images|photo|picture|logo|'
            r'favicon|video|icon|file|pic)\b', request, re.IGNORECASE):
            request = (f"{request}\n\n(Reference only — uploaded files in assets-new/: "
                       f"{', '.join(new_assets)}. Use one ONLY if this request asks for it.)")

    messages = [{"role": "system", "content": _SYSTEM},
                {"role": "user", "content": request}]
    # Planner pre-pass: a senior-engineer plan to guide the coder (degrades to "" if unavailable).
    if USE_PLANNER:
        plan_text = planner.plan(request, on_log)
        if plan_text:
            messages.append({"role": "user",
                             "content": "A senior engineer suggests this approach:\n" + plan_text +
                             "\nFollow it, adapting as needed. Always call a tool."})
    summary = "(agent finished without calling done)"
    no_tool = 0
    baseline = _baseline(job_dir, on_log)
    repair_budget = max_repair
    intent_budget = 1  # one planner-driven redo when the result doesn't match intent

    for _ in range(max_iters):
        resp = client.chat.completions.create(
            model=LLM_MODEL, messages=messages, tools=_TOOLS,
            tool_choice="auto", temperature=0, max_tokens=1500,
        )
        msg = resp.choices[0].message
        tcs = msg.tool_calls or []

        asst = {"role": "assistant", "content": msg.content or ""}
        if tcs:
            asst["tool_calls"] = [
                {"id": tc.id, "type": "function",
                 "function": {"name": tc.function.name, "arguments": tc.function.arguments}}
                for tc in tcs
            ]
        messages.append(asst)

        if not tcs:
            no_tool += 1
            if no_tool >= 3:
                return (msg.content or summary)[:500]
            messages.append({"role": "user",
                             "content": "Call a tool to make the change, or done() when finished."})
            continue
        no_tool = 0

        verdict_return = None  # set when done() clears the gate (or we're out of repair budget)
        for tc in tcs:
            try:
                args = json.loads(tc.function.arguments or "{}")
            except Exception:
                args = {}
            if tc.function.name == "done":
                summary = args.get("summary", "done")
                ok, regs = _gate(job_dir, baseline, on_log)
                # 1. Breakage gate first — a regressed page gets repaired before anything else.
                if not ok and repair_budget > 0:
                    repair_budget -= 1
                    messages.append({"role": "tool", "tool_call_id": tc.id,
                                     "content": "Your change introduced these problems (they were NOT on the "
                                     "page before your edit):\n- " + "\n- ".join(regs[:6]) +
                                     "\nFix ONLY these, then call done() again."})
                    continue
                # 2. Page is intact (or out of repair budget) — now the planner reviews INTENT:
                #    did the coder change the right thing? Catches wrong-section/wrong-target edits.
                if ok and USE_PLANNER and intent_budget > 0:
                    try:
                        after_html = (job_dir / "index.html").read_bytes()
                    except Exception:
                        after_html = b""
                    digest = _change_digest(before_html, after_html)
                    if on_log:
                        on_log("🔬 changes: " + digest[:160])
                    intent_ok, feedback = planner.verify_intent(request, summary, digest, on_log)
                    if not intent_ok:
                        intent_budget -= 1
                        messages.append({"role": "tool", "tool_call_id": tc.id,
                                         "content": "A reviewer checked your work against the request and it does "
                                         "NOT match: " + " ".join(feedback.split())[:300] +
                                         "\nRedo it so it satisfies the original request, then call done()."})
                        continue
                note = "" if ok else "  ⚠️ unresolved: " + "; ".join(regs[:3])
                messages.append({"role": "tool", "tool_call_id": tc.id,
                                 "content": "verified clean" if ok else "kept despite issues: " + "; ".join(regs[:3])})
                verdict_return = summary + note
                continue
            result = _exec(tc.function.name, args, job_dir)
            if on_log:
                on_log(f"🔧 {tc.function.name}({str(args)[:80]}) → {str(result)[:140]}")
            messages.append({"role": "tool", "tool_call_id": tc.id, "content": str(result)[:8000]})

        if verdict_return is not None:
            if on_log:
                on_log(f"✅ {verdict_return}")
            return verdict_return

    return summary
