#!/usr/bin/env python3
import json, re, html, os, sys, time, urllib.request, urllib.parse

UA = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0"}
BASE = "https://dobrotniy-dom.ru/product/serijnye-karkasnye-doma/"
SLUGS = ["troitskiy", "voskresenskiy", "voznesenskiy", "znamenskiy", "aleksandriya",
         "aleksandriya-1", "aleksandriya-2", "aleksandriya-barn", "sergievskiy", "radonezhskiy"]

token = None
for line in open("/root/site/yandex/.env", encoding="utf-8"):
    if line.startswith("YANDEX_ACCESS_TOKEN="):
        token = line.split("=", 1)[1].strip()
H = {"Authorization": f"OAuth {token}"}

WORK = "/root/site/video/site_renders"
os.makedirs(WORK, exist_ok=True)

def fetch(url, headers=None, binary=False):
    h = dict(UA)
    if headers: h.update(headers)
    req = urllib.request.Request(url, headers=h)
    with urllib.request.urlopen(req, timeout=90) as r:
        data = r.read()
    return data if binary else data.decode("utf-8", errors="replace")

def disk_mkdir(path):
    url = "https://cloud-api.yandex.net/v1/disk/resources?path=" + urllib.parse.quote(path)
    req = urllib.request.Request(url, method="PUT", headers=H)
    try:
        with urllib.request.urlopen(req, timeout=60) as r:
            return True
    except urllib.error.HTTPError as e:
        return "already" in e.read().decode(errors="replace")

def disk_upload(path, local_file):
    url = ("https://cloud-api.yandex.net/v1/disk/resources/upload?path=" +
           urllib.parse.quote(path) + "&overwrite=true")
    req = urllib.request.Request(url, headers=H)
    with urllib.request.urlopen(req, timeout=60) as r:
        href = json.loads(r.read().decode())["href"]
    with open(local_file, "rb") as f:
        data = f.read()
    req = urllib.request.Request(href, data=data, method="PUT")
    with urllib.request.urlopen(req, timeout=180) as r:
        return r.status

DST_ROOT = "/Рендеры с сайта"
disk_mkdir(DST_ROOT)

summary = {}
for slug in SLUGS:
    try:
        page = fetch(BASE + slug + "/")
    except Exception as e:
        print("СЕТЬ FAIL", slug, e); continue

    h1m = re.search(r'<h1[^>]*>(.*?)</h1>', page, flags=re.S)
    name = re.sub(r'<[^>]+>', '', h1m.group(1)).strip() if h1m else slug
    name = html.unescape(name)
    print(f"=== {slug} -> {name}")

    pos_plans = page.find("Планировки")
    if pos_plans == -1:
        pos_plans = len(page)

    imgs = {}
    for m in re.finditer(r'/upload/iblock/[a-z0-9]+/[a-z0-9]+\.(?:png|jpg|jpeg|webp)', page):
        url = "https://dobrotniy-dom.ru" + m.group(0)
        grp = "Планировки" if m.start() > pos_plans else "Внешний вид"
        if url not in imgs:
            imgs[url] = grp

    folder = os.path.join(WORK, name)
    os.makedirs(folder, exist_ok=True)
    counts = {"Внешний вид": 0, "Планировки": 0}
    for i, (url, grp) in enumerate(imgs.items(), 1):
        ext = url.rsplit(".", 1)[-1]
        sub = os.path.join(folder, grp)
        os.makedirs(sub, exist_ok=True)
        fname = f"{i:02d}.{ext}"
        local = os.path.join(sub, fname)
        if not os.path.exists(local):
            try:
                data = fetch(url, binary=True)
                open(local, "wb").write(data)
            except Exception as e:
                print("  DOWNLOAD FAIL", url, e); continue
        counts[grp] += 1

        # upload to disk
        disk_mkdir(DST_ROOT + "/" + name)
        disk_mkdir(DST_ROOT + "/" + name + "/" + grp)
        try:
            disk_upload(DST_ROOT + "/" + name + "/" + grp + "/" + fname, local)
        except Exception as e:
            print("  UPLOAD FAIL", fname, e)
        time.sleep(0.2)

    summary[name] = counts
    print(f"  готово: {counts}")

with open("/root/site/video/site_summary.json", "w", encoding="utf-8") as f:
    json.dump(summary, f, ensure_ascii=False, indent=1)
print("DONE. Проектов:", len(summary), "| всего файлов:", sum(sum(c.values()) for c in summary.values()))
