"""
Снимок сайта на Tilda для импорта: страницы из sitemap.xml, закрытые в robots.txt и все найденные внутренние ссылки.
Результат: storage/app/tilda-mirror/pages/*.html, crawl.json (статусы/редиректы), sitemap.xml, robots.txt, custom.css.

    pip install requests beautifulsoup4 lxml
    python tools/crawl.py            # докачать недостающее
    python tools/crawl.py --fresh    # полный свежий снимок

Сайт за DDoS-Guard: между запросами пауза 0.6 с; при ответах 403 подождите и запустите снова —
уже скачанные страницы повторно не запрашиваются.
"""
import requests, time, json, re, os, sys
from urllib.parse import urljoin, urlparse, urlunparse
from bs4 import BeautifulSoup

BASE = 'https://korzilla.ru'
ROOT = os.path.join(os.path.dirname(os.path.abspath(__file__)), '..', 'storage', 'app', 'tilda-mirror')
OUT = os.path.join(ROOT, 'pages')
UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0 Safari/537.36'
os.makedirs(OUT, exist_ok=True)
s = requests.Session()
s.headers.update({'User-Agent': UA, 'Accept-Language': 'ru-RU,ru;q=0.9'})

def norm(path):
    p = urlparse(path)
    path = p.path or '/'
    if path != '/' and path.endswith('/'):
        path = path.rstrip('/')
    return path

def fname(path):
    if path == '/': return 'index.html'
    return path.strip('/').replace('/', '__') + ('' if path.endswith('.html') else '.html')

queue = []
seen = set()
def add(path, src):
    path = norm(path)
    if path in seen: return
    seen.add(path); queue.append((path, src))

for u in [re.sub(r'^https?://[^/]+', '', u) for u in re.findall(r'<loc>([^<]+)</loc>', s.get(BASE + '/sitemap.xml', timeout=60).text)]:
    add(u, 'sitemap')
robots = s.get(BASE + '/robots.txt', timeout=30).text
for m in re.findall(r'Disallow:\s*(\S+)', robots):
    if m.startswith('/tilda/') or '*' in m: continue
    add(m, 'robots')

meta = {}
# --fresh: скачать все страницы заново (иначе уже скачанные пропускаются)
if os.path.exists(os.path.join(ROOT, 'crawl.json')) and '--fresh' not in sys.argv:
    meta = json.load(open(os.path.join(ROOT, 'crawl.json'), encoding='utf-8'))
assets_links = set()
i = 0
while i < len(queue):
    path, src = queue[i]; i += 1
    fn = fname(path)
    rec = meta.get(path)
    if rec and rec.get('status') == 200 and os.path.exists(os.path.join(OUT, fn)):
        html = open(os.path.join(OUT, fn), encoding='utf-8').read()
    else:
        try:
            r = s.get(BASE + path, timeout=60, allow_redirects=False)
        except Exception as e:
            meta[path] = {'status': 'ERR', 'err': str(e), 'src': src}; continue
        rec = {'status': r.status_code, 'src': src, 'location': r.headers.get('Location'),
               'last_modified': r.headers.get('last-modified'), 'len': len(r.content)}
        if r.status_code in (301, 302, 307, 308):
            loc = urljoin(BASE + path, r.headers.get('Location', ''))
            lp = urlparse(loc)
            if lp.netloc in ('korzilla.ru', 'www.korzilla.ru'):
                add(lp.path, 'redirect:' + path)
            meta[path] = rec
            print(r.status_code, path, '->', loc, flush=True)
            time.sleep(0.4); continue
        html = r.content.decode('utf-8', errors='replace')
        if r.status_code == 200:
            open(os.path.join(OUT, fn), 'w', encoding='utf-8').write(html)
        rec['file'] = fn
        meta[path] = rec
        print(r.status_code, path, len(r.content), flush=True)
        time.sleep(0.6)
    if rec.get('status') != 200: continue
    soup = BeautifulSoup(html, 'lxml')
    for a in soup.find_all('a', href=True):
        h = a['href'].strip()
        if not h or h.startswith(('#', 'tel:', 'mailto:', 'javascript:', 'whatsapp:', 'viber:', 'tg:', 'skype:')): continue
        full = urljoin(BASE + path, h)
        p = urlparse(full)
        if p.netloc not in ('korzilla.ru', 'www.korzilla.ru'): continue
        if p.path.startswith('/tilda/'): continue
        if re.search(r'\.(pdf|jpe?g|png|gif|webp|svg|zip|rar|docx?|xlsx?|pptx?|mp4|txt|xml)$', p.path, re.I):
            assets_links.add(p.path); continue
        add(p.path, 'link:' + path)
    if i % 20 == 0:
        json.dump(meta, open(os.path.join(ROOT, 'crawl.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)

json.dump(meta, open(os.path.join(ROOT, 'crawl.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)
json.dump(sorted(assets_links), open(os.path.join(ROOT, 'asset_links.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)
print('DONE', len(meta), 'paths;', sum(1 for v in meta.values() if v.get('status') == 200), 'ok')

# Служебные файлы сайта рядом с зеркалом
for name in ('sitemap.xml', 'robots.txt', 'custom.css'):
    r = s.get(BASE + '/' + name, timeout=60)
    if r.status_code == 200:
        open(os.path.join(ROOT, name), 'wb').write(r.content)
print('Готово:', ROOT)
