#!/usr/bin/env python3
"""Prawdziwe 1:1: curl zapisuje surowy body + nagłówki. Zero rewrite."""
import hashlib, os, re, subprocess, sys, time
from collections import deque
from html.parser import HTMLParser
from urllib.parse import urljoin, urlparse, unquote, quote

START = sys.argv[1] if len(sys.argv) > 1 else sys.exit("użycie: curl-1to1.py URL [katalog]")
OUT = os.path.abspath(sys.argv[2] if len(sys.argv) > 2 else "./klon")
HOST = urlparse(START).netloc
MAX, DELAY = 8000, 0.35
UA = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36"

class Href(HTMLParser):
    def __init__(self, base):
        super().__init__(); self.base, self.found = base, set()
    def handle_starttag(self, tag, attrs):
        a = dict(attrs)
        for k in ("href", "src", "poster", "data-src", "data-href", "action"):
            if a.get(k): self.found.add(urljoin(self.base, a[k]))
        if a.get("srcset"):
            for part in a["srcset"].split(","):
                u = part.strip().split(" ")[0]
                if u: self.found.add(urljoin(self.base, u))

def same(u):
    p = urlparse(u)
    return p.scheme in ("http", "https") and p.netloc == HOST

def rel_of(u):
    p = urlparse(u)
    path = unquote(p.path) or "/"
    if path.endswith("/"): path += "index.html"
    if p.query: path += "@" + quote(p.query, safe="")
    return os.path.join(p.netloc, path.lstrip("/"))

def curl_1to1(url, body, hdr):
    os.makedirs(os.path.dirname(body), exist_ok=True)
    os.makedirs(os.path.dirname(hdr), exist_ok=True)
    r = subprocess.run([
        "curl", "-sS", "-g", "--path-as-is", "--max-redirs", "0",
        "-H", "Accept-Encoding: identity",
        "-A", UA, "-D", hdr, "-o", body, "-R",
        "-w", "%{http_code}\t%{content_type}\t%{size_download}",
        "--connect-timeout", "20", "--max-time", "120",
        url,
    ], capture_output=True, text=True)
    parts = (r.stdout or "").strip().split("\t")
    code = parts[0] if parts else "000"
    ctype = parts[1] if len(parts) > 1 else ""
    size = parts[2] if len(parts) > 2 else "0"
    return r.returncode == 0, code, ctype, size

def peek_text(path):
    raw = open(path, "rb").read()
    if raw.startswith(b"\x1f\x8b"):
        import gzip
        try: raw = gzip.decompress(raw)
        except Exception: return None, raw
    if b"\0" in raw[:2048]:
        return None, raw
    return raw.decode("utf-8", "ignore"), raw

def links_from(text, base, css=False):
    out = set()
    if not text: return out
    if css:
        for m in re.findall(r"url\(\s*['\"]?([^'\")\s]+)['\"]?\s*\)", text):
            if not m.startswith("data:"): out.add(urljoin(base, m))
        for m in re.findall(r"@import\s+['\"]([^'\"]+)['\"]", text):
            out.add(urljoin(base, m))
        return out
    p = Href(base)
    try: p.feed(text)
    except Exception: pass
    out |= p.found
    out |= {urljoin(base, m) for m in re.findall(r"<loc>\s*([^<\s]+)\s*</loc>", text, re.I)}
    return out

def loc_header(hdr_path):
    try: raw = open(hdr_path, "rb").read().decode("iso-8859-1", "ignore")
    except Exception: return None
    m = re.search(r"^Location:\s*(\S+)", raw, re.I | re.M)
    return m.group(1).strip() if m else None

os.makedirs(OUT, exist_ok=True)
man = open(os.path.join(OUT, "manifest.tsv"), "w")
man.write("url\tstatus\tctype\tbytes\tsha256\tbody\theaders\n")

q, seen, n = deque([START]), set(), 0
for extra in (
    f"{urlparse(START).scheme}://{HOST}/",
    f"{urlparse(START).scheme}://{HOST}/robots.txt",
    f"{urlparse(START).scheme}://{HOST}/sitemap.xml",
    f"{urlparse(START).scheme}://{HOST}/sitemap_index.xml",
):
    q.append(extra)

while q and n < MAX:
    url = q.popleft().split("#", 1)[0]
    if not url or url in seen or not same(url):
        continue
    seen.add(url)
    rel = rel_of(url)
    body = os.path.join(OUT, "files", rel)
    hdr  = os.path.join(OUT, "headers", rel + ".headers")
    ok, code, ctype, size = curl_1to1(url, body, hdr)
    n += 1
    sha = ""
    if ok and os.path.isfile(body):
        h = hashlib.sha256()
        with open(body, "rb") as f:
            for chunk in iter(lambda: f.read(1 << 20), b""):
                h.update(chunk)
        sha = h.hexdigest()
    print(f"[{n}] {code} {url}")
    man.write(f"{url}\t{code}\t{ctype}\t{size}\t{sha}\t{body}\t{hdr}\n")
    man.flush()

    loc = loc_header(hdr) if os.path.isfile(hdr) else None
    if loc:
        loc = urljoin(url, loc).split("#", 1)[0]
        if same(loc) and loc not in seen:
            q.append(loc)

    if ok and os.path.isfile(body) and code.startswith(("2", "3")):
        text, _ = peek_text(body)
        css = "css" in (ctype or "") or body.endswith(".css")
        html = "html" in (ctype or "") or body.endswith((".html", ".htm"))
        xml = "xml" in (ctype or "") or body.endswith(".xml")
        if text and (html or css or xml or body.endswith("robots.txt")):
            if body.endswith("robots.txt"):
                for m in re.findall(r"(?i)^sitemap:\s*(\S+)", text, re.M):
                    q.append(m.strip())
            for u in links_from(text, url, css=css):
                u = u.split("#", 1)[0]
                if same(u) and u not in seen:
                    q.append(u)
    time.sleep(DELAY)

man.close()
print(f"\n1:1 gotowe: {n} odpowiedzi → {OUT}/files  manifest: {OUT}/manifest.tsv")

