#!/usr/bin/env bash
# użycie: ./curlclone.sh https://example.com [katalog_wyjściowy]
set -u
START="$1"
OUT="${2:-klon}"
HOST=$(sed -E 's#^https?://##; s#[/:?].*##' <<<"$START")
UA="Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120 Safari/537.36"
COOKIES="${COOKIES:-}"
ASSET_RE='\.(css|js|mjs|json|png|jpe?g|gif|svg|webp|avif|ico|woff2?|ttf|otf|eot|mp4|webm|mp3|pdf|xml|txt|map)$'
QUEUE=$(mktemp)
SEEN=$(mktemp)
trap 'rm -f "$QUEUE" "$SEEN"' EXIT
echo "$START" >"$QUEUE"

# URL -> ścieżka pliku 1:1 (query i fragment obcinane)
url2path() {
  local u="${1%%#*}"
  u="${u%%\?*}"
  u="${u#http://}"
  u="${u#https://}"
  [[ "$u" != */* ]] && u="$u/"
  [[ "$u" == */ ]] && u="${u}index.html"
  local base="${u##*/}"
  # tylko jeśli nie ma rozszerzenia i nie kończy się już na index.html
  if [[ "$base" != *.* && "$base" != "index.html" ]]; then
    u="${u%/}/index.html"
  fi
  printf '%s/%s' "$OUT" "$u"
}

# rozwiązywanie linków względnych
resolve() {
  python3 -c '
import sys
from urllib.parse import urljoin, urldefrag
base = sys.argv[1]
for l in sys.stdin:
    l = l.strip().strip("\"'\''")
    if not l or l.startswith(("data:", "mailto:", "tel:", "javascript:", "#", "blob:")):
        continue
    print(urldefrag(urljoin(base, l))[0])
' "$1"
}

# wyciąganie linków z HTML / CSS
extract() {
  local f="$1" ct="$2"
  case "$ct" in
    text/html*)
      # typowe atrybuty z URL-ami (bez content= bo meta to śmieci)
      grep -oiE '(href|src|data-src|data-lazy-src|poster)=["'"'"'][^"'"'"']+' "$f" 2>/dev/null \
        | sed -E 's/^[^=]+=["'"'"']//'
      # srcset
      grep -oiE '(srcset|data-srcset)=["'"'"'][^"'"'"']+' "$f" 2>/dev/null \
        | sed -E 's/^[^=]+=["'"'"']//' \
        | tr ',' '\n' \
        | awk '{print $1}'
      # url() w inline style
      grep -oE 'url\([^)]+\)' "$f" 2>/dev/null \
        | sed -E 's/^url\(//; s/\)$//'
      ;;
    text/css*)
      grep -oE 'url\([^)]+\)' "$f" 2>/dev/null \
        | sed -E 's/^url\(//; s/\)$//'
      grep -oE '@import\s+["'"'"'][^"'"'"']+' "$f" 2>/dev/null \
        | sed -E 's/@import\s+["'"'"']//'
      ;;
  esac
}

while [[ -s "$QUEUE" ]]; do
  NEXT=$(mktemp)
  sort -u "$QUEUE" | while read -r url; do
    grep -qxF "$url" "$SEEN" && continue
    echo "$url" >>"$SEEN"

    file=$(url2path "$url")
    mkdir -p "$(dirname "$file")"

    # -f = fail na HTTP errorach, -w zwraca kod + content-type
    http_ct=$(curl -sSL --fail --compressed --remote-time --retry 3 --max-time 60 \
              -A "$UA" ${COOKIES:+-b "$COOKIES"} \
              -w '%{http_code} %{content_type}' -o "$file" "$url" 2>/dev/null) || {
      echo "ERR  $url"
      continue
    }

    http_code="${http_ct%% *}"
    ct="${http_ct#* }"

    echo "OK   $url  ($http_code)"

    # tylko z sensownych content-type'ów wyciągamy linki
    if [[ "$ct" == text/html* || "$ct" == text/css* ]]; then
      extract "$file" "$ct" | resolve "$url" | while read -r link; do
        [[ -z "$link" ]] && continue
        lh=$(sed -E 's#^https?://##; s#[/:?].*##' <<<"$link")
        if [[ "$lh" == "$HOST" || "$lh" == "www.$HOST" || "www.$lh" == "$HOST" ]]; then
          echo "$link" >>"$NEXT"
        elif [[ "${link%%\?*}" =~ $ASSET_RE ]]; then
          echo "$link" >>"$NEXT"
        fi
      done
    fi
  done
  mv "$NEXT" "$QUEUE"
done

echo
echo "Kurwa, skończone."
echo "Plików: $(find "$OUT" -type f 2>/dev/null | wc -l), rozmiar: $(du -sh "$OUT" 2>/dev/null | cut -f1)"