#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""MMDCS bulk downloader / MMDCS 批量下载脚本 (Python 3, no extra packages / 无需第三方库)

Windows:       py -3 download_r101873.py --output "D:\\Downloads\\catalogue"
macOS / Linux: python3 download_r101873.py --output "/path/to/catalogue"
Help / 帮助:   python3 download_r101873.py --help

ENGLISH
Choose the folder that already contains your downloaded files with --output.
Without --output, files are saved beside this script. Relative subdirectories
are preserved. The file list covers the whole resource, not just one page.
Existing files with matching byte sizes are skipped. No server checksums are
available: this is SIZE CHECKING ONLY, not proof of content integrity. HTML
error pages are rejected for non-HTML files. Same-size corruption can go
undetected. Files with different or unknown sizes are preserved and reported
as conflicts. Inspect or move them first, or explicitly use --overwrite to
download replacements. --overwrite re-downloads ALL selected files.
Only this script's .part + .part.json files can resume; keep both when stopping.
HTTP errors (including 416) are never saved as completed files. Replacements
are installed only after a successful download and size check when available.
Run the same command again after interruption; do not remove successful files.
Use --match "tractor-gal-*.csv" to download only matching relative file paths.
The latest file list is fetched once at startup; each run uses that snapshot.
Run the same script again to pick up newly added files. --help works offline.

中文说明
使用 --output 指定已经存有下载文件的目录，避免因选择新目录而重复下载。
未指定时保存到脚本所在目录；保留相对子目录；清单覆盖整个资源而非当前分页。
已有文件字节数一致时跳过。服务端尚未提供校验和，因此仅比较大小，不能证明
内容完整正确，无法识别所有同大小损坏；非 HTML 文件中的 HTML 错误页会被拒绝。
大小不同或未知的同名文件默认保留并报告冲突。检查、移走冲突文件后重新执行，
或明确添加 --overwrite 重新下载并替换；该选项会重新下载全部选中的文件。
只续传本脚本创建的 .part 和 .part.json 临时文件，中断后请保留这两个文件。
HTTP 错误（包括 416）不会保存为正式文件；成功下载并核对可用大小后才替换文件。
中断后重新执行相同命令即可，不需要删除已经成功下载的文件。
可用 --match "tractor-gal-*.csv" 按相对路径筛选文件。
每次启动获取一次最新清单，本次执行使用同一份快照。
资源新增文件后重新执行本脚本即可；--help 无需网络。
"""

import argparse
import fnmatch
from http.client import HTTPException
import json
import os
from pathlib import Path
import re
import sys
import time
from urllib.error import HTTPError, URLError
from urllib.parse import urlsplit
from urllib.request import Request, urlopen

MANIFEST_URL = "https://nadc.china-vo.org/res/r101873/files/manifest.json"


def fetch_manifest():
    request = Request(MANIFEST_URL, headers={"Accept": "application/json",
                                            "User-Agent": "MMDCS-downloader/1.0"})
    with urlopen(request, timeout=60) as response:
        if response.status != 200:
            raise ValueError("Unexpected manifest status / 清单 HTTP 状态异常: " + str(response.status))
        manifest = json.loads(response.read().decode("utf-8"))
    if not isinstance(manifest, dict) or not isinstance(manifest.get("files"), list):
        raise ValueError("Invalid file manifest / 文件清单格式无效")
    for item in manifest["files"]:
        if (not isinstance(item, dict)
                or not isinstance(item.get("path"), str) or not item["path"]
                or "size" not in item
                or (item["size"] is not None and (type(item["size"]) is not int or item["size"] < 0))
                or not isinstance(item.get("url"), str)
                or urlsplit(item["url"]).scheme not in ("http", "https")
                or not urlsplit(item["url"]).netloc):
            raise ValueError("Invalid file entry / 文件清单条目无效")
    return manifest["files"]


def output_path(root, name):
    parts = name.replace("\\", "/").split("/")
    if any(part in ("", ".", "..") or ":" in part for part in parts):
        raise ValueError("Unsafe relative path / 不安全的相对路径: " + name)
    if os.name == "nt" and any(
        re.search(r'[<>"|?*\x00-\x1f]', part) or part.endswith((".", " "))
        or re.match(r"^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(?:\.|$)", part, re.I)
        for part in parts
    ):
        raise ValueError("Unsupported Windows filename / Windows 不支持的文件名: " + name)
    target = root.joinpath(*parts)
    if target.is_symlink():
        raise ValueError("Refusing symbolic link / 不写入符号链接: " + name)
    target.resolve().relative_to(root)
    return target


def html_error(prefix, target):
    # The service currently labels some CSV responses text/html; inspect the body.
    return target.suffix.lower() not in (".html", ".htm") and prefix.lstrip(
        b"\xef\xbb\xbf \t\r\n"
    ).lower().startswith((b"<html", b"<!doctype html"))


def download_file(item, root, overwrite=False):
    target = output_path(root, item["path"])
    size = item["size"]
    if target.exists() and not overwrite:
        with target.open("rb") as source:
            is_error = html_error(source.read(512), target)
        # ponytail: size-only skip until the service stores trusted checksums.
        if size is not None and target.stat().st_size == size and not is_error:
            return "skipped"
        raise ValueError("Existing file conflicts; inspect/move it or use --overwrite / "
                         "已有文件冲突，请检查、移走或使用 --overwrite: " + str(target))
    if urlsplit(item["url"]).scheme not in ("http", "https"):
        raise ValueError("Only HTTP(S) downloads are supported / 仅支持 HTTP(S) 下载")
    target.parent.mkdir(parents=True, exist_ok=True)
    part = output_path(root, item["path"] + ".part")
    state_path = output_path(root, item["path"] + ".part.json")
    state = {}
    if state_path.exists():
        with state_path.open(encoding="utf-8") as source:
            state = json.load(source)
    if part.exists() and state.get("item") != item:
        raise ValueError("Unrecognized partial file; move it first / "
                         "临时文件不属于本清单，请先移走: " + str(part))

    for attempt in range(3):
        offset = part.stat().st_size if part.exists() and state.get("validator") else 0
        if size is not None and offset >= size:
            offset = 0  # Never request a range at or beyond EOF / 不发起越界续传。
        headers = {"Accept-Encoding": "identity", "User-Agent": "MMDCS-downloader/1.0"}
        if offset:
            headers.update({"Range": "bytes={0}-".format(offset), "If-Range": state["validator"]})
        try:
            with urlopen(Request(item["url"], headers=headers), timeout=60) as response:
                if response.status not in (200, 206):
                    raise ValueError("Unexpected HTTP status / 异常 HTTP 状态: " + str(response.status))
                if response.headers.get("Content-Encoding", "identity").lower() != "identity":
                    raise ValueError("Encoded response cannot be resumed / 不支持压缩传输响应")
                length = response.headers.get("Content-Length")
                length = int(length) if length is not None else None
                if response.status == 206:
                    match = re.fullmatch(r"bytes (\d+)-(\d+)/(\d+)",
                                         response.headers.get("Content-Range", ""))
                    if not offset or not match:
                        raise ValueError("Invalid Content-Range / 无效的续传响应范围")
                    start, end, total = map(int, match.groups())
                    if start != offset or end != total - 1 or (length is not None and length != end - start + 1):
                        raise ValueError("Mismatched Content-Range / 续传响应范围不匹配")
                    if state.get("total") != total:
                        raise ValueError("Remote file changed; move partial files and rerun / 远端文件已变化，请移走临时文件后重新执行")
                    received_validator = response.headers.get(
                        "ETag" if state["validator"].startswith('"') else "Last-Modified")
                    if received_validator and received_validator != state["validator"]:
                        raise ValueError("Remote file changed during resume / 续传时远端文件已变化")
                else:
                    offset, total = 0, length  # Server ignored Range: restart, never append.
                if size is not None and total is not None and size != total:
                    raise ValueError("Remote size changed; rerun to refresh the list / 远端大小已变化，请重新执行以刷新清单")
                expected = size if size is not None else total
                prefix = response.read(512)
                if not offset and html_error(prefix, target):
                    raise ValueError("HTML error/login page received / 收到 HTML 错误页或登录页")
                etag = response.headers.get("ETag", "")
                validator = etag if etag and not etag.startswith("W/") else response.headers.get("Last-Modified")
                state = {"item": item, "validator": validator, "total": total}
                # Clear an old partial before saving the new representation's validator.
                with part.open("ab" if offset else "wb") as output:
                    with state_path.open("w", encoding="utf-8") as dest:
                        json.dump(state, dest, ensure_ascii=True)
                    output.write(prefix)
                    while True:
                        chunk = response.read(1024 * 1024)
                        if not chunk:
                            break
                        output.write(chunk)
                if expected is not None and part.stat().st_size != expected:
                    raise OSError("Incomplete download / 下载大小不完整")
            if target.exists() and not overwrite:
                raise ValueError("Destination appeared during download / 下载期间目标文件已存在")
            part.replace(target)
            state_path.unlink()
            return "downloaded"
        except HTTPError as error:
            error.close()
            if error.code == 416 and offset:
                state["validator"] = None
                with state_path.open("w", encoding="utf-8") as dest:
                    json.dump(state, dest)
            elif error.code not in (408, 429, 500, 502, 503, 504):
                raise
            if attempt == 2:
                raise
        except (OSError, URLError, HTTPException):
            if attempt == 2:
                raise
        print("Retry / 重试 ({0}/3): {1}".format(attempt + 2, item["path"]))
        time.sleep(attempt + 1)


def main():
    parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    parser.add_argument("--output", type=Path, default=Path(__file__).resolve().parent,
                        help="Existing download root / 已有文件的下载根目录（默认为脚本目录）")
    parser.add_argument("--match", default="*", help="Relative path wildcard / 相对路径通配符")
    parser.add_argument("--overwrite", action="store_true",
                        help="Re-download and replace selected files / 重新下载并替换所选文件")
    args = parser.parse_args()
    root = args.output.expanduser().resolve()
    try:
        files = fetch_manifest()
    except (OSError, URLError, ValueError, HTTPException) as error:
        print("Cannot fetch file list / 无法获取文件清单: " + str(error), file=sys.stderr)
        return 1
    if not files:
        print("No files in this resource / 该资源暂无文件")
        return 0
    selected = [item for item in files if fnmatch.fnmatchcase(item["path"], args.match)]
    if not selected:
        parser.error("No matching files / 没有匹配的文件")
    # Reserve data and temporary names before writing anything, including on Windows.
    reserved = set()
    for item in selected:
        for suffix in ("", ".part", ".part.json"):
            key = str(output_path(root, item["path"] + suffix)).casefold()
            if key in reserved or key == str(Path(__file__).resolve()).casefold():
                parser.error("Conflicting output paths / 输出路径冲突: " + item["path"])
            reserved.add(key)
    counts = {"downloaded": 0, "skipped": 0, "failed": 0}
    print("Existing files are checked by SIZE ONLY / 已有文件仅比较大小，不进行校验和验证")
    for index, item in enumerate(selected, 1):
        print("[{0}/{1}] {2}".format(index, len(selected), item["path"]), flush=True)
        try:
            result = download_file(item, root, args.overwrite)
            counts[result] += 1
            print("OK / 完成" if result == "downloaded" else "SKIP (size only) / 跳过（仅比较大小）")
        except (OSError, URLError, ValueError, HTTPException) as error:
            counts["failed"] += 1
            print("FAILED / 失败: " + str(error), file=sys.stderr)
    print("Downloaded / 下载: {downloaded}; Skipped / 跳过: {skipped}; Failed / 失败: {failed}".format(**counts))
    return 1 if counts["failed"] else 0


if __name__ == "__main__":
    try:
        sys.exit(main())
    except KeyboardInterrupt:
        print("\nInterrupted; run the same command to resume / 已中断，重新执行相同命令即可继续", file=sys.stderr)
        sys.exit(130)