#!/usr/bin/env python3
"""Compare a metadata-only Baidu inventory with the three local archives.

No remote file body is downloaded.  Matching is deliberately conservative:
only identical basename and byte size is counted as an exact local duplicate.
"""

from __future__ import annotations

import argparse
import collections
import datetime as dt
import gzip
import json
import os
import re
import shutil
from pathlib import Path


DEFAULT_LOCAL_ROOTS = [
    Path("/Users/zhuboyu/Desktop/龙门计划/投资/Quant/tushare_15000_history_by_api_packages_20260627"),
    Path("/Users/zhuboyu/Desktop/龙门计划/投资/Quant/【53】2026A股分钟日频-持续更新到年底"),
    Path("/Users/zhuboyu/Desktop/龙门计划/投资/Quant/【49】A股5800股票日线持续更新"),
]


def human_bytes(value: int) -> str:
    size = float(value)
    for unit in ["B", "KB", "MB", "GB", "TB"]:
        if size < 1024 or unit == "TB":
            return f"{size:.2f} {unit}"
        size /= 1024
    return f"{value} B"


def category(path: str) -> str:
    # Ignore the generic share title (which itself contains “分钟+日频”); classify
    # from the filename and its nearest folders so every file is not mislabeled.
    parts = [part for part in path.lower().split("/") if part]
    text = "/".join(parts[-4:])
    if any(token in text for token in ["level2", "level-2", "逐笔", "tick", "委托队列", "盘口"]):
        return "level2_or_tick"
    if any(token in text for token in ["竞价", "auction"]):
        return "auction"
    if any(token in text for token in ["财务", "利润", "资产负债", "现金流", "fundamental", "income", "balancesheet"]):
        return "fundamental"
    if any(token in text for token in ["etf", "基金", "iopv"]):
        return "fund_or_etf"
    if any(token in text for token in ["股票日线", "日线", "daily", "日k", "1d_price", "day"]):
        return "daily"
    if any(token in text for token in ["指数", "index"]):
        return "index"
    if any(token in text for token in ["分钟", "分时", "min", "1m", "5m", "15m", "30m", "60m"]):
        return "minute"
    if any(token in text for token in ["复权", "adj_factor", "factor"]):
        return "adjustment"
    return "other"


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--inventory", type=Path, required=True)
    parser.add_argument("--result", type=Path, required=True)
    parser.add_argument("--report", type=Path, required=True)
    args = parser.parse_args()

    inventory = json.loads(args.inventory.read_text(encoding="utf-8"))
    corrected_years = collections.Counter()
    for record in inventory["records"]:
        if record["isDirectory"]:
            continue
        for year in set(re.findall(r"(?<!\d)((?:199\d|20[01]\d|202[0-6]))(?:[01]\d[0-3]\d)?(?!\d)", record.get("name") or "")):
            corrected_years[year] += 1
    inventory["byYear"] = dict(sorted(corrected_years.items()))
    # Normalize the first-run pretty JSON to compact JSON before publishing.
    args.inventory.write_text(json.dumps(inventory, ensure_ascii=False, separators=(",", ":")) + "\n", encoding="utf-8")
    compressed_inventory = args.inventory.with_suffix(args.inventory.suffix + ".gz")
    with args.inventory.open("rb") as source, gzip.open(compressed_inventory, "wb", compresslevel=9) as target:
        shutil.copyfileobj(source, target)
    local_files = []
    local_keys = set()
    local_total = 0
    for root in DEFAULT_LOCAL_ROOTS:
        for directory, _, names in os.walk(root):
            for name in names:
                path = Path(directory) / name
                try:
                    size = path.stat().st_size
                except OSError:
                    continue
                local_total += size
                local_keys.add((name.casefold(), size))
                local_files.append({"path": str(path), "name": name, "size": size})

    remote_files = [record for record in inventory["records"] if not record["isDirectory"]]
    exact = []
    missing = []
    category_counts = collections.Counter()
    category_bytes = collections.Counter()
    missing_category_counts = collections.Counter()
    missing_category_bytes = collections.Counter()
    for item in remote_files:
        item_category = category(item.get("path") or item.get("name") or "")
        category_counts[item_category] += 1
        category_bytes[item_category] += item["size"]
        if ((item.get("name") or "").casefold(), item["size"]) in local_keys:
            exact.append(item)
        else:
            missing.append(item)
            missing_category_counts[item_category] += 1
            missing_category_bytes[item_category] += item["size"]

    priority_order = {
        "auction": 0,
        "minute": 1,
        "level2_or_tick": 2,
        "fund_or_etf": 3,
        "fundamental": 4,
        "adjustment": 5,
        "daily": 6,
        "index": 7,
        "other": 8,
    }
    candidates = sorted(
        missing,
        key=lambda item: (
            priority_order[category(item.get("path") or "")],
            -item["size"],
            item.get("path") or "",
        ),
    )[:300]
    generated_at = dt.datetime.now(dt.timezone.utc).isoformat()
    result = {
        "version": "baidu-local-gap-analysis-v1",
        "generatedAt": generated_at,
        "mode": "METADATA_ONLY_NO_REMOTE_BODY_DOWNLOAD",
        "matchingRule": "exact case-insensitive basename plus byte size",
        "local": {
            "roots": [str(path) for path in DEFAULT_LOCAL_ROOTS],
            "files": len(local_files),
            "bytes": local_total,
            "human": human_bytes(local_total),
        },
        "remote": {
            "files": len(remote_files),
            "bytes": inventory["totalBytes"],
            "human": inventory["totalHuman"],
            "compressedInventory": str(compressed_inventory),
            "filesByDetectedYear": inventory["byYear"],
        },
        "exactLocalDuplicates": {
            "files": len(exact),
            "bytes": sum(item["size"] for item in exact),
        },
        "notLocallyMatched": {
            "files": len(missing),
            "bytes": sum(item["size"] for item in missing),
            "human": human_bytes(sum(item["size"] for item in missing)),
        },
        "categories": {
            key: {
                "remoteFiles": category_counts[key],
                "remoteBytes": category_bytes[key],
                "unmatchedFiles": missing_category_counts[key],
                "unmatchedBytes": missing_category_bytes[key],
            }
            for key in sorted(category_counts)
        },
        "semanticDecisions": [
            "远程日线与本地TuShare 2000-2026日线研究范围高度重叠，当前策略无需再次整包下载",
            "本地已有adj_factor、stk_factor与日线收益链；远程复权类文件在字段核验前不替换主数据",
            "远程历史分钟是主要新增价值，但应先按市场状态选择少数年度分片做质量与策略增量检验",
            "文件名与大小不同不等于数值内容不同；精确重复数只是保守下限，不能解释为419GB都必须下载",
        ],
        "priorityCandidates": candidates,
        "safety": {
            "remoteFileBodiesDownloaded": 0,
            "brokerConnected": False,
            "ordersSubmitted": 0,
        },
    }
    args.result.parent.mkdir(parents=True, exist_ok=True)
    args.result.write_text(json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")

    rows = []
    for key, values in result["categories"].items():
        rows.append(
            f"| {key} | {values['remoteFiles']:,} | {human_bytes(values['remoteBytes'])} | "
            f"{values['unmatchedFiles']:,} | {human_bytes(values['unmatchedBytes'])} |"
        )
    report = [
        "# 百度网盘与本地三目录差异分析",
        "",
        f"- 生成时间：{generated_at}",
        f"- 网盘清单：{len(remote_files):,}个文件，{inventory['totalHuman']}（只读元数据）。",
        f"- 本地三目录：{len(local_files):,}个文件，{human_bytes(local_total)}。",
        f"- 保守判定的本地精确重复：{len(exact):,}个文件；未本地匹配：{len(missing):,}个文件，{human_bytes(sum(item['size'] for item in missing))}。",
        "",
        "## 类型差异",
        "",
        "| 类型 | 远程文件 | 远程大小 | 未匹配文件 | 未匹配大小 |",
        "|---|---:|---:|---:|---:|",
        *rows,
        "",
        "## 研究决策",
        "",
        "- 远程日线与本地TuShare 2000–2026日线范围高度重叠，当前全市场日线策略不再整包下载。",
        "- 本地已有adj_factor、stk_factor和无幸存者日期分区收益链；远程复权文件在字段核验前不替换主数据。",
        "- 远程历史分钟是主要新增价值，但应先选2008、2015、2020、2024/2025等不同市场状态的年度1分钟分片做质量审计和增量回测；只有确实提升样本外结果才扩大。",
        "- 仅有7个文件满足“同名且同字节数”的严格重复条件，这只是重复下限；压缩方式或命名不同仍可能是相同数据，不能据此认定419GB都要下载。",
        "",
        "## 计算边界",
        "",
        "可以不下载整套419.56GB完成目录、年份、文件类型、大小和重复项学习；不能不传输文件正文就对其中K线、逐笔或财务数值做回测。后续只应按研究缺口选择最小分片，并在下载后完成字段、复权、时间戳和泄漏审计。",
        "",
        "真实资金仍为LOCKED；本分析不连接券商、不提交订单，也不把文件名当作已验证的数据质量。",
        "",
    ]
    args.report.parent.mkdir(parents=True, exist_ok=True)
    args.report.write_text("\n".join(report), encoding="utf-8")


if __name__ == "__main__":
    main()
