#!/usr/bin/env python3
"""freeze_sheet_parse.py — convert a posted day-N freeze sheet (board text)
into the sheet.json shape freeze_rewalk.py consumes.

The freeze-row spec settled in musemoneychallenge #102125/#102279:
  whole sheet : pinned day-close block, pool address(es) pinned at weigh-in
  per row     : muse_id, claimed total, one line per instrument,
                claim timestamp
The exact TEXT layout is the publisher's choice, so this parser is
tolerant: markdown tables, "key: value" lines, and bullet lists all parse.
Anything it cannot extract is reported LOUD in "missing" — a re-walk run
on an incomplete cfg is worse than no run (unknown fields guess silently).

Usage:
  python3 freeze_sheet_parse.py sheet.txt [-o outdir] [--numeraire USDC]
Emits sheet_cfg.json + prints a field-coverage report.
"""
import json, os, re, sys

ADDR = r"0x[0-9a-fA-F]{40}"
KNOWN_TOKENS = {  # lowercase symbol -> (addr, decimals); extend as pinned
    "usdc": ("0x833589fcd6edb6e08f4c7c32d4f71b54bda02913", 6),
    "weth": ("0x4200000000000000000000000000000000000006", 18),
}
NUM_SYMBOLS = {"usdc": ("0x833589fcd6edb6e08f4c7c32d4f71b54bda02913", 6),
               "weth": ("0x4200000000000000000000000000000000000006", 18)}

RE_BLOCK = re.compile(r"(?:close|close[-_ ]?block|day[-_ ]?close|"
                      r"pinned[-_ ]?block|block)\D{0,15}?\b(\d{6,9})\b", re.I)
RE_POOL = re.compile(r"pool\D{0,30}?(" + ADDR + ")", re.I)
RE_ADDR = re.compile(ADDR)
RE_ROWHEAD = re.compile(
    r"(?:^|\|\s*)([A-Za-z][\w .'-]{1,30}?)\s*[:|]\s*(?:claimed(?:_total)?|"
    r"total)?\s*\$?([\d,]+(?:\.\d+)?)\s*(?:([A-Za-z]{2,10})\b)?", re.I)
RE_TS = re.compile(r"\b(\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}(?::\d{2})?Z?)\b")
RE_INSTR = re.compile(
    r"(?:^|\|\s*|[-*•]\s*)([A-Za-z]{2,12}|" + ADDR + r")\s*[:=]?\s*"
    r"([\d,]+(?:\.\d+)?)\s*(?:raw\s*[:=]?\s*(\d+))?", re.I)
RE_MUSEID = re.compile(r"\b(muse_[a-z0-9]+)\b")


def num(s):
    return float(s.replace(",", ""))


def parse_text(text):
    cfg = {"close_block": None, "numeraire": None, "pools": {},
           "rows": [], "missing": [], "source_chars": len(text)}
    m = RE_BLOCK.search(text)
    if m:
        cfg["close_block"] = int(m.group(1))
    else:
        cfg["missing"].append("close_block")

    for pm in RE_POOL.finditer(text):
        cfg["pools"].setdefault("unassigned", []).append(pm.group(1))
    if not cfg["pools"]:
        cfg["missing"].append("mark pool pin (freeze spec requires a pool "
                              "pinned at weigh-in; none found in text)")
    elif "unassigned" in cfg["pools"]:
        cfg["missing"].append("pool->token assignment (pool addrs found "
                              "but not bound to tokens)")

    # split into row chunks: a new row starts at a 'muse:'/'row:'/name-total
    # header line, a markdown table separator-free row, or 'muse_xxx' id.
    lines = [l.strip() for l in text.splitlines() if l.strip()]
    cur = None
    header_seen = False
    for l in lines:
        low = l.lower()
        if l.startswith("|"):
            cells = [c.strip() for c in l.strip("|").split("|")]
            if not header_seen:  # first pipe row = header, learn columns
                header_seen = True
                continue
            if all(re.match(r"^[-: ]*$", c) for c in cells):
                continue
            cur = {"muse_id": None, "name": cells[0] or None, "wallet": None,
                   "claimed_total": None, "claim_ts": None,
                   "instruments": []}
            cfg["rows"].append(cur)
            for c in cells[1:]:
                cl = c.lower()
                a = RE_ADDR.search(c)
                if a and not cur["wallet"] and "pool" not in cl:
                    cur["wallet"] = a.group(0)
                    continue
                ts = RE_TS.search(c)
                if ts and not cur["claim_ts"]:
                    cur["claim_ts"] = ts.group(1)
                    continue
                for iv in re.finditer(r"([A-Za-z]{2,12}|" + ADDR + r")\s*[:=]"
                                      r"\s*([\d,]+(?:\.\d+)?)", c):
                    tok, amt = iv.group(1), num(iv.group(2))
                    ins = {"symbol": tok, "amount": amt}
                    if tok.lower() in KNOWN_TOKENS:
                        ins["token"], ins["decimals"] = \
                            KNOWN_TOKENS[tok.lower()]
                    elif re.match(ADDR + "$", tok):
                        ins["token"] = tok.lower()
                    cur["instruments"].append(ins)
                if cur["claimed_total"] is None:
                    hm = re.match(r"^\$?([\d,]+(?:\.\d+)?)$", cl)
                    if hm:
                        cur["claimed_total"] = num(hm.group(1))
            continue
        if re.match(r"^[-*•]?\s*(?:row|muse|entry|name)\b", low) or \
                RE_MUSEID.search(l):
            m2 = RE_MUSEID.search(l)
            mid = m2.group(1) if m2 else None
            if cur is None or (mid and mid != cur.get("muse_id")):
                cur = {"muse_id": mid, "name": None, "wallet": None,
                       "claimed_total": None, "claim_ts": None,
                       "instruments": []}
                cfg["rows"].append(cur)
            elif mid:
                cur["muse_id"] = mid
        if cur is None:
            continue
        a = RE_ADDR.search(l)
        if a and not cur["wallet"] and "pool" not in low:
            cur["wallet"] = a.group(0)
        ts = RE_TS.search(l)
        if ts and not cur["claim_ts"]:
            cur["claim_ts"] = ts.group(1)
        hm = re.search(r"(?:claimed(?:_total)?|total)\s*[:=]?\s*\$?"
                       r"([\d,]+(?:\.\d+)?)", low)
        if hm and cur["claimed_total"] is None:
            cur["claimed_total"] = num(hm.group(1))
        im = RE_INSTR.match(l.lstrip("| -*•"))
        if im and not re.match(r"^(?:claimed|total|close|pool|block|muse|"
                               r"row|name|wallet|entry)\b",
                               im.group(1).lower()):
            tok, amt, raw = im.group(1), num(im.group(2)), im.group(3)
            ins = {"symbol": tok, "amount": amt}
            if tok.lower() in KNOWN_TOKENS:
                ins["token"], ins["decimals"] = KNOWN_TOKENS[tok.lower()]
            elif re.match(ADDR + "$", tok):
                ins["token"] = tok.lower()
            if raw:
                ins["amount_raw"] = raw
            cur["instruments"].append(ins)
    # table fallback: rows parsed from pipes carry instrument columns
    cfg["rows"] = [r for r in cfg["rows"]
                   if r["muse_id"] or r["name"] or r["instruments"]]
    for r in cfg["rows"]:
        if r["wallet"] is None:
            r["_missing_wallet"] = True
        if r["claimed_total"] is None:
            r["_missing_total"] = True
    if not cfg["rows"]:
        cfg["missing"].append("rows")
    if any(r.get("_missing_wallet") for r in cfg["rows"]):
        cfg["missing"].append("wallet per row (wallet registry needed)")
    if any(r.get("_missing_total") for r in cfg["rows"]):
        cfg["missing"].append("claimed_total per row")
    unbound = [i["symbol"] for r in cfg["rows"] for i in r["instruments"]
               if "token" not in i]
    if unbound:
        cfg["missing"].append("token addr for symbols: "
                              + ",".join(sorted(set(unbound))))
    return cfg


def main():
    args = [a for a in sys.argv[1:] if not a.startswith("--")]
    num_arg = "usdc"
    if "--numeraire" in sys.argv:
        num_arg = sys.argv[sys.argv.index("--numeraire") + 1].lower()
    outdir = "."
    if "-o" in args:
        outdir = args[args.index("-o") + 1]
    text = open(args[0]).read()
    cfg = parse_text(text)
    ntok, ndec = NUM_SYMBOLS.get(num_arg, NUM_SYMBOLS["usdc"])
    cfg["numeraire"] = {"token": ntok, "decimals": ndec,
                        "symbol": num_arg.upper()}
    os.makedirs(outdir, exist_ok=True)
    out = os.path.join(outdir, "sheet_cfg.json")
    json.dump(cfg, open(out, "w"), indent=1)
    print(f"rows={len(cfg['rows'])} close_block={cfg['close_block']} "
          f"pools={len(cfg['pools'].get('unassigned', []))} "
          f"missing={cfg['missing'] or 'none'} -> {out}")
    for r in cfg["rows"]:
        print(f"  {r.get('name') or r.get('muse_id')}: "
              f"total={r['claimed_total']} wallet={r['wallet']} "
              f"instruments={len(r['instruments'])}")


if __name__ == "__main__":
    main()
