#!/usr/bin/env python3
"""Uptober judging aggregator (standard library only; no randomness anywhere).

Inputs
  Three judge sheets, CSV with header: code,originality,craft,lore,supporter,total
    originality 0-40, craft 0-30, lore 0-20, supporter 0-10, whole points; total = their sum.
  A recusal list (optional), CSV or plain text, one "judge,code" pair per line, e.g. "J2,a1b2c3".
    A recused judge's score for that code is ignored even if it appears on their sheet.

Ranking (Official Rules, rule 9; spec section 5.3)
  final score  = median of the three judges' totals; mean of the other two when one judge recused
  tie-breaks   = 1) mean of the totals  2) median Originality  3) median Craft
                 4) still tied: flagged "needs_judge_vote" (a recorded, published majority vote of
                    the judges decides). Never random.

Hash order
  gallery_code(goId)        = first 6 hex characters of sha256(goId)
  review_order(goIds, "J1") = goIds sorted by sha256(goId + ":J1") (":J2", ":J3" for the others)

CLI
  aggregator.py rank --j1 J1.csv --j2 J2.csv --j3 J3.csv [--recusals recusals.csv] [--out ranking.json]
  aggregator.py codes  --entries go_ids.txt          # goId -> gallery code (checks for collisions)
  aggregator.py order  --entries go_ids.txt --judge J2
"""
import argparse
import csv
import hashlib
import io
import json
import os
import sys
from fractions import Fraction

VERSION = "1.0.0"
JUDGES = ("J1", "J2", "J3")
CRITERIA = (("originality", 40), ("craft", 30), ("lore", 20), ("supporter", 10))
COLUMNS = ["code"] + [c for c, _ in CRITERIA] + ["total"]


class SheetError(ValueError):
    pass


# ----------------------------------------------------------------------------- hash order

def gallery_code(go_id):
    """First 6 hex characters of sha256(goId)."""
    return hashlib.sha256(str(go_id).strip().encode("utf-8")).hexdigest()[:6]


def review_key(go_id, judge):
    judge = judge.upper()
    if judge not in JUDGES:
        raise ValueError("judge must be one of %s" % ", ".join(JUDGES))
    return hashlib.sha256((str(go_id).strip() + ":" + judge).encode("utf-8")).hexdigest()


def review_order(go_ids, judge):
    """The order in which `judge` reviews the entries: ascending sha256(goId + ":J<n>") hex."""
    ids = [str(g).strip() for g in go_ids if str(g).strip()]
    return sorted(ids, key=lambda g: (review_key(g, judge), g))


def gallery(go_ids):
    """[{go_id, code, order: {J1: n, J2: n, J3: n}}], 1-based positions. Raises on a code collision."""
    ids = list(dict.fromkeys(str(g).strip() for g in go_ids if str(g).strip()))
    codes = {}
    for g in ids:
        c = gallery_code(g)
        if c in codes:
            raise ValueError("gallery code collision: %s and %s both give %s" % (codes[c], g, c))
        codes[c] = g
    pos = {j: {g: i + 1 for i, g in enumerate(review_order(ids, j))} for j in JUDGES}
    return [{"go_id": g, "code": gallery_code(g), "order": {j: pos[j][g] for j in JUDGES}} for g in ids]


# ----------------------------------------------------------------------------- sheets

def _int(v, field, where):
    s = str(v).strip()
    if not s.lstrip("-").isdigit():
        raise SheetError("%s: %s=%r is not a whole number" % (where, field, v))
    return int(s)


def parse_sheet(text, judge="sheet"):
    """CSV text -> {code: {originality, craft, lore, supporter, total}}. Strict: bad rows raise SheetError."""
    if text.startswith("﻿"):
        text = text[1:]
    rows = list(csv.reader(io.StringIO(text)))
    rows = [r for r in rows if any(c.strip() for c in r)]
    if not rows:
        return {}
    header = [h.strip().lower() for h in rows[0]]
    if header != COLUMNS:
        raise SheetError("%s: header must be %s, got %s" % (judge, ",".join(COLUMNS), ",".join(header)))
    out, problems = {}, []
    for n, r in enumerate(rows[1:], start=2):
        where = "%s line %d" % (judge, n)
        try:
            if len(r) != len(COLUMNS):
                raise SheetError("%s: expected %d columns, got %d" % (where, len(COLUMNS), len(r)))
            code = r[0].strip().lower()
            if not code:
                raise SheetError("%s: empty code" % where)
            if code in out:
                raise SheetError("%s: code %s appears twice" % (where, code))
            s = {}
            for (c, mx), v in zip(CRITERIA, r[1:5]):
                x = _int(v, c, where)
                if not 0 <= x <= mx:
                    raise SheetError("%s: %s=%d outside 0-%d" % (where, c, x, mx))
                s[c] = x
            s["total"] = _int(r[5], "total", where)
            if s["total"] != sum(s[c] for c, _ in CRITERIA):
                raise SheetError("%s: total %d != sum of criteria %d" % (where, s["total"],
                                                                         sum(s[c] for c, _ in CRITERIA)))
            out[code] = s
        except SheetError as e:
            problems.append(str(e))
    if problems:
        raise SheetError("; ".join(problems))
    return out


def parse_recusals(text):
    """'judge,code' per line (header 'judge,code' optional, '#' comments ok) -> set of (judge, code)."""
    out = set()
    for line in (text or "").splitlines():
        line = line.split("#", 1)[0].strip()
        if not line:
            continue
        parts = [p.strip() for p in line.replace(";", ",").split(",")]
        if len(parts) != 2:
            raise SheetError("recusal line %r: expected 'judge,code'" % line)
        j, c = parts[0].upper(), parts[1].lower()
        if (j, c) == ("JUDGE", "code"):
            continue
        if j not in JUDGES:
            raise SheetError("recusal line %r: judge must be one of %s" % (line, ", ".join(JUDGES)))
        out.add((j, c))
    return out


# ----------------------------------------------------------------------------- ranking

def median(xs):
    xs = sorted(Fraction(x) for x in xs)
    if not xs:
        raise ValueError("median of nothing")
    m = len(xs) // 2
    return xs[m] if len(xs) % 2 else (xs[m - 1] + xs[m]) / 2


def mean(xs):
    xs = [Fraction(x) for x in xs]
    return sum(xs) / len(xs)


def num(f):
    """Fraction -> int when whole, else float rounded to 4 places (display only; ranking uses exact values)."""
    f = Fraction(f)
    return int(f) if f.denominator == 1 else round(float(f), 4)


def rank(sheets, recusals=()):
    """sheets: {"J1": {code: scores}, "J2": ..., "J3": ...}; recusals: iterable of (judge, code).

    Returns a list of entries, best first:
      {rank, code, final, mean_total, median_originality, median_craft, judges, recused, totals,
       tie_group, needs_judge_vote}
    rank is shared by entries that stay tied after every tie-break (1, 2, 2, 4 ...).
    """
    missing = [j for j in JUDGES if j not in sheets]
    if missing:
        raise SheetError("missing sheet(s): %s" % ", ".join(missing))
    rec = {(j.upper(), c.lower()) for j, c in recusals}
    codes = sorted(set().union(*(set(sheets[j]) for j in JUDGES)) | {c for _, c in rec})
    problems, rows = [], []
    for code in codes:
        used = [j for j in JUDGES if (j, code) not in rec]
        absent = [j for j in used if code not in sheets[j]]
        if absent:
            problems.append("code %s: no score from %s (and no recusal)" % (code, ", ".join(absent)))
            continue
        if len(used) < 2:
            problems.append("code %s: only %d judge(s) left after recusals; needs at least 2" % (code, len(used)))
            continue
        sc = [sheets[j][code] for j in used]
        totals = [s["total"] for s in sc]
        final = median(totals) if len(used) == 3 else mean(totals)
        rows.append({"code": code, "judges": used, "recused": [j for j in JUDGES if j not in used],
                     "totals": {j: sheets[j][code]["total"] for j in used},
                     "_key": (final, mean(totals), median(s["originality"] for s in sc),
                              median(s["craft"] for s in sc))})
    if problems:
        raise SheetError("; ".join(problems))
    rows.sort(key=lambda r: (tuple(-k for k in r["_key"]), r["code"]))
    count = {}
    for r in rows:
        count[r["_key"]] = count.get(r["_key"], 0) + 1
    out, first_rank, tie_ids = [], {}, {}
    for i, r in enumerate(rows):
        k = r["_key"]
        first_rank.setdefault(k, i + 1)
        tied = count[k] > 1
        if tied and k not in tie_ids:
            tie_ids[k] = len(tie_ids) + 1
        out.append({"rank": first_rank[k], "code": r["code"], "final": num(k[0]), "mean_total": num(k[1]),
                    "median_originality": num(k[2]), "median_craft": num(k[3]), "judges": r["judges"],
                    "recused": r["recused"], "totals": r["totals"], "tie_group": tie_ids.get(k),
                    "needs_judge_vote": tied})
    return out


# ----------------------------------------------------------------------------- CLI

def sha256_file(path):
    with open(path, "rb") as f:
        return hashlib.sha256(f.read()).hexdigest()


def read_text(path):
    with open(path, encoding="utf-8") as f:
        return f.read()


def read_ids(path):
    out = []
    for line in read_text(path).splitlines():
        s = line.split("#", 1)[0].strip().split(",")[0].strip()
        if s and s.lower() not in ("go_id", "goid", "id"):
            out.append(s)
    return out


def main(argv=None):
    ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    sub = ap.add_subparsers(dest="cmd", required=True)
    p = sub.add_parser("rank")
    p.add_argument("--j1", required=True)
    p.add_argument("--j2", required=True)
    p.add_argument("--j3", required=True)
    p.add_argument("--recusals")
    p.add_argument("--out")
    p = sub.add_parser("codes")
    p.add_argument("--entries", required=True, help="one goId per line (first CSV column also works)")
    p = sub.add_parser("order")
    p.add_argument("--entries", required=True)
    p.add_argument("--judge", required=True, choices=JUDGES)
    args = ap.parse_args(argv)

    if args.cmd == "codes":
        print(json.dumps(gallery(read_ids(args.entries)), indent=1))
        return 0
    if args.cmd == "order":
        ids = review_order(read_ids(args.entries), args.judge)
        print(json.dumps([{"position": i + 1, "go_id": g, "code": gallery_code(g)} for i, g in enumerate(ids)],
                         indent=1))
        return 0

    paths = {"J1": args.j1, "J2": args.j2, "J3": args.j3}
    try:
        sheets = {j: parse_sheet(read_text(p), j) for j, p in paths.items()}
        rec = parse_recusals(read_text(args.recusals)) if args.recusals else set()
        ranking = rank(sheets, rec)
    except SheetError as e:
        print("refusing to rank: %s" % e, file=sys.stderr)
        return 2
    doc = {"aggregator_version": VERSION,
           "aggregator_sha256": sha256_file(os.path.abspath(__file__)),
           "sheets": {j: {"file": os.path.basename(p), "sha256": sha256_file(p)} for j, p in paths.items()},
           "recusals": sorted("%s,%s" % x for x in rec),
           "recusals_sha256": sha256_file(args.recusals) if args.recusals else None,
           "rule": "final = median of 3 totals (mean of 2 if a judge recused); ties: mean of totals, "
                   "median originality, median craft, then a recorded judge vote. No randomness.",
           "unresolved_ties": sorted({r["tie_group"] for r in ranking if r["needs_judge_vote"]}),
           "ranking": ranking}
    text = json.dumps(doc, indent=1)
    if args.out:
        with open(args.out, "w") as f:
            f.write(text + "\n")
    print(text)
    return 0


if __name__ == "__main__":
    sys.exit(main())
