#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
# Copyright 2025-2026 73 Holding B.V. -- CommitMotion. The one open file of CommitMotion: read it, run it, keep it.
# Licensed under the Apache License, Version 2.0: https://www.apache.org/licenses/LICENSE-2.0
"""
extract.py — turn one year of a git repo into timeline.json for the viewer.

Usage:
    python3 extract.py <repo-path> <year> [branch] [-o out.json] [--remote github.com/owner/repo] [--previous N]

For a private repository, run it where the repository is and look at what it wrote before you share it:
    python3 extract.py . 2025 --no-emails -o 2025.json      # names stay, e-mail addresses go
    python3 extract.py . 2025 --anonymise --no-subjects --no-code --no-timezones -o 2025.json   # as little as possible

--previous N (0 to 5) also writes a short summary of each of the N years before it, for the visualizations that set
years side by side: no codebase map for those, only the days, the releases and the people. With --month it is the same
month of each of those years (March 2025 beside March 2024 and March 2023):
    python3 extract.py . 2025 --previous 2 -o 2025.json
    python3 extract.py . 2025 --month 3 --previous 2 -o 2025-03.json

Everything is computed from `git` plumbing; nothing hits the network.
Avatars are resolved by the viewer from author e-mails (GitHub noreply ids
or the avatars.githubusercontent.com e-mail lookup), with initials as fallback.
"""
import argparse, json, os, re, shutil, subprocess, sys
from collections import Counter, defaultdict
from datetime import datetime, date, timedelta

SCHEMA = 1        # the shape of the file; raised when a viewer could no longer read an older one
MAX_PREVIOUS = 5  # --previous N: at most this many of the years before the one asked for
HEX = re.compile(r"[0-9a-f]{4,64}")
ISO = re.compile(r"\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d(Z|[+-]\d\d:\d\d)")


def git(repo, *args):
    # git writes UTF-8 (commit messages are re-encoded to it, path names are stored as bytes, mostly UTF-8): read it as
    # that on every system, and let a name in some other encoding come out with a replacement character, not a crash
    return subprocess.run(["git", "-C", repo, *args], check=True,
                          capture_output=True, text=True, encoding="utf-8", errors="replace").stdout


# What separates one thing from the next in git's output is NUL and nothing else: git refuses a NUL in a commit message, a
# name or a path, so no commit can write one into its own text. (Control characters like \x1e it does not refuse: a message
# that carried one, and a made-up record after it, used to be read as a second commit.)

def git_z(repo, *args, header=1):
    """Yield (fields, paths) for every commit of `git log -z <args>`, whose --format is "%x00" and then `header` fields
    separated by "%x00" ("%x00%cI%x00%an" is two). `fields` are the pieces right after the empty piece that starts a commit
    (the first is the committer date); `paths` are what the commit changed (--name-only), or its status and path in turn
    (--name-status).

    -z gives path names as they are, one per NUL. Without it git C-quotes a name with a space, a quote, a tab or a
    letter outside ASCII -- "caf\\303\\251/menu.txt" -- and the opening quote would become part of the folder.
    A commit starts with an empty piece (a path is never empty) and the committer date, which has a form and is checked."""
    fields_wanted = header
    pieces = git(repo, "log", "-z", *args).split("\0")
    i = 0
    while i < len(pieces):
        if pieces[i] != "":                           # not the start of a commit: nothing to read it as
            i += 1
            continue
        fields = pieces[i + 1:i + 1 + fields_wanted]
        j = k = i + 1 + fields_wanted
        while k < len(pieces) and pieces[k] != "":
            k += 1
        paths = pieces[j:k]
        if paths and paths[0].startswith("\n"):     # the line between the header and the first name
            paths[0] = paths[0][1:]
        if len(fields) == fields_wanted and ISO.fullmatch(fields[0]):
            yield fields, [p for p in paths if p]
        i = max(k, i + 1)


def parse_log(repo, *revargs):
    """Yield dicts for every commit matched by revargs."""
    fmt = "%x00".join(["%H", "%h", "%P", "%an", "%ae", "%cI", "%aI", "%s", "%b"])
    f = git(repo, "log", "-z", f"--format={fmt}", "--date=iso-strict", *revargs).split("\0")
    for i in range(0, len(f) - 8, 9):             # nine fields to a commit, each ended by a NUL (the last one by -z's)
        sha, short, parents, author, email, cdate, adate, subject, body = f[i:i + 9]
        sha = sha.lstrip("\n")
        # what has a form is in it, or it is not a commit: a record that is not is dropped, never trusted
        if not (HEX.fullmatch(sha) and HEX.fullmatch(short) and all(HEX.fullmatch(p) for p in parents.split()) and ISO.fullmatch(cdate)):
            continue
        yield {
            "sha": sha, "short": short, "parents": parents.split(),
            "author": author, "email": email, "date": cdate, "tz": tz_of(adate),
            "subject": subject, "body": body.rstrip("\n"),
        }


def tz_of(iso):
    """The UTC offset of an ISO date in minutes: "+02:00" is 120, "-05:00" is -300, "Z" is 0. `date` is the committer's
    clock (the merger's, for a merge); the author's date says where the work was written, which is what `tz` keeps."""
    m = re.search(r"([+-])(\d\d):?(\d\d)$", iso or "")
    return (1 if m.group(1) == "+" else -1) * (int(m.group(2)) * 60 + int(m.group(3))) if m else 0


CO_AUTHOR = re.compile(r"^co-authored-by:\s*(.*?)\s*<([^>]*)>\s*$", re.I | re.M)


def day_of(iso):
    return iso[:10]


def releases_of(mainline_all, tags, since, until):
    """The tags on the main line inside [since, until), each with the work behind it since the tag before."""
    releases = []
    prev_tag_idx = None
    # find the last tagged mainline commit before the period, for "since previous tag"
    for i, c in enumerate(mainline_all):
        if c["sha"] in tags:
            if c["date"] < since:
                prev_tag_idx = i
            elif since <= c["date"] < until:
                n_since = i - (prev_tag_idx if prev_tag_idx is not None else -1)
                span = mainline_all[(prev_tag_idx + 1) if prev_tag_idx is not None else 0:i + 1]
                auth = Counter(s["author"] for s in span)
                days = 0
                if prev_tag_idx is not None:
                    d0 = datetime.fromisoformat(mainline_all[prev_tag_idx]["date"])
                    d1 = datetime.fromisoformat(c["date"])
                    days = (d1 - d0).days
                releases.append({
                    "tag": tags[c["sha"]][0], "sha": c["sha"], "short": c["short"],
                    "date": c["date"], "author": c["author"], "email": c["email"],
                    "commitsSincePrev": n_since, "authors": len(auth),
                    "topAuthor": auth.most_common(1)[0][0] if auth else None,
                    "days": days,
                })
                prev_tag_idx = i
    return releases


def reverts_of(mainline, mainline_all, main_index):
    """The reverts among `mainline`, each with the commit it undoes when that one is on the main line too."""
    reverts = []
    for c in mainline:
        if not c["subject"].startswith("Revert "):
            continue
        m = re.search(r"reverts commit ([0-9a-f]{7,40})", c["body"])
        target = None
        gap_days = None
        gap_commits = None
        if m:
            target = m.group(1)
            for t in mainline_all:
                if t["sha"].startswith(target):
                    target = t
                    gap_days = (datetime.fromisoformat(c["date"]) - datetime.fromisoformat(t["date"])).days
                    gap_commits = main_index[c["sha"]] - main_index[t["sha"]] - 1
                    break
        reverts.append({
            "sha": c["sha"], "short": c["short"], "date": c["date"],
            "author": c["author"], "email": c["email"], "subject": c["subject"],
            "targetShort": target["short"] if isinstance(target, dict) else target,
            "targetSubject": target["subject"] if isinstance(target, dict) else None,
            "targetAuthor": target["author"] if isinstance(target, dict) else None,
            "gapDays": gap_days, "gapCommits": gap_commits,
        })
    return reverts


def moments_of(releases, reverts):
    """The three moments of a period: the release with the most work behind it, the cleanest revert, the last release."""
    moments = []
    if releases:
        big = max(releases, key=lambda r: r["commitsSincePrev"])
        moments.append({"kind": "release", "title": "the release with the most work behind it",
                        "date": big["date"], "ref": big["tag"]})
    clean = [r for r in reverts if r["gapCommits"] is not None]
    if clean:
        c = min(clean, key=lambda r: (r["gapCommits"], r["gapDays"]))
        moments.append({"kind": "revert", "title": "the cleanest revert",
                        "date": c["date"], "ref": c["short"]})
    if releases:
        last = releases[-1]
        if not moments or moments[0]["ref"] != last["tag"]:
            moments.append({"kind": "release", "title": "the last release of the year",
                            "date": last["date"], "ref": last["tag"]})
        else:
            moments[0]["title"] = "the last release, and the one with the most work behind it"
    moments.sort(key=lambda m: m["date"])
    return moments


def days_of(per_day, since, until):
    """Commits per day from `since` up to (not including) `until`, as a list."""
    days = []
    d = date.fromisoformat(since[:10])
    while d.isoformat() < until[:10]:
        days.append(per_day.get(d.isoformat(), 0))
        d += timedelta(days=1)
    return days


def previous_years(repo, year, count, remote, branch, tags, mainline_all, main_index, reach, per_day_all, per_month_all, month=None):
    """A short summary of each of the `count` years before `year`, oldest first, in the shape build.py gives every year of a
    comparison (`window.TIMELINE_ALL`): the days, the totals, the releases and the moments, the months of the whole history,
    and the sorted names of the people on the main line. The numbers are the ones a full run for that year writes -- the same
    functions on the same history. There is no codebase map, which is what makes a full year slow, and no branch list: only how
    many branches were merged, counted without reading them. A year with no commits on the branch is left out.

    With `month` (1 to 12) it is the same calendar month of each of those years instead -- 2023-03 and 2022-03 for March 2024 --
    each as a full run with --month for that month summarises it: its own days (February has 28 or 29), `period` the month,
    and a month with no commits is left out like a year with none."""
    months = sorted(per_month_all)
    out = []
    for y in range(year - count, year):
        if month:
            since = f"{y}-{month:02d}-01T00:00:00"
            until = f"{y + 1}-01-01T00:00:00" if month == 12 else f"{y}-{month + 1:02d}-01T00:00:00"
            period = f"{y}-{month:02d}"
        else:
            since, until, period = f"{y}-01-01T00:00:00", f"{y + 1}-01-01T00:00:00", str(y)
        per_day = Counter({d: n for d, n in per_day_all.items() if since[:10] <= d < until[:10]})
        if not per_day:
            continue
        mainline = [c for c in mainline_all if since <= c["date"] < until]
        merged = 0
        for c in mainline:
            if len(c["parents"]) > 1 and int(git(repo, "rev-list", "--count", c["parents"][1], f"^{c['parents'][0]}").strip()) > 0:
                merged += 1               # a merge brings in a branch when its second parent has commits the first has not
        releases = releases_of(mainline_all, tags, since, until)
        moments = moments_of(releases, reverts_of(mainline, mainline_all, main_index))
        out.append({
            "repo": remote, "branch": branch, "year": y, "period": period, "periodStart": since[:10],
            "totals": {"commitsThisYear": sum(per_day.values()), "releases": len(releases), "mergedBranches": merged,
                       "repoCommits": len(reach), "firstYear": int(months[0][:4]) if months else y,
                       "lastYear": int(months[-1][:4]) if months else y},
            "days": days_of(per_day, since, until), "releases": releases, "moments": moments,
            "repoMonths": [{"m": m, "n": per_month_all[m]} for m in months],
            "authors": sorted({c["author"] for c in mainline}), "mainlineCount": len(mainline),
        })
    return out


def main(argv=None):
    ap = argparse.ArgumentParser()
    ap.add_argument("repo")
    ap.add_argument("year", type=int)
    ap.add_argument("branch", nargs="?", default="HEAD")
    ap.add_argument("-o", "--out", default="timeline.json")
    ap.add_argument("--remote", default=None, help="e.g. github.com/mrdoob/three.js")
    ap.add_argument("--month", type=int, default=None, help="only this month (1-12) instead of the whole year")
    ap.add_argument("--previous", type=int, default=0, metavar="N",
                    help="also summarise each of the N years before this one (0 to 5; with --month, the same month of each): their days, releases and people, never the codebase map -- for the visualizations that set years side by side")
    ap.add_argument("--no-code", action="store_true", help="skip the codebase map (git log --name-only)")
    ap.add_argument("--anonymise", action="store_true", help="replace author names with 'Author 1..n' (busiest first), drop e-mail addresses and PR/branch labels")
    ap.add_argument("--no-emails", action="store_true", help="drop every e-mail address, keep the names (no avatars then)")
    ap.add_argument("--no-subjects", action="store_true", help="leave out the commit messages (and the co-authors named in them): the words, the report and the pair dance have less to tell")
    ap.add_argument("--no-timezones", action="store_true", help="every date in UTC, and no author time zones: nothing says where people worked, the night shift and daylight lose their clocks")
    ap.add_argument("--teams", default=None, help="JSON file {team: [author name, email or @domain, ...]} → stored as teams {author: team}")
    ap.add_argument("--code-depth", type=int, default=2, help="folder depth for the codebase map")
    ap.add_argument("--max-branches", type=int, default=60,
                    help="how many side branches (largest first) get a drawn lane")
    a = ap.parse_args(argv)
    if not 0 <= a.previous <= MAX_PREVIOUS:
        ap.error(f"--previous takes a number from 0 to {MAX_PREVIOUS}: how many of the years before {a.year} to summarise as well")
    if a.month is not None and not 1 <= a.month <= 12:
        ap.error("--month takes a number from 1 to 12")
    repo, year, branch = a.repo, a.year, a.branch
    if branch.startswith("-"):      # git would read it as an option (a branch named "--output=..." is one a remote can make)
        ap.error("a branch name never starts with '-'")

    remote = a.remote
    if not remote:
        try:
            url = git(repo, "remote", "get-url", "origin").strip()
            m = re.search(r"([\w.-]+\.\w+)[:/]([\w.-]+)/([\w.-]+?)(?:\.git)?$", url)
            if m:
                remote = f"{m.group(1)}/{m.group(2)}/{m.group(3)}"
        except subprocess.CalledProcessError:
            pass
    remote = remote or "local repository"

    if a.month:
        m = a.month
        since = f"{year}-{m:02d}-01T00:00:00"
        until = f"{year + 1}-01-01T00:00:00" if m == 12 else f"{year}-{m + 1:02d}-01T00:00:00"
        period = f"{year}-{m:02d}"
    else:
        since, until = f"{year}-01-01T00:00:00", f"{year + 1}-01-01T00:00:00"
        period = str(year)

    # ---- tags → commit map -------------------------------------------------
    tags = {}
    for line in git(repo, "for-each-ref", "refs/tags",
                    "--format=%(refname:short)%00%(objectname)%00%(*objectname)").splitlines():
        name, obj, peeled = (line.split("\0") + ["", ""])[:3]
        if HEX.fullmatch(peeled or obj):                  # a ref name has no NUL and no line break: a line is one tag
            tags.setdefault(peeled or obj, []).append(name)

    # ---- main line (first-parent walk) for the whole branch ---------------
    mainline_all = list(parse_log(repo, "--first-parent", branch))
    mainline_all.reverse()  # oldest → newest
    mainline = [c for c in mainline_all if since <= c["date"] < until]
    main_index = {c["sha"]: i for i, c in enumerate(mainline_all)}

    # ---- every commit reachable from the branch (for pulse + repo summary) --
    reach = list(parse_log(repo, branch))
    per_day_all = Counter()
    per_month_all = Counter()
    for c in reach:
        d = day_of(c["date"])
        per_month_all[d[:7]] += 1
        per_day_all[d] += 1
    per_day = Counter({d: n for d, n in per_day_all.items() if since[:10] <= d < until[:10]})

    # ---- side branches: for each merge on the main line ------------------
    branches = []
    for c in mainline:
        if len(c["parents"]) < 2:
            continue
        p1, p2 = c["parents"][0], c["parents"][1]
        side = list(parse_log(repo, p2, f"^{p1}"))
        side.reverse()
        if not side:
            continue
        m = re.match(r"Merge pull request #(\d+) from (\S+)", c["subject"])
        m2 = re.match(r"Merge (?:remote-tracking )?branch '([^']+)'", c["subject"])
        label = None
        pr = None
        if m:
            pr, label = int(m.group(1)), m.group(2)
        elif m2:
            label = m2.group(1)
        authors = Counter(s["author"] for s in side)
        branches.append({
            "merge": c["sha"],
            "mergeDate": c["date"],
            "start": side[0]["date"],
            "count": len(side),
            "pr": pr,
            "label": label,
            "author": authors.most_common(1)[0][0],
            "email": next(s["email"] for s in side if s["author"] == authors.most_common(1)[0][0]),
            "commits": [{"date": s["date"], "tz": s["tz"], "short": s["short"], "subject": s["subject"],
                         "author": s["author"], "email": s["email"]} for s in side[:400]],
        })
    branches.sort(key=lambda b: -b["count"])
    for i, b in enumerate(branches):
        b["lane"] = i < a.max_branches  # the ones that "gave the year its shape"

    # ---- releases (tags on the main line), reverts, and the three moments ------
    releases = releases_of(mainline_all, tags, since, until)
    reverts = reverts_of(mainline, mainline_all, main_index)
    moments = moments_of(releases, reverts)

    # ---- code map: which directories were touched, per day (tree objects only, works on blobless clones)
    code = None
    if not a.no_code:
        try:
            depth = a.code_depth
            dirs = {}          # path -> {"touches": n, "byDay": {d: n}, "files": set(), "authors": Counter()}
            files = Counter()
            pairs = {}         # (dir a, dir b) -> {"count": n, "byDay": Counter}
            ext_touch, ext_week = Counter(), {}          # file extension -> touches, and {week: touches}
            born, gone = {}, {}                          # folder -> Counter(day): files added, files deleted
            weave = {}         # top-level folder -> {week: Counter(extension)}: which kind of file changed there that week
            # --no-renames: finding a rename compares file contents, which a blobless clone would fetch
            for (cdate, cauthor), paths in git_z(repo, "--name-only", "--no-renames", "--format=%x00%cI%x00%an",
                                                 f"--since={since}", f"--until={until}", branch, header=2):
                if not (since <= cdate < until):
                    continue
                dday = (date.fromisoformat(cdate[:10]) - date.fromisoformat(since[:10])).days
                touched = set()
                for path in sorted(set(paths)):      # sorted: folders that tie keep one order, whatever the hash seed
                    parts = path.split("/")
                    e, week = ext_of(path), dday // 7
                    ext_touch[e] += 1
                    ext_week.setdefault(e, Counter())[week] += 1
                    weave.setdefault(parts[0] if len(parts) > 1 else "(root)", {}).setdefault(week, Counter())[e] += 1
                    key = "/".join(parts[:-1][:depth]) or "(root)"
                    node = dirs.setdefault(key, {"touches": 0, "byDay": Counter(), "files": set(), "authors": Counter()})
                    node["touches"] += 1
                    node["byDay"][dday] += 1
                    node["files"].add(path)
                    node["authors"][cauthor] += 1
                    files[path] += 1
                    touched.add(key)
                for x in sorted(touched):
                    for y in sorted(touched):
                        if x < y:
                            pr = pairs.setdefault((x, y), {"count": 0, "byDay": Counter()})
                            pr["count"] += 1
                            pr["byDay"][dday] += 1
            # files born and gone: an added or a deleted path is a difference between two trees, so --name-status needs no
            # blob; --no-renames keeps it that way (rename detection compares contents), and a rename is then a birth and a death
            for (cdate,), fields in git_z(repo, "--name-status", "--no-renames", "--diff-filter=AD", "--format=%x00%cI",
                                          f"--since={since}", f"--until={until}", branch):
                if not (since <= cdate < until):
                    continue
                dday = (date.fromisoformat(cdate[:10]) - date.fromisoformat(since[:10])).days
                for status, path in zip(fields[::2], fields[1::2]):      # with -z: the status, then the name
                    if status not in ("A", "D"):
                        continue
                    key = "/".join(path.split("/")[:-1][:depth]) or "(root)"
                    (born if status == "A" else gone).setdefault(key, Counter())[dday] += 1
            # tree size at the end of the period, for the footprint
            tip = mainline[-1]["sha"] if mainline else branch
            size, ext_files, top_files = Counter(), Counter(), Counter()
            for path in filter(None, git(repo, "ls-tree", "-r", "-z", "--name-only", tip).split("\0")):
                parts = path.split("/")
                size["/".join(parts[:-1][:depth]) or "(root)"] += 1
                ext_files[ext_of(path)] += 1
                top_files[parts[0] if len(parts) > 1 else "(root)"] += 1
            code = {
                "depth": depth,
                "dirs": [{"path": k, "touches": v["touches"], "filesTouched": len(v["files"]), "size": size.get(k, 0),
                          "authors": [{"name": n, "n": c} for n, c in v["authors"].most_common(3)],
                          "byDay": {str(d): n for d, n in sorted(v["byDay"].items())}} for k, v in sorted(dirs.items(), key=lambda kv: -kv[1]["touches"])],
                "pairs": [{"a": k[0], "b": k[1], "count": v["count"], "byDay": {str(d): n for d, n in sorted(v["byDay"].items())}}
                          for k, v in sorted(pairs.items(), key=lambda kv: -kv[1]["count"])[:48]],
                "untouched": [{"path": k, "size": n} for k, n in size.items() if k not in dirs],
                "topFiles": [{"path": p, "touches": n} for p, n in files.most_common(40)],
                "filesTotal": sum(size.values()),
                # what kinds of file: per extension (from the names only, never the contents), and per top-level
                # folder and week the extension that changed most there -- the loom's threads, the core's strata
                "ext": [{"ext": e, "files": ext_files.get(e, 0), "touches": ext_touch.get(e, 0),
                         "byWeek": {str(w): n for w, n in sorted(ext_week.get(e, Counter()).items())}}
                        for e in sorted(set(ext_touch) | set(ext_files), key=lambda e: (-ext_touch.get(e, 0), -ext_files.get(e, 0), e))[:40]],
                "born": [{"path": k, "n": sum(c.values()), "byDay": {str(d): n for d, n in sorted(c.items())}}
                         for k, c in sorted(born.items(), key=lambda kv: -sum(kv[1].values()))[:80]],
                "gone": [{"path": k, "n": sum(c.values()), "byDay": {str(d): n for d, n in sorted(c.items())}}
                         for k, c in sorted(gone.items(), key=lambda kv: -sum(kv[1].values()))[:80]],
                "weave": [{"dir": k, "files": top_files.get(k, 0),
                           "byWeek": {str(w): {"ext": min(c, key=lambda e: (-c[e], e)), "n": sum(c.values())} for w, c in sorted(weave[k].items())}}
                          for k in sorted(weave, key=lambda k: -sum(sum(c.values()) for c in weave[k].values()))[:24]],
            }
        except subprocess.CalledProcessError as e:
            print(f"code map overgeslagen: {e}", file=sys.stderr)

    # ---- people who signed a commit together (Co-authored-by trailers) ----------
    # A trailer names a co-author as "Name <e-mail>"; the e-mail finds the name that person commits under, so one person is
    # one name. Only the names are kept, never the addresses, and only with the messages: a trailer is part of one.
    coauthors = None
    if not a.no_subjects:
        known = {}
        for c in reach:
            known.setdefault(c["email"].lower(), Counter())[c["author"]] += 1
        pairs = {}
        for c in reach:
            if not (since <= c["date"] < until):
                continue
            people = {c["author"]}
            for name, email in CO_AUTHOR.findall(c["body"]):
                by = known.get(email.strip().lower())
                people.add(by.most_common(1)[0][0] if by else (name.strip() or email.strip()))
            dday = (date.fromisoformat(c["date"][:10]) - date.fromisoformat(since[:10])).days
            for x in sorted(people):
                for y in sorted(people):
                    if x < y:
                        pr = pairs.setdefault((x, y), Counter())
                        pr[dday] += 1
        coauthors = [{"a": x, "b": y, "n": sum(c.values()), "byDay": {str(d): n for d, n in sorted(c.items())}}
                     for (x, y), c in sorted(pairs.items(), key=lambda kv: -sum(kv[1].values()))[:300]]

    # ---- output --------------------------------------------------------------
    days = days_of(per_day, since, until)

    months = sorted(per_month_all)
    branch_name = branch if branch != "HEAD" else git(repo, "rev-parse", "--abbrev-ref", "HEAD").strip()
    out = {
        "schema": SCHEMA,
        "repo": remote, "branch": branch_name,
        "year": year, "period": period, "periodStart": since[:10],
        "totals": {"commitsThisYear": sum(per_day.values()), "releases": len(releases),
                   "mergedBranches": len(branches), "repoCommits": len(reach),
                   "firstYear": int(months[0][:4]) if months else year,
                   "lastYear": int(months[-1][:4]) if months else year},
        "mainline": [{"sha": c["sha"], "short": c["short"], "date": c["date"], "tz": c["tz"], "author": c["author"],
                      "email": c["email"], "subject": c["subject"], "merge": len(c["parents"]) > 1,
                      "tags": tags.get(c["sha"], [])} for c in mainline],
        "branches": branches,
        "releases": releases,
        "reverts": reverts,
        "moments": moments,
        "days": days,
        "repoMonths": [{"m": m, "n": per_month_all[m]} for m in months],
        "code": code,
        "coauthors": coauthors,
        "teams": teams_for(a.teams, reach),
    }
    if a.previous:                                   # the years before (or their same month), summarised (no codebase map): absent unless asked for
        out["previous"] = previous_years(repo, year, a.previous, remote, branch_name, tags, mainline_all, main_index, reach,
                                         per_day_all, per_month_all, a.month)
    redact(out, names=a.anonymise, emails=a.no_emails, subjects=a.no_subjects, code=a.no_code, timezones=a.no_timezones)
    with open(a.out, "w") as f:
        json.dump(out, f)
    print(f"{a.out}: {len(mainline)} main-line commits, {len(branches)} merged branches, "
          f"{len(releases)} releases, {len(reverts)} reverts, {sum(per_day.values())} commits in {period}",
          file=sys.stderr)


def ext_of(path: str) -> str:
    """A file's extension from its name alone, lower case: "js" for three.module.js, "ts" for x.d.ts, "" for a
    Makefile, a .gitignore or anything whose ending is not a plain short word."""
    stem, dot, suffix = path.rsplit("/", 1)[-1].rpartition(".")
    return suffix.lower() if dot and stem and 0 < len(suffix) <= 10 and suffix.isalnum() else ""


def utc(out):
    """Every date as the same moment in UTC, and no `tz`: what is left says when, never where."""
    from datetime import timezone
    def walk(o):
        if isinstance(o, dict):
            for k, v in o.items():
                if k in ("date", "mergeDate", "start") and isinstance(v, str) and "T" in v:
                    o[k] = datetime.fromisoformat(v.replace("Z", "+00:00")).astimezone(timezone.utc).isoformat()
                elif k == "tz":
                    o[k] = None
                else:
                    walk(v)
        elif isinstance(o, list):
            for v in o:
                walk(v)
    walk(out)


def drop(out, *keys):
    """Blank these fields wherever the history has them: commits, branches, releases, reverts, and the years before."""
    def walk(o):
        if isinstance(o, dict):
            for k in keys:
                if k in o:
                    o[k] = "" if isinstance(o[k], str) else None
            for v in o.values():
                walk(v)
        elif isinstance(o, list):
            for v in o:
                walk(v)
    for part in ("mainline", "branches", "releases", "reverts", "previous"):
        walk(out.get(part))


ADDRESS = re.compile(r"^\s*([^@\s]+)@[^@\s]+\.[^@\s]+\s*$")


def unaddress(out):
    """Some people commit with their e-mail address as their name. Without e-mails, keep the part before the @."""
    def name(v):
        m = ADDRESS.match(v) if isinstance(v, str) else None
        return m.group(1) if m else v
    def walk(o):
        if isinstance(o, dict):
            for k, v in o.items():
                if k in ("author", "topAuthor", "targetAuthor", "name"):
                    o[k] = name(v)
                elif k == "authors" and isinstance(v, list) and all(isinstance(x, str) for x in v):
                    o[k] = sorted({name(x) for x in v})      # the people of an earlier year (`previous`): the same names, as a set again
                else:
                    walk(v)
        elif isinstance(o, list):
            for v in o:
                walk(v)
    walk(out)
    for p in out.get("coauthors") or []:
        p["a"], p["b"] = name(p["a"]), name(p["b"])
    if out.get("teams"):
        out["teams"] = {name(k): v for k, v in out["teams"].items()}


LEAVE_OUT = ("names", "emails", "subjects", "code", "timezones")


def redact(out, names=False, emails=False, subjects=False, code=False, timezones=False):
    """Leave out what was asked, in place, and add it to `redacted`, so whoever reads the file (a viewer, the
    site) knows it is missing on purpose. The options of this script call it, and so does the site for a file
    that left out less than the person now wants -- one set of rules for both."""
    if names and not out.get("anonymised"):
        anonymise(out)
    if emails:
        drop(out, "email")
        unaddress(out)
    if subjects:
        drop(out, "subject", "targetSubject")
        out["coauthors"] = None                     # a trailer is part of a message
    if code:
        out["code"] = None
    if timezones:
        utc(out)
    done = set(out.get("redacted") or []) | {what for what, on in (("names", names), ("emails", names or emails),
                                                                    ("subjects", subjects), ("code", code), ("timezones", timezones)) if on}
    out["redacted"] = [what for what in LEAVE_OUT if what in done]


def anonymise(out):
    """Strip identities in place: names → 'Author n' ordered by activity, e-mails removed, branch labels reduced to their PR number
    (or 'branch'), subjects kept (they are the story), teams kept but re-keyed."""
    from collections import Counter
    count = Counter(c["author"] for c in out["mainline"])
    for b in out["branches"]:
        for c in b["commits"]:
            count[c["author"]] += 1
    names = {name: f"Author {i + 1}" for i, (name, _) in enumerate(count.most_common())}
    alias = lambda n: names.setdefault(n, f"Author {len(names) + 1}")
    def strip(c):
        c["author"] = alias(c["author"]); c.pop("email", None)
    for c in out["mainline"]:
        strip(c)
    for b in out["branches"]:
        strip(b); b["label"] = f"#{b['pr']}" if b.get("pr") else "branch"
        for c in b["commits"]:
            strip(c)
    for r in out["releases"]:
        strip(r)
        if r.get("topAuthor"):
            r["topAuthor"] = alias(r["topAuthor"])
    for r in out["reverts"]:
        strip(r)
        if r.get("targetAuthor"):
            r["targetAuthor"] = alias(r["targetAuthor"])
    if out.get("code"):
        for d in out["code"]["dirs"]:
            for x in d.get("authors", []):
                x["name"] = alias(x["name"])
    for p in out.get("coauthors") or []:
        p["a"], p["b"] = alias(p["a"]), alias(p["b"])
    if out.get("teams"):
        out["teams"] = {alias(k): v for k, v in out["teams"].items()}
    anonymise_previous(out, alias)
    out["anonymised"] = True


def anonymise_previous(out, alias):
    """The years before (`previous`) with the same numbering as the rest of the file: a person who is on the main line of the
    year itself keeps the number they have there, and one who only appears in an earlier year gets the next numbers. Those are
    given by how many of the earlier years the person is in (most first) and then by a hash of the name, never by the alphabet:
    the order of the names must not leak through the numbers, and the summaries hold no commit counts to rank them by."""
    import hashlib
    summaries = out.get("previous") or []
    present = Counter(n for p in summaries for n in set(p.get("authors") or []) | {r.get(k) for r in p.get("releases", []) for k in ("author", "topAuthor")} - {None})
    for n in sorted(present, key=lambda n: (-present[n], hashlib.sha256(n.encode("utf-8")).hexdigest())):
        alias(n)
    for p in summaries:
        p["authors"] = sorted({alias(n) for n in p.get("authors") or []})
        for r in p.get("releases", []):
            r["author"] = alias(r["author"]); r.pop("email", None)
            if r.get("topAuthor"):
                r["topAuthor"] = alias(r["topAuthor"])


def teams_for(path, commits):
    """Map every author seen in the history to a team, from a JSON file {team: [patterns]}.
    A pattern is an exact author name, an exact e-mail, or '@domain' matching every e-mail at that domain. Unmatched authors get no entry."""
    if not path:
        return None
    with open(path) as f:
        spec = json.load(f)
    out = {}
    for c in commits:
        name, email = c["author"], (c.get("email") or "").lower()
        for team, pats in spec.items():
            for pat in pats:
                p = str(pat).lower()
                if (p.startswith("@") and email.endswith(p)) or p == name.lower() or p == email:
                    out[name] = team
                    break
            if name in out:
                break
    return out


def ask(question, default=""):
    shown = f" [{default}]" if default != "" else ""
    try:
        answer = input(f"{question}{shown}: ").strip()
    except EOFError:
        raise SystemExit("\nstopped: no answer") from None
    return answer or str(default)


def yes(question, default):
    while True:
        answer = ask(question + (" (Y/n)" if default else " (y/N)")).lower()
        if not answer:
            return default
        if answer in ("y", "yes", "j", "ja"):
            return True
        if answer in ("n", "no", "nee"):
            return False
        print("  y or n, please")


def interactive():
    """No options given: ask for them, one at a time, with a sensible answer already filled in. Standard
    library only, like the rest of this file, so all it needs is Python 3 and git."""
    print("CommitMotion -- turn a year of a git repository into a data file.\n"
          "Nothing leaves this computer: the file is written here, you can read it, and you decide what to do with it.\n")
    if not shutil.which("git"):
        raise SystemExit("git is not installed or not on the PATH; it is the one thing this script needs besides Python.")
    while True:
        repo = ask("Repository folder", ".")
        try:
            top = git(repo, "rev-parse", "--show-toplevel").strip() or repo
            break
        except (subprocess.CalledProcessError, FileNotFoundError, NotADirectoryError):
            try:
                git(repo, "rev-parse", "--git-dir"); top = repo; break          # a bare clone has no top level
            except (subprocess.CalledProcessError, FileNotFoundError, NotADirectoryError):
                print(f"  {repo} is not a git repository")
    from datetime import date as _date
    while True:
        year = ask("Year", _date.today().year - 1)
        if re.fullmatch(r"(19|20)\d\d", year):
            break
        print("  a year like 2025, please")
    while True:
        month = ask("Only one month? 1-12, or empty for the whole year", "")
        if month == "" or (month.isdigit() and 1 <= int(month) <= 12):
            break
        print("  a number from 1 to 12, or nothing")
    try:
        current = git(top, "rev-parse", "--abbrev-ref", "HEAD").strip()
    except subprocess.CalledProcessError:
        current = "HEAD"
    branch = ask("Branch", current)
    before = ""
    since = f"{year}-{int(month):02d}-01" if month else f"{year}-01-01"
    try:                                                    # only worth asking when there is history before the period to set beside it
        has_before = bool(git(top, "log", "-1", "--format=%H", f"--until={since}T00:00:00", branch).strip())
    except subprocess.CalledProcessError:
        has_before = True
    while has_before:
        before = ask(f"Also set it beside the years before? How many, 0-{MAX_PREVIOUS}, or empty for none"
                     + (" (the same month of each)" if month else ""), "")
        if before == "" or (before.isdigit() and 0 <= int(before) <= MAX_PREVIOUS):
            break
        print(f"  a number from 0 to {MAX_PREVIOUS}, or nothing")
    try:
        url = git(top, "remote", "get-url", "origin").strip()
        m = re.search(r"([\w.-]+\.\w+)[:/]([\w.-]+)/([\w.-]+?)(?:\.git)?$", url)
        remote = f"{m.group(1)}/{m.group(2)}/{m.group(3)}" if m else os.path.basename(os.path.abspath(top))
    except subprocess.CalledProcessError:
        remote = os.path.basename(os.path.abspath(top))
    remote = ask("Name to show on the pictures", remote)
    print("\nThe file holds the history, never the contents of your files: commit messages, author names and\n"
          "e-mail addresses, the time zones of the dates, branch and tag names, and folder and file names.\n"
          "Leave out what you would rather keep:")
    emails = yes("  Keep e-mail addresses? They are only used for avatars", False)
    names = yes("  Keep author names? Otherwise they become Author 1, Author 2, ...", True)
    subjects = yes("  Keep commit messages? The words and the report are made of them (and co-authors named in them)", True)
    code = yes("  Keep folder and file names? The code map, the loom and the core sample are made of them", True)
    zones = yes("  Keep time zones? They say at what hour, on whose clock -- and so roughly where", True)
    out = ask("\nWrite the file to", f"commitmotion-{year}{'-%02d' % int(month) if month else ''}.json")
    argv = [top, year, branch, "-o", out, "--remote", remote]
    argv += (["--month", month] if month else []) + (["--previous", before] if before not in ("", "0") else []) + ([] if emails else ["--no-emails"]) + ([] if names else ["--anonymise"])
    argv += ([] if subjects else ["--no-subjects"]) + ([] if code else ["--no-code"]) + ([] if zones else ["--no-timezones"])
    print("\nThe same, without questions next time:\n  python3 extract.py " + " ".join(
        a if re.fullmatch(r"[\w./:@-]+", a) else '"' + a.replace('"', '\\"') + '"' for a in argv) + "\n")
    return argv


if __name__ == "__main__":
    if len(sys.argv) == 1:
        argv = interactive()
        main(argv)
        size = os.path.getsize(argv[argv.index("-o") + 1])
        print(f"\nDone: {argv[argv.index('-o') + 1]} ({size / 1e6:.1f} MB). It is plain JSON: open it in any editor to see\n"
              "exactly what is in it. Upload it at https://commitmotion.com/private to see your year.")
    else:
        main()
