# Builds the page data from the git clones.
# usage: python3 build_data.py <repos_dir> <tzfull_dir> <out_dir>
import subprocess, json, re, sys, collections, datetime, os

REPOS, TZ, OUT = sys.argv[1], sys.argv[2], sys.argv[3]
BOT = re.compile(r'bot|dependabot|github-actions|renovate|weblate', re.I)
END = datetime.date(2026, 10, 7)
START12 = datetime.date(2025, 10, 7)

# one person can appear under several spellings
ALIAS = {'jiat75': 'Jia Tan', 'Jia Cheong Tan': 'Jia Tan', 'djm@openbsd.org': 'Damien Miller',
         'Daniel Stenberg': 'Daniel Stenberg'}


def commits(repo, since=None, until=None):
    a = ['git', '-C', os.path.join(REPOS, repo + '.git'), 'log', '--no-merges',
         '--format=%ad\x1f%an\x1f%s', '--date=short']
    if since: a.append('--since=' + since)
    if until: a.append('--until=' + until)
    out = subprocess.run(a, capture_output=True, text=True, errors='replace').stdout
    rows = []
    for l in out.splitlines():
        p = l.split('\x1f')
        if len(p) != 3 or BOT.search(p[1]): continue
        rows.append((p[0], ALIAS.get(p[1], p[1]), p[2][:90]))
    rows.reverse()  # oldest first
    return rows


def wall(repo, since, until=None, top=3, anonymous=False, subjects=True):
    # git --since filters on commit date; keep only changes WRITTEN inside the window
    rows = [r for r in commits(repo, since, until) if r[0] >= since and (not until or r[0] <= until)]
    cnt = collections.Counter(r[1] for r in rows)
    people = [n for n, _ in cnt.most_common()]
    named = people[:top]
    out = dict(people=[dict(name=('Developer ' + 'ABCDEFG'[i] if anonymous else n), n=cnt[n])
                       for i, n in enumerate(people)],
               named=len(named), total=len(rows),
               c=[[r[0], people.index(r[1]), r[2][:70] if (subjects and not anonymous) else ''] for r in rows])
    return out


def per_year(repo, y0, y1, top=4):
    rows = commits(repo, f'{y0}-01-01', f'{y1}-12-31')
    by = collections.defaultdict(collections.Counter)
    for d, a, _ in rows: by[d[:4]][a] += 1
    allc = collections.Counter(r[1] for r in rows)
    out = []
    for y in sorted(by):
        c = by[y]
        out.append(dict(y=int(y), total=sum(c.values()), core=sum(1 for v in c.values() if v >= 10),
                        people=[[n, v] for n, v in c.most_common()]))
    return out


data = {}
data['tz'] = wall('tz', '2025-10-07', top=2)
data['xz'] = wall('xz', '2021-01-01', top=2)
data['corejs'] = wall('corejs', '2019-06-01', '2021-03-31', top=2)
data['sudo'] = wall('sudo', '2008-01-01', '2018-12-31', top=1, subjects=False)
data['libjpeg'] = wall('libjpeg', '2025-10-07', top=1)
data['sqlite'] = wall('sqlite', '2025-10-07', top=4, anonymous=True)
data['zlib'] = wall('zlib', '2025-10-07', top=2)
data['harfbuzz'] = wall('harfbuzz', '2025-10-07', top=3)
data['curl'] = wall('curl', '2025-10-07', top=11)
data['libxml2'] = wall('libxml2', '2025-06-01', top=3)
data['nghttp2'] = wall('nghttp2', '2025-10-07', top=1)
data['openssl_years'] = per_year('openssl', 2008, 2026)
data['xz_years'] = per_year('xz', 2008, 2026)
for k, v in data.items():
    if isinstance(v, dict):
        v['c'] = [x for x in v['c']]
json.dump(data, open(os.path.join(OUT, 'walls.json'), 'w'), separators=(',', ':'), ensure_ascii=False)
print('walls:', {k: (v['total'] if isinstance(v, dict) else len(v)) for k, v in data.items()})

# ---- tz: blame every Zone block, so a reader sees who last touched each line of their zone
files = ['africa', 'antarctica', 'asia', 'australasia', 'europe', 'northamerica', 'southamerica', 'etcetera', 'backward']
zones, links = {}, {}
for f in files:
    out = subprocess.run(['git', '-C', TZ, 'blame', '--line-porcelain', 'HEAD', '--', f],
                         capture_output=True, text=True, errors='replace').stdout
    lines, cur = [], {}
    for l in out.splitlines():
        if l.startswith('\t'):
            lines.append((l[1:], cur.get('author', '?'), cur.get('date', '')))
        elif l.startswith('author '):
            cur['author'] = l[7:]
        elif l.startswith('author-time '):
            cur['date'] = datetime.datetime.fromtimestamp(int(l[12:]), datetime.UTC).strftime('%Y-%m-%d')
    zone = None
    for text, auth, date in lines:
        code = text.split('#', 1)[0].rstrip()
        if text.startswith('Zone'):
            zone = text.split()[1]
            zones[zone] = [[text.rstrip(), auth, date]]
        elif text.startswith('Link'):
            p = code.split()
            if len(p) >= 3: links[p[2]] = p[1]
            zone = None
        elif zone and code.strip() and (text[0] in ' \t') and not text.strip().startswith('#'):
            zones[zone].append([text.rstrip(), auth, date])
        elif code.strip():
            zone = None
json.dump(dict(zones=zones, links=links), open(os.path.join(OUT, 'zones.json'), 'w'), separators=(',', ':'), ensure_ascii=False)
print('zones:', len(zones), 'links:', len(links))
