#!/usr/bin/env python3
"""Taken apart #1: what the most-starred shadcn-ui repos actually default to.

For the top N GitHub repos tagged `shadcn-ui` (by stars), read the app's font setup and its
`--primary` colour, and classify both. Every number in the write-up comes from results.json.
Method, kept honest:
  - fonts: `next/font/google` and `next/font/local` imports, Google Fonts @import URLs, and
    `font-family` / Tailwind `fontFamily.sans` declarations, in layout, _app, tailwind config and CSS.
  - primary: the first `--primary:` in the root CSS (light theme), parsed as HSL or OKLCH.
    Hue buckets: neutral (saturation < 12% or chroma < 0.05), violet/indigo (HSL 240-300,
    OKLCH 265-320), blue (HSL 200-240, OKLCH 225-265), other.
  - repos where neither signal is found are counted as "unreadable", never guessed.
Run: python3 scan_shadcn.py [N]
"""
import base64, json, re, subprocess, sys
from collections import Counter
from pathlib import Path

N = int(sys.argv[1]) if len(sys.argv) > 1 else 150
OUT = Path(__file__).with_name("results.json")
FONT_FILES = re.compile(r"(^|/)(app/layout\.(t|j)sx?|pages/_app\.(t|j)sx?|tailwind\.config\.(t|j|m|c)?s|"
                        r"globals?\.css|index\.css|app\.css|global\.css|styles\.css)$")
SKIP_DIRS = re.compile(r"(node_modules|/dist/|/build/|/\.next/|/examples?/|/test|/docs?/|/apps/www/)")


def gh(path):
    r = subprocess.run(["gh", "api", path], capture_output=True, text=True)
    return json.loads(r.stdout) if r.returncode == 0 and r.stdout.strip() else None


def repos():
    out, page = [], 1
    while len(out) < N:
        d = gh(f"search/repositories?q=topic:shadcn-ui&sort=stars&order=desc&per_page=100&page={page}")
        if not d or not d.get("items"):
            break
        out += [(i["full_name"], i["stargazers_count"], i["default_branch"]) for i in d["items"]]
        page += 1
    return out[:N]


FONT_NAMES = r"(Inter|Geist(?: Mono)?|Roboto|Poppins|Manrope|DM Sans|Plus Jakarta Sans|Outfit|Montserrat|" \
             r"Open Sans|Lato|Nunito|IBM Plex Sans|Space Grotesk|Figtree|Work Sans|Satoshi|Cal Sans|Onest|" \
             r"Noto Sans|Rubik|Urbanist|Lexend|Sora|Mona Sans|Instrument Sans|Source Sans 3|Raleway)"


def fonts_in(text):
    found = set()
    for m in re.finditer(r"import\s*\{([^}]+)\}\s*from\s*['\"]next/font/google['\"]", text):
        found |= {x.strip().split(" as ")[0].replace("_", " ") for x in m.group(1).split(",") if x.strip()}
    if re.search(r"from\s*['\"]geist/font", text):
        found.add("Geist")
    for m in re.finditer(r"fonts\.googleapis\.com/css2?\?family=([^&:'\")]+)", text):
        found.add(m.group(1).replace("+", " "))
    for m in re.finditer(r"font-family:\s*['\"]?" + FONT_NAMES, text):
        found.add(m.group(1))
    for m in re.finditer(r"sans:\s*\[[^\]]*?['\"]" + FONT_NAMES, text):
        found.add(m.group(1))
    return {f for f in found if re.fullmatch(FONT_NAMES, f)} or {f for f in found if f and len(f) < 30}


def primary_in(text):
    m = re.search(r"--primary:\s*([^;]+);", text)
    if not m:
        return None
    v = m.group(1).strip()
    o = re.search(r"oklch\(\s*([\d.]+%?)\s+([\d.]+)\s+([\d.]+)", v)
    if o:
        c, h = float(o.group(2)), float(o.group(3))
        if c < 0.05:
            return ("neutral", v)
        return ("violet/indigo" if 265 <= h <= 320 else "blue" if 225 <= h < 265 else "other", v)
    hs = re.search(r"(?:hsl\()?\s*([\d.]+)(?:deg)?[\s,]+([\d.]+)%[\s,]+([\d.]+)%", v)
    if hs:
        h, s = float(hs.group(1)), float(hs.group(2))
        if s < 12:
            return ("neutral", v)
        return ("violet/indigo" if 240 <= h <= 300 else "blue" if 200 <= h < 240 else "other", v)
    if re.match(r"#[0-9a-fA-F]{6}", v):
        r, g, b = (int(v[i:i + 2], 16) for i in (1, 3, 5))
        mx, mn = max(r, g, b), min(r, g, b)
        if mx - mn < 25:
            return ("neutral", v)
        import colorsys
        h = colorsys.rgb_to_hls(r / 255, g / 255, b / 255)[0] * 360
        return ("violet/indigo" if 240 <= h <= 300 else "blue" if 200 <= h < 240 else "other", v)
    return ("unparsed", v)


def scan(full, branch):
    tree = gh(f"repos/{full}/git/trees/{branch}?recursive=1")
    if not tree:
        return None
    paths = [t["path"] for t in tree.get("tree", []) if t["type"] == "blob"
             and FONT_FILES.search(t["path"]) and not SKIP_DIRS.search("/" + t["path"])]
    paths = sorted(paths, key=len)[:8]
    fonts, primary = set(), None
    for p in paths:
        f = gh(f"repos/{full}/contents/{p}?ref={branch}")
        if not f or "content" not in f:
            continue
        text = base64.b64decode(f["content"]).decode("utf-8", "replace")
        fonts |= fonts_in(text)
        if primary is None and p.endswith(".css"):
            primary = primary_in(text)
    return {"fonts": sorted(fonts), "primary": primary, "files": paths}


def main():
    rows = []
    for full, stars, branch in repos():
        r = scan(full, branch)
        rows.append({"repo": full, "stars": stars, **(r or {"fonts": [], "primary": None, "files": []})})
        print(full, stars, (r or {}).get("fonts"), (r or {}).get("primary"), flush=True)
    OUT.write_text(json.dumps(rows, indent=1))
    font_first = Counter((r["fonts"][0] if len(r["fonts"]) == 1 else ("+".join(r["fonts"][:2]) if r["fonts"] else "none found"))
                         for r in rows)
    any_font = Counter(f for r in rows for f in set(r["fonts"]))
    prim = Counter((r["primary"] or ("none found", ""))[0] for r in rows)
    print("\nrepos", len(rows))
    print("font mentioned (any):", any_font.most_common(12))
    print("primary bucket:", prim.most_common())


if __name__ == "__main__":
    main()
