"""Classify the PSG directory snapshot. Every figure published in the article is reproducible from this script and the public CSV.
Reads psg_entries.json (collected by psg_census.py on 2026-09-30). Writes the public CSV (one anonymised row per entry, no vendor names)
and a private entries_classified.json (titles and vendors kept, for hand checks only).

All flags are matched, case-insensitively unless noted, against title + description.
Pre-registered (01b-signature.md, before counting): seo, sem_paid, dm_package.
Added AFTER counting, NOT pre-registered (see DEVIATIONS.md): ad_spend, gbp_local, reporting, social, content, website, ai_search.
"""
import json, os, re, csv, random
from collections import Counter

HERE = os.path.dirname(os.path.abspath(__file__))
d = json.load(open(os.path.join(HERE, "psg_entries.json"), encoding="utf-8"))
E = d["entries"]

DEFS = {
    "seo": re.compile(r"\bSEO\b|search engine optimi[sz]ation", re.I),
    "sem_paid": re.compile(r"\bSEM\b|search engine marketing|Google Ads|pay[- ]per[- ]click|\bPPC\b", re.I),
    "ad_spend": re.compile(r"ad[- ]?spend|media (spend|budget|cost)|advertising (budget|spend|cost)|ad budget", re.I),
    "gbp_local": re.compile(r"Google Business|GBP|Google Maps|local SEO", re.I),
    "reporting": re.compile(r"report|track|analytic|dashboard", re.I),
    "social": re.compile(r"social media", re.I),
    "content": re.compile(r"content", re.I),
    "website": re.compile(r"website|web design|web development", re.I),
    # GEO / AEO are matched upper-case only (a case-insensitive match hit "geo" in fleet-tracking products). Changes no published figure.
    "ai_search": re.compile(r"(?i:ChatGPT|Gemini|Perplexity|generative engine|answer engine)|\bGEO\b|\bAEO\b"),
}
for e in E:
    t = e["title"] + " " + e["description"]
    for k, rx in DEFS.items():
        e[k] = bool(rx.search(t))
    e["dm_package"] = e["category"] == "Digital Marketing Packages"

n = len(E)
dm = [e for e in E if e["dm_package"]]
neither = [e for e in dm if not e["seo"] and not e["sem_paid"]]
print("n", n, "collectedUtc", d["collectedUtc"], "types", dict(Counter(e["type"] for e in E)))
print("dm packages", len(dm), "vendors", len({e["vendor"] for e in dm}), "types", dict(Counter(e["type"] for e in dm)))
for k in ("seo", "sem_paid", "ad_spend", "gbp_local", "reporting", "ai_search"):
    print(f"{k:10s} all {sum(e[k] for e in E)}/{n}   dm {sum(e[k] for e in dm)}/{len(dm)}")
print("dm seo+sem", sum(e["seo"] and e["sem_paid"] for e in dm), "seo only", sum(e["seo"] and not e["sem_paid"] for e in dm),
      "sem only", sum(e["sem_paid"] and not e["seo"] for e in dm), "neither", len(neither))
print("neither: social", sum(e["social"] for e in neither), "content", sum(e["content"] for e in neither), "website", sum(e["website"] for e in neither))

random.seed(20260930)
pos, neg = [e for e in E if e["seo"]], [e for e in E if not e["seo"]]
json.dump({"pos": random.sample(pos, min(30, len(pos))), "neg": random.sample(neg, 30)}, open(os.path.join(HERE, "validation_sample.json"), "w", encoding="utf-8"), indent=1, ensure_ascii=False)
json.dump(E, open(os.path.join(HERE, "entries_classified.json"), "w", encoding="utf-8"), indent=1, ensure_ascii=False)

cols = ["entry_id", "type", "sector_generic", "category", "dm_package", "seo", "sem_paid", "ad_spend", "gbp_local", "reporting",
        "social", "content", "website", "ai_search", "collected_at"]
with open(os.path.join(HERE, "singrank-psg-directory-census-2026.csv"), "w", newline="", encoding="utf-8") as f:
    w = csv.writer(f)
    w.writerow(cols)
    for i, e in enumerate(E, 1):
        row = {"entry_id": f"P{i:04d}", "type": e["type"], "sector_generic": int((e["sector"] or "").startswith("Generic")),
               "category": e["category"], "collected_at": d["collectedUtc"]}
        w.writerow([row[c] if c in row else int(e[c]) for c in cols])
print("csv rows", n)
