"""Counting AI-tool use and writing the report (HTML, JSON, CSV). The HTML report is one self-contained file: inline CSS, no scripts, no web fonts, no images, no external links other than plain hyperlinks. """ import csv import hashlib import json from collections import Counter from datetime import datetime from html import escape from . import (FREE_EDITION_NOTE, FREE_LIST_SIZE, FULL_EDITION_URL, FULL_REGISTER_TEXT, TOOL_LIST_ATTRIBUTION, TOOL_LIST_LICENSE, TOOL_LIST_LICENSE_URL, TOOL_LIST_NAME, __version__) from .parse import FileStats, Matcher, load_tools, read_log TOP_USERS = 25 # rows in the per-user table of the HTML report MAX_HOSTS_PER_TOOL = 50 # host names listed per tool in JSON output FLAG_LEVELS = ("high", "medium") COVERAGE_NOTE = ( "This free edition recognises only the %d AI tools of the free list. Requests to AI " "tools that are not on the list are counted here as ordinary traffic, so a low or zero " "count does not mean low or zero AI use." % FREE_LIST_SIZE ) # --------------------------------------------------------------------------- # Counting # --------------------------------------------------------------------------- class ToolUsage(object): __slots__ = ("tool", "requests", "users", "blocked", "first", "last", "hosts") def __init__(self, tool): self.tool = tool self.requests = 0 self.users = set() self.blocked = 0 self.first = None self.last = None self.hosts = set() class Analysis(object): """Streams records from log files and keeps only the counts the report needs.""" def __init__(self, tools=None): self.tools = load_tools() if tools is None else tools self.matcher = Matcher(self.tools) self.files = [] self.total = 0 self.domains = {} # distinct normalised domain -> Tool or None self.users_all = set() # every user/device/IP seen (for "N of M") self.usage = {} # tool domain -> ToolUsage self.user_usage = {} # user -> Counter(tool domain -> requests) self.ai_requests = 0 self.ai_blocked = 0 self.blocked_known = False self.start = None self.end = None def add_file(self, path, fmt="auto"): stats = FileStats(path) self.files.append(stats) add = self.add for domain, user, ts, blocked in read_log(path, stats, fmt): add(domain, user, ts, blocked) return stats def add(self, domain, user=None, ts=None, blocked=None): self.total += 1 try: tool = self.domains[domain] except KeyError: tool = self.domains[domain] = self.matcher.match(domain) if user is not None: self.users_all.add(user) if ts is not None: if self.start is None or ts < self.start: self.start = ts if self.end is None or ts > self.end: self.end = ts if blocked is not None: self.blocked_known = True if tool is None: return self.ai_requests += 1 u = self.usage.get(tool.domain) if u is None: u = self.usage[tool.domain] = ToolUsage(tool) u.requests += 1 if len(u.hosts) < MAX_HOSTS_PER_TOOL: u.hosts.add(domain) if blocked: u.blocked += 1 self.ai_blocked += 1 if ts is not None: if u.first is None or ts < u.first: u.first = ts if u.last is None or ts > u.last: u.last = ts if user is not None: u.users.add(user) c = self.user_usage.get(user) if c is None: c = self.user_usage[user] = Counter() c[tool.domain] += 1 def is_flagged(tool): return tool.risk_level in FLAG_LEVELS or tool.data_sovereignty in FLAG_LEVELS def is_high(tool): return tool.risk_level == "high" or tool.data_sovereignty == "high" def _iso(dt): return dt.isoformat() if dt is not None else None def pseudonyms(user_usage): """Stable pseudonyms: User 1 is the identity with the most AI-tool requests. Ties are broken by a hash of the identity, so the same log always gives the same pseudonyms and the order reveals nothing about the real names. """ order = sorted(user_usage, key=lambda u: (-sum(user_usage[u].values()), hashlib.sha256(u.encode("utf-8", "replace")).hexdigest())) return {u: "User %d" % (i + 1) for i, u in enumerate(order)} def build_summary(analysis, org="Your organisation", anonymise=False, generated=None): """Everything the outputs need, as plain JSON-serialisable data.""" a = analysis generated = generated or datetime.now().replace(microsecond=0) has_users = bool(a.users_all) blocked_known = a.blocked_known usages = sorted(a.usage.values(), key=lambda u: (-u.requests, u.tool.name.lower())) tools = [] for u in usages: row = u.tool.as_dict() row.update({ "requests": u.requests, "users": len(u.users) if has_users else None, "blocked": u.blocked if blocked_known else None, "first_seen": _iso(u.first), "last_seen": _iso(u.last), "flagged": is_flagged(u.tool), "hosts_seen": sorted(u.hosts), }) tools.append(row) cats = {} for u in usages: c = cats.setdefault(u.tool.category or "Uncategorised", {"category": u.tool.category or "Uncategorised", "tools": 0, "requests": 0, "_users": set()}) c["tools"] += 1 c["requests"] += u.requests c["_users"] |= u.users categories = [] for c in sorted(cats.values(), key=lambda c: (-c["requests"], c["category"])): users = c.pop("_users") c["users"] = len(users) if has_users else None categories.append(c) names = pseudonyms(a.user_usage) if anonymise else None users = [] for user, counter in a.user_usage.items(): top = counter.most_common() users.append({ "user": names[user] if names else user, "ai_requests": sum(counter.values()), "tools": len(counter), "top_tools": [a.tools[d].name for d, _ in top[:3]], "flagged_tools": sum(1 for d in counter if is_flagged(a.tools[d])), }) if names: users.sort(key=lambda r: int(r["user"].split()[-1])) else: users.sort(key=lambda r: (-r["ai_requests"], r["user"])) lines = sum(f.lines for f in a.files) skipped = sum(f.skipped for f in a.files) bad_time = sum(f.bad_time for f in a.files) period = {"start": _iso(a.start), "end": _iso(a.end), "days": None} if a.start is not None: period["days"] = (a.end.date() - a.start.date()).days + 1 return { "generator": {"name": "shadow-ai-report", "version": __version__, "edition": "free"}, "organisation": org, "generated": generated.isoformat(), "period": period, "anonymised": bool(anonymise), "totals": { "lines_read": lines, "lines_skipped": skipped, "requests": a.total, "distinct_domains": len(a.domains), "unreadable_timestamps": bad_time, "ai_requests": a.ai_requests, "ai_share": round(a.ai_requests / a.total, 6) if a.total else 0.0, "ai_tools": len(usages), "users_seen": len(a.users_all) if has_users else None, "ai_users": len(a.user_usage) if has_users else None, "ai_blocked": a.ai_blocked if blocked_known else None, "flagged_tools": sum(1 for u in usages if is_flagged(u.tool)), "high_risk_or_exposure_tools": sum(1 for u in usages if is_high(u.tool)), }, "tools": tools, "categories": categories, "users": users, "inputs": [f.as_dict() for f in a.files], "tool_list": {"name": TOOL_LIST_NAME, "tools": len(a.tools), "attribution": TOOL_LIST_ATTRIBUTION, "license": TOOL_LIST_LICENSE}, "coverage_note": COVERAGE_NOTE, "full_edition": {"register": FULL_REGISTER_TEXT, "url": FULL_EDITION_URL, "note": FREE_EDITION_NOTE}, } # --------------------------------------------------------------------------- # JSON and CSV # --------------------------------------------------------------------------- def write_json(summary, path): with open(path, "w", encoding="utf-8") as f: json.dump(summary, f, indent=2, ensure_ascii=False) f.write("\n") CSV_COLUMNS = ("tool", "domain", "category", "subcategory", "ai_type", "risk_level", "data_sovereignty", "requests", "users", "blocked", "first_seen", "last_seen") def write_csv(summary, path): with open(path, "w", encoding="utf-8", newline="") as f: w = csv.writer(f) w.writerow(CSV_COLUMNS) for t in summary["tools"]: w.writerow([t["name"], t["domain"], t["category"], t["subcategory"], t["ai_type"], t["risk_level"], t["data_sovereignty"], t["requests"], "" if t["users"] is None else t["users"], "" if t["blocked"] is None else t["blocked"], t["first_seen"] or "", t["last_seen"] or ""]) # --------------------------------------------------------------------------- # HTML # --------------------------------------------------------------------------- def fmt_int(n): return "{:,}".format(n) def fmt_pct(x): if not x: return "0%" if x < 0.001: return "<0.1%" return "{:.1f}%".format(x * 100) def fmt_date(iso): dt = datetime.fromisoformat(iso) if isinstance(iso, str) else iso return "%d %s" % (dt.day, dt.strftime("%b %Y")) def period_text(period): if not period.get("start"): return "Not stated in the log" s = datetime.fromisoformat(period["start"]) e = datetime.fromisoformat(period["end"]) if s.date() == e.date(): return fmt_date(s) if (s.year, s.month) == (e.year, e.month): return "%d–%s" % (s.day, fmt_date(e)) if s.year == e.year: return "%d %s – %s" % (s.day, s.strftime("%b"), fmt_date(e)) return "%s – %s" % (fmt_date(s), fmt_date(e)) _LEVEL_LABEL = {"high": "High", "medium": "Medium", "low": "Low", "unknown": "Unknown", "not_assessed": "Not assessed"} def pill(level): level = (level or "not_assessed").lower() cls = level if level in ("high", "medium", "low") else "unknown" return '%s' % (cls, escape(_LEVEL_LABEL.get(level, level))) def _num(n): return "—" if n is None else fmt_int(n) def _plural(n, one, many): return one if n == 1 else many CSS = """ :root{--bg:#f8fafc;--card:#fff;--line:#e2e8f0;--line2:#f1f5f9;--ink:#0f172a;--ink2:#334155; --muted:#64748b;--primary:#2563eb;--cyan:#22d3ee;--violet:#8b5cf6;--danger:#ef4444;--warn:#f59e0b; --safe:#10b981;--navy:#040814; --sans:-apple-system,BlinkMacSystemFont,"Segoe UI",Inter,Roboto,"Helvetica Neue",Arial,sans-serif; --mono:ui-monospace,SFMono-Regular,"SF Mono",Menlo,Consolas,"Liberation Mono",monospace;color-scheme:light} *{box-sizing:border-box} html{-webkit-text-size-adjust:100%;text-size-adjust:100%} body{margin:0;background:var(--bg);color:var(--ink);font:15px/1.55 var(--sans);overflow-wrap:break-word} a{color:var(--primary)} .wrap{max-width:1080px;margin:0 auto;padding:0 24px} .label,.eyebrow,th,dt,.pill,.chip{font-family:var(--mono);text-transform:uppercase;letter-spacing:.09em} .label{font-size:11px;font-weight:600;color:var(--muted)} .band{background:var(--navy);color:#fff;padding:36px 0 84px;position:relative;overflow:hidden} .band:before{content:"";position:absolute;inset:0;background:radial-gradient(600px 260px at 88% 0%,rgba(37,99,235,.35),transparent 70%),radial-gradient(420px 220px at 70% 100%,rgba(139,92,246,.22),transparent 70%);pointer-events:none} .band:after{content:"";position:absolute;left:0;right:0;bottom:0;height:3px;background:linear-gradient(90deg,var(--primary),var(--cyan),var(--violet))} .band .wrap{position:relative} .toprow{display:flex;flex-wrap:wrap;gap:8px 16px;align-items:center;justify-content:space-between} .eyebrow{font-size:11px;font-weight:600;color:var(--cyan)} .chip{display:inline-block;font-size:10px;font-weight:600;color:#cbd5e1;border:1px solid rgba(255,255,255,.22);border-radius:999px;padding:3px 10px} h1{font-size:40px;line-height:1.1;letter-spacing:-.025em;margin:14px 0 6px;font-weight:700} .org{font-size:20px;font-weight:600;color:#cbd5e1;margin:0} .meta{display:grid;grid-template-columns:repeat(auto-fit,minmax(170px,1fr));gap:14px 24px;margin:28px 0 0;padding:0} .meta div{border-left:2px solid rgba(34,211,238,.55);padding-left:12px} .meta dt{font-size:10px;font-weight:600;color:rgba(255,255,255,.55)} .meta dd{margin:3px 0 0;font-weight:600;font-size:15px} .meta .days{color:rgba(255,255,255,.55);font-weight:500;white-space:nowrap} .tiles{display:grid;grid-template-columns:repeat(auto-fit,minmax(180px,1fr));gap:14px;margin-top:-52px;position:relative} .tile{background:var(--card);border:1px solid var(--line);border-radius:14px;padding:16px 18px 18px;box-shadow:0 1px 2px rgba(15,23,42,.04),0 8px 24px -12px rgba(15,23,42,.12)} .tile .value{font-size:32px;font-weight:700;letter-spacing:-.02em;line-height:1.15;margin-top:8px;font-variant-numeric:tabular-nums} .tile .sub{font-size:13px;color:var(--muted);margin-top:2px} .tile.danger .value{color:var(--danger)} .tile.safe .value{color:var(--safe)} .tile .dot{display:inline-block;width:8px;height:8px;border-radius:50%;margin-right:7px;vertical-align:1px} .card{background:var(--card);border:1px solid var(--line);border-radius:16px;padding:26px 28px;margin-top:18px} .card h2{font-size:21px;line-height:1.25;letter-spacing:-.015em;margin:6px 0 6px} .card h3{font-size:15px;margin:0 0 6px} .intro{color:var(--ink2);margin:0 0 18px;max-width:760px} .notice{border:1px solid #fde68a;background:#fffbeb;border-radius:12px;padding:14px 16px;margin-top:18px;color:#78350f} .tablewrap{overflow-x:auto;-webkit-overflow-scrolling:touch;margin:0 -4px;padding:0 4px} table{width:100%;border-collapse:collapse;font-size:14px} th{font-size:10.5px;font-weight:600;color:var(--muted);text-align:left;padding:9px 12px;border-bottom:1px solid var(--line);white-space:nowrap} td{padding:11px 12px;border-bottom:1px solid var(--line2);vertical-align:top} tr:last-child td{border-bottom:0} th.num,td.num{text-align:right;font-variant-numeric:tabular-nums;white-space:nowrap} td .name{font-weight:600} td .dom,.mono{font-family:var(--mono);font-size:12.5px;color:var(--muted)} td.user{font-family:var(--mono);font-size:13px;color:var(--ink2)} td.small{font-size:13px;color:var(--muted)} td .subcat{font-size:12.5px;color:var(--muted)} .pill{display:inline-block;font-size:10px;font-weight:700;padding:3px 8px;border-radius:999px;white-space:nowrap} .pill.high{background:#fee2e2;color:#b91c1c} .pill.medium{background:#fef3c7;color:#92400e} .pill.low{background:#d1fae5;color:#065f46} .pill.unknown{background:#f1f5f9;color:#475569} .bars{margin:0;padding:0;list-style:none} .bars li{display:grid;grid-template-columns:minmax(150px,230px) 1fr minmax(150px,auto);gap:14px;align-items:center;padding:9px 0;border-bottom:1px solid var(--line2)} .bars li:last-child{border-bottom:0} .bars .cat{font-weight:600;font-size:14px} .bar{height:10px;background:#eef2f7;border-radius:999px;overflow:hidden} .bar span{display:block;height:100%;min-width:4px;border-radius:999px;background:linear-gradient(90deg,var(--primary),var(--violet))} .bars .fig{font-size:13px;color:var(--muted);text-align:right;font-variant-numeric:tabular-nums;white-space:nowrap} .bars .fig b{color:var(--ink);font-weight:700} .explain{display:grid;grid-template-columns:repeat(auto-fit,minmax(280px,1fr));gap:14px;margin:0 0 20px} .box{border:1px solid var(--line);border-radius:12px;padding:16px 18px;background:#fbfdff} .box p{margin:0 0 10px;color:var(--ink2);font-size:14px} .box dl{margin:0;display:grid;grid-template-columns:auto 1fr;gap:8px 12px;align-items:baseline;font-size:14px} .box dd{margin:0;color:var(--ink2)} .why{font-size:13px;color:var(--ink2)} ol.steps{margin:0;padding:0;list-style:none;counter-reset:s;display:grid;gap:12px} ol.steps li{counter-increment:s;position:relative;padding:14px 16px 14px 56px;border:1px solid var(--line);border-radius:12px} ol.steps li:before{content:counter(s,decimal-leading-zero);position:absolute;left:16px;top:14px;font:700 13px var(--mono);color:var(--primary)} ol.steps b{display:block;margin-bottom:2px} ol.steps span{color:var(--ink2);font-size:14px} .facts{display:grid;grid-template-columns:repeat(auto-fit,minmax(170px,1fr));gap:12px;margin:18px 0} .fact{border:1px solid var(--line);border-radius:12px;padding:12px 14px} .fact .v{font-size:20px;font-weight:700;font-variant-numeric:tabular-nums;margin-top:4px} .method{margin-top:18px} .method p{color:var(--ink2);margin:0 0 10px;max-width:820px;font-size:14px} .callout{border-left:3px solid var(--primary);background:#eff6ff;border-radius:0 12px 12px 0;padding:14px 18px;color:#1e3a8a;margin:4px 0 14px;max-width:860px} .full{margin-top:20px;border-radius:14px;padding:20px 22px;background:linear-gradient(135deg,#eff6ff,#f5f3ff);border:1px solid #dbeafe} .full h3{font-size:16px} .full p{margin:4px 0 6px;color:var(--ink2);font-size:14px} .full ul{margin:6px 0 12px;padding-left:20px;color:var(--ink2);font-size:14px} .full li{margin:3px 0} .full a{font-weight:600;word-break:break-all} .small{font-size:13px;color:var(--muted)} .foot{padding:26px 24px 48px;font-size:13px;color:var(--muted)} .foot p{margin:0 0 8px;max-width:900px} .foot a{word-break:break-all} @media (max-width:640px){ .wrap{padding:0 16px} .band{padding:28px 0 72px} h1{font-size:30px} .org{font-size:17px} .tiles{grid-template-columns:repeat(2,minmax(0,1fr));gap:10px;margin-top:-44px} .tile{padding:14px} .tile .value{font-size:26px} .card{padding:20px 16px;border-radius:14px} .bars li{grid-template-columns:1fr auto;gap:6px 12px} .bars .bar{grid-column:1/-1;grid-row:2} .meta{grid-template-columns:1fr 1fr;gap:14px 16px} .facts{grid-template-columns:repeat(2,minmax(0,1fr))} .fact .v{font-size:18px} .tile:last-child:nth-child(odd){grid-column:1/-1} .tablewrap{overflow:visible;margin:0;padding:0} table.rt thead{display:none} table.rt,table.rt tbody{display:block;width:100%} table.rt tr{display:grid;grid-template-columns:repeat(6,minmax(0,1fr));grid-auto-flow:row dense;gap:10px 12px;padding:14px 0;border-bottom:1px solid var(--line)} table.rt tr:first-child{padding-top:2px} table.rt tr:last-child{border-bottom:0} table.rt td{display:block;grid-column:span 3;border:0;padding:0;text-align:left;white-space:normal;min-width:0} table.rt td.num{grid-column:span 2;text-align:left} table.rt td.lead,table.rt td.wide{grid-column:1/-1} table.rt td:before{content:attr(data-label);display:block;font:600 9.5px/1.7 var(--mono);text-transform:uppercase;letter-spacing:.09em;color:var(--muted)} table.rt td.lead{font-size:15px} table.rt td.lead:before,table.rt td.rank{display:none} .foot{padding:22px 16px 40px} } @media print{ @page{size:A4;margin:12mm} body{background:#fff;font-size:12px} .band,.band:before,.band:after,.tile,.pill,.bar,.bar span,.full,.notice,.callout{-webkit-print-color-adjust:exact;print-color-adjust:exact} .band{padding:24px 0 64px} .wrap{max-width:none;padding:0 4px} .tile,.box,.fact,.full,ol.steps li,.bars li,tr{break-inside:avoid} .card{border-color:#cbd5e1} .card>.label,.card h2,.card h3,.card .intro{break-after:avoid;page-break-after:avoid} .explain{break-inside:avoid} thead{display:table-header-group} .tablewrap{overflow:visible} .tile{box-shadow:none;padding:10px 12px} .tiles{grid-template-columns:repeat(5,minmax(0,1fr));gap:8px} .tile .value{font-size:22px} .tile .sub{font-size:10.5px} .card{padding:18px 20px} table{font-size:11px} th,td{padding:6px 8px} td .dom,.mono,td .subcat,td.small,.why{font-size:10.5px} a{color:inherit;text-decoration:none} } """ def _table(columns, rows): """A table that turns into stacked cards on narrow screens. ``columns`` is a list of ``(label, css_class)``; ``rows`` holds cell HTML. The column with class ``lead`` becomes the card heading on phones. """ def cls(c, tag): c = " ".join(x for x in c.split() if tag == "td" or x == "num") return ' class="%s"' % c if c else "" head = "".join('
An overall rating from the AI Tools Blocklist register of how much caution a tool needs before people use it with work data.
How far the tool’s ownership, hosting or legal jurisdiction places what people type or upload under foreign data-access laws.
None of the AI tools detected carries a high or medium rating for ' 'risk or data sovereignty. This covers only the tools on the free list; see ' 'coverage.
') else: # High ratings first, then by requests. flagged.sort(key=lambda r: (-(r["risk_level"] == "high" or r["data_sovereignty"] == "high"), -r["requests"], r["name"].lower())) columns = [("Tool", "lead"), ("Risk", ""), ("Data sovereignty", ""), ("Why it is listed", "why wide"), ("Requests", "num"), ("Users", "num")] rows = [[_tool_cell(r), pill(r["risk_level"]), pill(r["data_sovereignty"]), escape(_why(r)), fmt_int(r["requests"]), _num(r["users"]) if has_users else "—"] for r in flagged] body = _table(columns, rows) n = len(flagged) intro = ("%d of the AI tools found %s a high or medium rating for risk or data sovereignty. " "Review these first." % (n, _plural(n, "carries", "carry"))) if n else "" return ('%s
' % intro if intro else "", explain, body)) def _users_section(s): t = s["totals"] if t["users_seen"] is None or not s["users"]: return "" users = s["users"][:TOP_USERS] columns = [("#", "num rank"), ("User", "user lead"), ("AI requests", "num"), ("Tools", "num"), ("Most used", "wide"), ("Flagged tools", "num")] rows = [[str(i + 1), escape(u["user"]), fmt_int(u["ai_requests"]), str(u["tools"]), escape(", ".join(u["top_tools"])), ('%d' % u["flagged_tools"]) if u["flagged_tools"] else "0"] for i, u in enumerate(users)] shown = ("the top %d of %s" % (len(users), fmt_int(len(s["users"])))) if len(s["users"]) > len(users) \ else "all %s" % fmt_int(len(s["users"])) if s["anonymised"]: who = ("Identities are replaced by pseudonyms (User 1 has the most AI-tool requests). " "The same log always gives the same pseudonyms.") else: who = ("Identities are shown as recorded in the log: e-mail, user name, device or IP address. " "Run with --anonymise to replace them with pseudonyms.") return ('Showing %s users and devices that reached ' 'an AI tool on the list. %s Use this to understand demand, not to single people out.
' '%sWhat counts as a match. A request counts for a tool when the domain in the log is the " "tool’s domain or a subdomain of it (api.example.ai counts for " "example.ai). Domains are lower-cased, and a leading " "www. and trailing dots are removed first.
" "What a request is. One line of the log. DNS logs record look-ups, not page views or " "prompts: one visit can cause several look-ups, and devices cache answers. Treat the counts as a " "measure of activity, not as an exact number of uses.
" "Users. Counted from the user, e-mail, device or IP column of the log. One person on " "several devices counts more than once; several people behind one IP address count once.
" "Skipped lines could not be read or had no valid domain. Times are shown in UTC " "when the log states a time zone, otherwise as written in the log.
") full = ( 'The full register covers %s. Compared with this free edition it adds:
" "AI-tool list: %s. Sources: %s, licensed %s.
' '%d %s from the free list, ' 'sorted by number of requests. Requests include attempts that your filter blocked.
' '%sRequests per category of AI tool, ' 'with the number of distinct tools in each.
%s%(org)s