{e(org)}
Summary
{' '.join(sentences)}
#!/usr/bin/env python3 # Ghostbuster: find ghost accounts and excess access in a GitHub organization. # Created by Jens Naterman at Adcyma. Latest version: https://adcyma.com/tools/ghostbuster # # SPDX-License-Identifier: MIT # # MIT License # # Copyright (c) 2026 Jens Naterman, Adcyma # # Permission is hereby granted, free of charge, to any person obtaining a copy # of this software and associated documentation files (the "Software"), to deal # in the Software without restriction, including without limitation the rights # to use, copy, modify, merge, publish, distribute, sublicense, and/or sell # copies of the Software, and to permit persons to whom the Software is # furnished to do so, subject to the following conditions: # # The above copyright notice and this permission notice shall be included in all # copies or substantial portions of the Software. # # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE # SOFTWARE. """ Ghostbuster: find ghost accounts and excess access in a GitHub organization. It answers four questions about your org: 1. Who has access to what? 2. Are they active? 3. Are they still with the company? (with --mapping, or SAML SSO on Enterprise Cloud) 4. Do they still need that access? Read-only: it never changes anything in GitHub. It signs in through your GitHub CLI (gh) login, talks only to GitHub's API, and writes its results to a local folder. Suggested fixes are written as commented-out `gh` commands for you to review. Needs Python 3.8+ and the GitHub CLI (it offers to install gh if it's missing). python ghostbuster.py python ghostbuster.py --org my-org --days 90 python ghostbuster.py --org my-org --mapping people.csv Created by Jens Naterman at Adcyma. Free to use, change and share under the MIT license above, as long as the copyright notice stays in. """ from __future__ import annotations import argparse import csv import datetime as dt import html import json import math import os import platform import re import shutil import subprocess import sys import threading import time import urllib.error import urllib.parse import urllib.request import webbrowser from collections import Counter, defaultdict from concurrent.futures import ThreadPoolExecutor, as_completed VERSION = "0.1.0" AUTHOR = "Jens Naterman, Adcyma" HOME_URL = "https://adcyma.com/tools/ghostbuster" API_URL = "https://api.github.com" API_VERSION = "2022-11-28" GH_INSTALL_URL = "https://cli.github.com" GH_LINUX_URL = "https://github.com/cli/cli/blob/trunk/docs/install_linux.md" SEVERITIES = ("critical", "high", "medium", "low", "info") SEV_COLOR = {"critical": "1;31", "high": "31", "medium": "33", "low": "36", "info": "2"} VERDICT_ORDER = {"departed": 0, "dormant": 1, "over-privileged": 2, "unknown": 3, "active": 4} PERMISSION_RANK = {"read": 1, "triage": 2, "write": 3, "maintain": 4, "admin": 5} PERMISSION_NAME = {rank: name for name, rank in PERMISSION_RANK.items()} BASE_RANK = {"none": 0, "read": 1, "write": 3, "admin": 5} PLAN_LABEL = {"free": "Free", "team": "Team", "enterprise": "Enterprise Cloud"} ACTIVITY_SOURCES = ( ("pushes", "Pushes, merges and branch changes"), ("issues", "Issues and pull requests opened"), ("comments", "Issue and pull request comments"), ("review-comments", "Code review comments"), ("reviews", "Pull request reviews and merges"), ) SIGNAL_LABELS = {"push": "push", "pr": "pull request", "talk": "comment", "audit": "audit log", "git": "git clone/fetch", "copilot": "Copilot", "credential": "token/SSH key use", "search": "found by search"} LEFT_VALUES = {"left", "departed", "disabled", "inactive", "terminated", "offboarded", "false", "no", "0", "slutat"} ACTIVE_VALUES = {"active", "enabled", "employed", "true", "yes", "1", "aktiv"} # App permissions that let an app change code, settings or people across the org. SENSITIVE_APP_PERMISSIONS = { "administration", "contents", "members", "organization_administration", "workflows", "secrets", "organization_secrets", "actions", "environments", "organization_user_blocking", } USE_COLOR = False # ---------------------------------------------------------------- console def _setup_console(): global USE_COLOR for stream in (sys.stdout, sys.stderr): try: stream.reconfigure(errors="replace") except (AttributeError, ValueError): pass if os.name == "nt": os.system("") # switches older Windows consoles into ANSI mode USE_COLOR = sys.stdout.isatty() and "NO_COLOR" not in os.environ def paint(text, code): return f"\033[{code}m{text}\033[0m" if USE_COLOR else text def say(msg=""): print(msg, flush=True) def step(msg): say(f"{paint('>', '36')} {msg}") def note(msg): say(f" {paint(msg, '2')}") def warn(msg): say(f"{paint('!', '33')} {msg}") def die(msg): print(f"{paint('x', '31')} {msg}", file=sys.stderr, flush=True) sys.exit(1) def ask_yes_no(question, default=False): if not sys.stdin.isatty(): return False suffix = " [Y/n] " if default else " [y/N] " try: answer = input(question + suffix).strip().lower() except EOFError: return False if not answer: return default return answer in ("y", "yes", "j", "ja") class Progress: def __init__(self, label, total): self.label, self.total, self.done = label, total, 0 self.lock = threading.Lock() self.tty = sys.stdout.isatty() def tick(self): with self.lock: self.done += 1 if self.tty: print(f"\r {self.label}: {self.done}/{self.total}", end="", flush=True) def finish(self): if self.tty and self.total: print(flush=True) # ---------------------------------------------------------------- time helpers def utcnow(): return dt.datetime.now(dt.timezone.utc) def parse_ts(value): if value in (None, ""): return None if isinstance(value, (int, float)): # audit log uses epoch milliseconds return dt.datetime.fromtimestamp(value / 1000, dt.timezone.utc) text = str(value).replace("Z", "+00:00") text = re.sub(r"\.(\d{1,6})\d*(?=[+-])", lambda m: "." + m.group(1).ljust(6, "0"), text) try: ts = dt.datetime.fromisoformat(text) except ValueError: return None return ts if ts.tzinfo else ts.replace(tzinfo=dt.timezone.utc) def later(a, b): if a is None: return b if b is None: return a return max(a, b) def fmt_date(ts): return ts.strftime("%Y-%m-%d") if ts else "" def iso(ts): return ts.isoformat().replace("+00:00", "Z") if ts else None def record(bucket, login, ts, since): if login and ts and ts >= since: bucket[login] = later(bucket.get(login), ts) def login_of(obj): return (obj or {}).get("login") def nodes_of(connection): return [n for n in (connection or {}).get("nodes") or [] if n] def perm_rank(role_name, permissions=None): if permissions: for key, rank in (("admin", 5), ("maintain", 4), ("push", 3), ("triage", 2), ("pull", 1)): if permissions.get(key): return rank return PERMISSION_RANK.get(role_name or "", {"pull": 1, "push": 3}.get(role_name or "", 0)) def activity_period(days): for limit, period in ((1, "day"), (7, "week"), (30, "month"), (90, "quarter")): if days <= limit: return period return "year" # ---------------------------------------------------------------- GitHub CLI bootstrap def find_gh(): path = shutil.which("gh") if path: return path candidates = [] if os.name == "nt": for base in (os.environ.get("ProgramFiles"), os.environ.get("ProgramFiles(x86)"), os.environ.get("LOCALAPPDATA")): if base: candidates += [os.path.join(base, "GitHub CLI", "gh.exe"), os.path.join(base, "Programs", "GitHub CLI", "gh.exe")] else: candidates += ["/opt/homebrew/bin/gh", "/usr/local/bin/gh", "/usr/bin/gh"] return next((c for c in candidates if os.path.isfile(c)), None) def install_gh(): system = platform.system() say("Ghostbuster signs in to GitHub through the GitHub CLI (gh), which isn't installed.") if system == "Windows" and shutil.which("winget"): cmd = ["winget", "install", "--id", "GitHub.cli", "-e", "--source", "winget", "--accept-package-agreements", "--accept-source-agreements"] elif system == "Darwin" and shutil.which("brew"): cmd = ["brew", "install", "gh"] else: url = GH_LINUX_URL if system == "Linux" else GH_INSTALL_URL die(f"Install it from {url}, run `gh auth login`, then run Ghostbuster again.") if not ask_yes_no(f"Install it now with `{' '.join(cmd[:4])}`?", default=True): die(f"Install gh from {GH_INSTALL_URL}, run `gh auth login`, then run Ghostbuster again.") if subprocess.run(cmd).returncode != 0: die(f"Installing gh failed. Install it from {GH_INSTALL_URL} and run Ghostbuster again.") gh = find_gh() if not gh: die("gh is installed but this terminal can't see it yet. Open a new terminal and run Ghostbuster again.") return gh def gh_token(gh, account=None): cmd = [gh, "auth", "token", "--hostname", "github.com"] if account: cmd += ["--user", account] result = subprocess.run(cmd, capture_output=True, text=True) token = result.stdout.strip() return token if result.returncode == 0 and token else None def ensure_token(gh, args): token = gh_token(gh, args.account) if token: return token if args.account: die(f"gh has no login for '{args.account}'. Run `gh auth login`, or check `gh auth status`.") say("You're not signed in to the GitHub CLI.") if not ask_yes_no("Sign in now with `gh auth login`?", default=True): die("Run `gh auth login`, then run Ghostbuster again.") scopes = "read:org,repo" + (",read:audit_log" if args.audit_log else "") subprocess.run([gh, "auth", "login", "--hostname", "github.com", "--git-protocol", "https", "--web", "--scopes", scopes]) token = gh_token(gh) if not token: die("Still not signed in. Run `gh auth login`, then run Ghostbuster again.") return token def token_from_env(): return next((name for name in ("GH_TOKEN", "GITHUB_TOKEN") if os.environ.get(name)), None) def check_scopes(gh, headers, args): """Returns True if the token was refreshed and must be fetched again.""" raw = headers.get("X-OAuth-Scopes") if raw is None: warn("This token doesn't report its scopes (fine-grained token?). Checks it can't reach are listed as skipped.") return False scopes = {s.strip() for s in raw.split(",") if s.strip()} missing = [] if "repo" not in scopes: missing.append("repo") if not scopes & {"read:org", "write:org", "admin:org"}: missing.append("read:org") if args.audit_log and "read:audit_log" not in scopes: missing.append("read:audit_log") if not missing: return False env = token_from_env() if env: die(f"The token in {env} is missing scopes: {', '.join(missing)}. Unset {env} to use your gh login, or give the token those scopes.") refresh = f"gh auth refresh -h github.com -s {','.join(missing)}" if args.account: # gh can only refresh the active account die(f"The gh login for {args.account} is missing scopes: {', '.join(missing)}. " f"Run `gh auth switch --user {args.account}` and `{refresh}`, then run Ghostbuster again.") say(f"Your gh login is missing scopes Ghostbuster needs: {', '.join(missing)}.") if not ask_yes_no("Add them now with `gh auth refresh`?", default=True): die(f"Run `{refresh}`, then run Ghostbuster again.") subprocess.run([gh, "auth", "refresh", "--hostname", "github.com", "--scopes", ",".join(missing)]) return True # ---------------------------------------------------------------- GitHub API client class GitHubError(Exception): def __init__(self, status, message, url=""): super().__init__(f"HTTP {status}: {message}") self.status, self.message, self.url = status, message, url def reason(self): if self.status == 403: return "no permission (needs an org owner and the right token scopes)" if self.status == 404: return "not available (plan, feature or permission)" if self.status == 0: return f"network error: {self.message}" return f"HTTP {self.status}: {self.message}" def _error_message(raw, fallback): try: return json.loads(raw).get("message") or str(fallback) except (ValueError, AttributeError): return str(fallback) def _next_link(link_header): for part in (link_header or "").split(","): match = re.match(r'\s*<([^>]+)>\s*;\s*rel="next"', part) if match: return match.group(1) return None class GitHub: def __init__(self, token): self._token = token self._lock = threading.Lock() self._limits = {} self._notice_until = 0.0 self.calls = 0 def url(self, path, params=None): url = path if path.startswith("http") else API_URL + path if params: url += ("&" if "?" in url else "?") + urllib.parse.urlencode(params) return url def _note_limits(self, headers): if not headers or headers.get("x-ratelimit-remaining") is None: return try: remaining, reset = int(headers["x-ratelimit-remaining"]), int(headers["x-ratelimit-reset"]) limit = int(headers.get("x-ratelimit-limit") or 0) except (KeyError, TypeError, ValueError): return # Keep a small reserve: 25 calls for the hourly limits, fewer for search's 30 per minute. floor = max(1, min(25, limit // 10)) if limit else 25 with self._lock: self._limits[headers.get("x-ratelimit-resource", "core")] = (remaining, reset, floor) def _sleep(self, seconds, why): if seconds > 3600: raise GitHubError(429, "rate limit won't reset within an hour; try again later") with self._lock: show = time.time() >= self._notice_until if show: self._notice_until = time.time() + seconds if show: say("") warn(f"{why}; pausing {int(seconds)}s.") time.sleep(seconds) def _pace(self, resource): with self._lock: remaining, reset, floor = self._limits.get(resource, (None, None, 0)) if remaining is not None and remaining < floor and reset - time.time() > 0: self._sleep(reset - time.time() + 2, "GitHub rate limit nearly used up") with self._lock: self._limits.pop(resource, None) def request(self, method, url, body=None, resource="core"): data = json.dumps(body).encode() if body is not None else None headers = { "Authorization": f"Bearer {self._token}", "Accept": "application/vnd.github+json", "X-GitHub-Api-Version": API_VERSION, "User-Agent": f"ghostbuster/{VERSION}", } if data is not None: headers["Content-Type"] = "application/json" for attempt in range(6): self._pace(resource) req = urllib.request.Request(url, data=data, headers=headers, method=method) try: with urllib.request.urlopen(req, timeout=60) as resp: raw = resp.read() self._note_limits(resp.headers) with self._lock: self.calls += 1 return (json.loads(raw) if raw else None), resp.headers except urllib.error.HTTPError as err: raw = err.read() self._note_limits(err.headers) with self._lock: self.calls += 1 message = _error_message(raw, err.reason) limited = err.headers is not None and ( err.headers.get("x-ratelimit-remaining") == "0" or err.headers.get("retry-after")) if err.code in (403, 429) and (limited or "rate limit" in message.lower()): self._sleep(self._backoff(err.headers, attempt), "GitHub asked us to slow down") continue if err.code >= 500 and attempt < 3: time.sleep(2 ** attempt) continue raise GitHubError(err.code, message, url) from None except (urllib.error.URLError, OSError) as err: if attempt < 3: time.sleep(2 ** attempt) continue raise GitHubError(0, str(getattr(err, "reason", err)), url) from None raise GitHubError(429, "still rate limited after several retries", url) @staticmethod def _backoff(headers, attempt): retry_after = (headers or {}).get("retry-after") if retry_after and str(retry_after).isdigit(): return max(1, int(retry_after)) if headers and headers.get("x-ratelimit-remaining") == "0": return max(1, int(headers.get("x-ratelimit-reset", "0")) - time.time() + 2) return 60 * (attempt + 1) def get(self, path, params=None): return self.request("GET", self.url(path, params))[0] def paginate(self, path, params=None, key=None, max_pages=None): url = self.url(path, dict({"per_page": 100}, **(params or {}))) pages = 0 while url: data, headers = self.request("GET", url) items = (data or {}).get(key) or [] if key else (data or []) for item in items: yield item pages += 1 if max_pages and pages >= max_pages: return url = _next_link(headers.get("Link")) def list(self, path, params=None, key=None): return list(self.paginate(path, params, key)) def graphql(self, query, variables=None): for attempt in range(4): data, _ = self.request("POST", API_URL + "/graphql", {"query": query, "variables": variables or {}}, resource="graphql") data = data or {} errors = data.get("errors") or [] if any(e.get("type") == "RATE_LIMITED" for e in errors): self._sleep(60 * (attempt + 1), "GitHub GraphQL rate limit hit") continue if errors and not data.get("data"): raise GitHubError(200, "; ".join(e.get("message", "") for e in errors[:3]), "graphql") return data.get("data") or {}, errors raise GitHubError(429, "GraphQL still rate limited after several retries", "graphql") def run_parallel(fn, items, workers, label): results, errors = [], [] progress = Progress(label, len(items)) with ThreadPoolExecutor(max_workers=max(1, workers)) as pool: futures = {pool.submit(fn, item): item for item in items} for future in as_completed(futures): item = futures[future] try: results.append((item, future.result())) except GitHubError as err: errors.append((item, err)) progress.tick() progress.finish() return results, errors class Coverage: def __init__(self): self.rows = [] def add(self, check, status, detail=""): self.rows.append({"check": check, "status": status, "detail": detail}) def guarded(cov, check, fn, describe=None): try: result = fn() except GitHubError as err: cov.add(check, "skipped", err.reason()) return None cov.add(check, "checked", describe(result) if describe else "") return result def part_status(failed, total): return "checked" if not failed else ("skipped" if failed == total else "partial") # ---------------------------------------------------------------- GraphQL queries PR_QUERY = """ query($owner: String!, $name: String!, $cursor: String) { repository(owner: $owner, name: $name) { pullRequests(first: 50, after: $cursor, orderBy: {field: UPDATED_AT, direction: DESC}) { pageInfo { hasNextPage endCursor } nodes { updatedAt mergedAt mergedBy { login } reviews(last: 30) { nodes { submittedAt author { login } } } } } } }""" SAML_QUERY = """ query($org: String!, $cursor: String) { organization(login: $org) { samlIdentityProvider { externalIdentities(first: 100, after: $cursor) { pageInfo { hasNextPage endCursor } nodes { user { login } samlIdentity { nameId emails { value } } scimIdentity { username } } } } } }""" def page_graphql(gh, query, field, owner, name, since, max_pages, handle): """Walks a repo's pull requests, most recently updated first, until they're older than `since`. Returns True if it stopped at the page cap rather than the window edge.""" cursor = None for _ in range(max_pages): data, _errors = gh.graphql(query, {"owner": owner, "name": name, "cursor": cursor}) connection = ((data or {}).get("repository") or {}).get(field) or {} nodes = nodes_of(connection) for node in nodes: handle(node) info = connection.get("pageInfo") or {} oldest = parse_ts(nodes[-1].get("updatedAt")) if nodes else None if not nodes or not info.get("hasNextPage") or (oldest and oldest < since): return False cursor = info.get("endCursor") return True def handle_pr(node, out, since): record(out["pr"], login_of(node.get("mergedBy")), parse_ts(node.get("mergedAt")), since) for review in nodes_of(node.get("reviews")): record(out["pr"], login_of(review.get("author")), parse_ts(review.get("submittedAt")), since) def confirm_with_search(gh, org, logins, since): """Second opinion for people the repository scan found no activity for. GitHub search sees across every repository, including ones the scan only partly read. Returns {login: (timestamp, what was found)}.""" found, day = {}, f"{since:%Y-%m-%d}" progress = Progress("search", len(logins)) for login in logins: progress.tick() checks = ( ("/search/issues", f"org:{org} author:{login} created:>={day}", "created", "issue or pull request"), ("/search/commits", f"org:{org} author:{login} committer-date:>={day}", "committer-date", "commit"), ) for path, query, sort, label in checks: try: data = gh.get(path, {"q": query, "sort": sort, "order": "desc", "per_page": 1}) except GitHubError: continue items = (data or {}).get("items") or [] if items: item = items[0] ts = parse_ts(item.get("created_at") or ((item.get("commit") or {}).get("committer") or {}).get("date")) found[login] = (ts, label) break if login in found: continue # `commenter:` matches issues they ever commented on, so confirm the comment itself is recent. try: data = gh.get("/search/issues", {"q": f"org:{org} commenter:{login} updated:>={day}", "sort": "updated", "order": "desc", "per_page": 5}) except GitHubError: continue for item in (data or {}).get("items") or []: try: comments = gh.get(item["comments_url"], {"since": since.strftime("%Y-%m-%dT%H:%M:%SZ"), "per_page": 100}) except GitHubError: continue stamps = [parse_ts(c.get("created_at")) for c in comments or [] if login_of(c.get("user")) == login and parse_ts(c.get("created_at"))] recent = [ts for ts in stamps if ts >= since] if recent: found[login] = (max(recent), "comment") break progress.finish() return found def load_saml(gh, org): identities, cursor = {}, None while True: data, _ = gh.graphql(SAML_QUERY, {"org": org, "cursor": cursor}) provider = (data.get("organization") or {}).get("samlIdentityProvider") if provider is None: return identities or None connection = provider.get("externalIdentities") or {} for node in nodes_of(connection): login = login_of(node.get("user")) if not login: continue # provisioned in the IdP but never linked to a GitHub account saml = node.get("samlIdentity") or {} emails = [e.get("value") for e in saml.get("emails") or [] if e and e.get("value")] identities[login] = saml.get("nameId") or (emails[0] if emails else "") or \ (node.get("scimIdentity") or {}).get("username") or "" info = connection.get("pageInfo") or {} if not info.get("hasNextPage"): return identities cursor = info.get("endCursor") def load_profiles(gh, logins, org): profiles, with_domain_emails = {}, True pending, i = sorted(logins), 0 while i < len(pending): chunk = pending[i:i + 40] fields = "login name email" + (" organizationVerifiedDomainEmails(login: $org)" if with_domain_emails else "") header = "query($org: String!)" if with_domain_emails else "query" body = " ".join(f"u{n}: user(login: {json.dumps(login)}) {{ {fields} }}" for n, login in enumerate(chunk)) try: data, errors = gh.graphql(f"{header} {{ {body} }}", {"org": org} if with_domain_emails else {}) except GitHubError: if with_domain_emails: with_domain_emails = False continue data, errors = {}, [] if with_domain_emails and any("organizationVerifiedDomainEmails" in (e.get("message") or "") for e in errors): with_domain_emails = False continue for n, login in enumerate(chunk): node = data.get(f"u{n}") or {} profiles[login] = { "name": node.get("name") or "", "email": node.get("email") or "", "verified_emails": node.get("organizationVerifiedDomainEmails") or [], } i += len(chunk) return profiles # ---------------------------------------------------------------- collection def collect(gh, org, args, now, viewer): since = now - dt.timedelta(days=args.days) cov = Coverage() step(f"Reading organization {org}") try: org_info = gh.get(f"/orgs/{org}") except GitHubError as err: if err.status == 404: die(f"Organization '{org}' not found, or your login can't see it.") raise org = org_info.get("login", org) try: membership = gh.get(f"/user/memberships/orgs/{org}") except GitHubError: membership = {} is_owner = membership.get("role") == "admin" plan = ((org_info.get("plan") or {}).get("name") or "").lower() or None if not is_owner: warn("You're not an owner of this organization, so several checks will be skipped. Run it as an org owner for a full scan.") step("Listing members, owners and outside collaborators") members = [m["login"] for m in gh.paginate(f"/orgs/{org}/members")] owners = {m["login"] for m in gh.paginate(f"/orgs/{org}/members", {"role": "admin"})} cov.add("Members and owners", "checked", f"{len(members)} members, {len(owners)} owners") outside = guarded(cov, "Outside collaborators", lambda: [u["login"] for u in gh.paginate(f"/orgs/{org}/outside_collaborators")], lambda r: f"{len(r)} found") or [] tfa_required = org_info.get("two_factor_requirement_enabled") tfa_disabled, tfa_insecure, tfa_checked = set(), set(), False if tfa_required: cov.add("Two-factor authentication", "checked", "required for everyone by org policy") tfa_checked = True elif not is_owner: cov.add("Two-factor authentication", "skipped", "only org owners can see who has 2FA turned off") else: def load_tfa(): off = {u["login"] for u in gh.paginate(f"/orgs/{org}/members", {"filter": "2fa_disabled"})} off |= {u["login"] for u in gh.paginate(f"/orgs/{org}/outside_collaborators", {"filter": "2fa_disabled"})} return off result = guarded(cov, "Two-factor authentication", load_tfa, lambda r: f"{len(r)} without 2FA") if result is not None: tfa_disabled, tfa_checked = result, True if is_owner: try: tfa_insecure = {u["login"] for u in gh.paginate(f"/orgs/{org}/members", {"filter": "2fa_insecure"})} except GitHubError: pass invitations = guarded(cov, "Pending invitations", lambda: gh.list(f"/orgs/{org}/invitations"), lambda r: f"{len(r)} pending") or [] step("Reading teams") team_list = guarded(cov, "Teams", lambda: gh.list(f"/orgs/{org}/teams"), lambda r: f"{len(r)} teams") or [] def load_team(team): slug = team["slug"] return { "name": team.get("name") or slug, "members": [u["login"] for u in gh.paginate(f"/orgs/{org}/teams/{slug}/members")], "repos": {r["full_name"]: perm_rank(r.get("role_name"), r.get("permissions")) for r in gh.paginate(f"/orgs/{org}/teams/{slug}/repos")}, } team_results, team_errors = run_parallel(load_team, team_list, args.workers, "teams") teams = {team["slug"]: data for team, data in team_results} if team_list: cov.add("Team members and team access", part_status(len(team_errors), len(team_list)), f"{len(teams)} of {len(team_list)} teams read") step("Listing repositories") repo_list = gh.list(f"/orgs/{org}/repos", {"type": "all", "sort": "pushed", "direction": "desc"}) total_repos = len(repo_list) if args.max_repos and total_repos > args.max_repos: repo_list = repo_list[:args.max_repos] warn(f"Scanning only the {args.max_repos} most recently pushed of {total_repos} repositories (--max-repos).") repos = {r["full_name"]: r for r in repo_list} step(f"Reading collaborators and deploy keys for {len(repos)} repositories") def load_repo(repo): full, out = repo["full_name"], {} try: out["direct"] = [(c["login"], perm_rank(c.get("role_name"), c.get("permissions")), c.get("role_name") or "") for c in gh.paginate(f"/repos/{full}/collaborators", {"affiliation": "direct"})] except GitHubError: pass try: out["keys"] = gh.list(f"/repos/{full}/keys") except GitHubError: pass return out repo_results, _ = run_parallel(load_repo, repo_list, args.workers, "repositories") direct = {r["full_name"]: out["direct"] for r, out in repo_results if "direct" in out} keys = {r["full_name"]: out["keys"] for r, out in repo_results if "keys" in out} cov.add("Direct repository collaborators", part_status(len(repos) - len(direct), len(repos)), f"{len(direct)} of {len(repos)} repositories read") cov.add("Deploy keys", part_status(len(repos) - len(keys), len(repos)), f"{sum(len(k) for k in keys.values())} keys in {len(keys)} of {len(repos)} repositories") activity = {} if args.skip_activity: for _, label in ACTIVITY_SOURCES: cov.add(label, "not requested", "--skip-activity") else: live = [r for r in repo_list if not r.get("archived")] step(f"Reading the last {args.days} days of activity in {len(live)} active repositories") period = activity_period(args.days) page_cap = args.max_pages * 100 def load_activity(repo): full = repo["full_name"] owner_login, name = full.split("/", 1) out = {"push": {}, "push_count": {}, "pr": {}, "talk": {}, "truncated": [], "failed": []} def walk(source, path, params, stamp, handle): # Every list here is sorted newest-first by `stamp`, so stop at the window edge. try: count = 0 for item in gh.paginate(path, params, max_pages=args.max_pages): ts = parse_ts(item.get(stamp)) if ts is None or ts < since: return count += 1 handle(item, ts) if count >= page_cap: out["truncated"].append(source) except GitHubError as err: if err.status not in (409, 410): # empty repository, or issues turned off out["failed"].append(source) def on_push(event, ts): actor = login_of(event.get("actor")) if actor: record(out["push"], actor, ts, since) out["push_count"][actor] = out["push_count"].get(actor, 0) + 1 pushed_at = parse_ts(repo.get("pushed_at")) if pushed_at and pushed_at >= since: walk("pushes", f"/repos/{full}/activity", {"time_period": period}, "timestamp", on_push) newest_first = {"sort": "created", "direction": "desc"} walk("issues", f"/repos/{full}/issues", dict(newest_first, state="all"), "created_at", lambda item, ts: record(out["pr" if "pull_request" in item else "talk"], login_of(item.get("user")), ts, since)) walk("comments", f"/repos/{full}/issues/comments", newest_first, "created_at", lambda item, ts: record(out["talk"], login_of(item.get("user")), ts, since)) walk("review-comments", f"/repos/{full}/pulls/comments", newest_first, "created_at", lambda item, ts: record(out["pr"], login_of(item.get("user")), ts, since)) try: # approvals without comments only show up through GraphQL if page_graphql(gh, PR_QUERY, "pullRequests", owner_login, name, since, args.max_pages, lambda node: handle_pr(node, out, since)): out["truncated"].append("reviews") except GitHubError: out["failed"].append("reviews") return out activity_results, activity_errors = run_parallel(load_activity, live, args.workers, "activity") activity = {r["full_name"]: out for r, out in activity_results} for source, label in ACTIVITY_SOURCES: failed = sum(1 for out in activity.values() if source in out["failed"]) + len(activity_errors) capped = sum(1 for out in activity.values() if source in out["truncated"]) status = part_status(failed, len(live)) detail = f"{len(live) - failed} of {len(live)} non-archived repositories" if capped: status = "partial" if status == "checked" else status detail += f"; {capped} very busy ones only partly read (raise --max-pages)" cov.add(label, status, detail) step("Checking apps, Copilot, SSO identities and credentials") installs = guarded(cov, "Installed GitHub Apps", lambda: gh.list(f"/orgs/{org}/installations", key="installations"), lambda r: f"{len(r)} apps") or [] copilot = None try: seats = gh.list(f"/orgs/{org}/copilot/billing/seats", key="seats") copilot = {} for seat in seats: login = login_of(seat.get("assignee")) if login: copilot[login] = later(parse_ts(seat.get("last_activity_at")), copilot.get(login)) cov.add("Copilot seat activity", "checked", f"{len(seats)} seats") except GitHubError as err: cov.add("Copilot seat activity", "not available", "no Copilot Business seats, or no access" if err.status in (403, 404, 422) else err.reason()) saml = None try: saml = load_saml(gh, org) if saml is None: cov.add("SAML SSO identities", "not available", "no org-level SAML SSO (an Enterprise Cloud feature)") else: cov.add("SAML SSO identities", "checked", f"{len(saml)} linked identities") except GitHubError as err: cov.add("SAML SSO identities", "skipped", err.reason()) credentials = [] if saml is not None: credentials = guarded(cov, "SSO-authorized tokens and SSH keys", lambda: gh.list(f"/orgs/{org}/credential-authorizations"), lambda r: f"{len(r)} credentials") or [] else: cov.add("SSO-authorized tokens and SSH keys", "not available", "needs SAML SSO (Enterprise Cloud)") audit, audit_git = None, None if args.audit_log: def load_audit(): last, last_git, count = {}, {}, 0 params = {"include": "all", "phrase": f"created:>={since:%Y-%m-%d}", "order": "desc"} for event in gh.paginate(f"/orgs/{org}/audit-log", params, max_pages=args.audit_pages): count += 1 actor, ts = event.get("actor"), parse_ts(event.get("@timestamp")) if actor and ts: last[actor] = later(last.get(actor), ts) if str(event.get("action", "")).startswith("git."): last_git[actor] = later(last_git.get(actor), ts) return last, last_git, count result = guarded(cov, "Audit log", load_audit, lambda r: f"{r[2]} events read" + ("; capped at --audit-pages" if r[2] >= args.audit_pages * 100 else "") + "; git clone/fetch events are only kept 7 days") if result: audit, audit_git = result[0], result[1] else: cov.add("Audit log", "not requested", "add --audit-log to read it" if plan == "enterprise" else "API is Enterprise Cloud only") if plan == "enterprise": note("Tip: this org is on Enterprise Cloud. Add --audit-log to also see git clone/fetch activity.") cov.add("Fine-grained personal access tokens", "not available", "only a GitHub App can list them") everyone = set(members) | set(outside) | {login for rows in direct.values() for login, _, _ in rows} search_hits = {} if not args.skip_activity: seen = set() for out in activity.values(): for bucket in ("push", "pr", "talk"): seen |= set(out[bucket]) for source in (copilot, audit): seen |= {login for login, ts in (source or {}).items() if ts and ts >= since} seen |= {c.get("login") for c in credentials if parse_ts(c.get("credential_accessed_at")) and parse_ts(c.get("credential_accessed_at")) >= since} candidates = sorted(everyone - seen, key=str.lower) if candidates and args.skip_search_check: cov.add("Search double-check", "not requested", "--skip-search-check") elif candidates: step(f"Double-checking {len(candidates)} people with no activity found, using GitHub search") search_hits = confirm_with_search(gh, org, candidates, since) cov.add("Search double-check", "checked", f"{len(candidates)} people with no activity found; search found activity for {len(search_hits)}") uncertain = {full for full, out in activity.items() if {"pushes", "reviews"} & set(out["truncated"] + out["failed"])} step(f"Reading profiles for {len(everyone)} people") profiles = load_profiles(gh, everyone, org) return { "org": org, "org_info": org_info, "plan": plan, "is_owner": is_owner, "viewer": viewer, "now": now, "since": since, "days": args.days, "members": members, "owners": owners, "outside": outside, "tfa_required": tfa_required, "tfa_checked": tfa_checked, "tfa_disabled": tfa_disabled, "tfa_insecure": tfa_insecure, "invitations": invitations, "teams": teams, "repos": repos, "total_repos": total_repos, "direct": direct, "keys": keys, "activity": activity, "activity_scanned": not args.skip_activity, "search_hits": search_hits, "uncertain_repos": uncertain, "search_checked": not args.skip_activity and not args.skip_search_check, "installs": installs, "copilot": copilot, "saml": saml, "credentials": credentials, "audit": audit, "audit_git": audit_git, "profiles": profiles, "coverage": cov.rows, } # ---------------------------------------------------------------- mapping file def load_mapping(path): try: with open(path, newline="", encoding="utf-8-sig") as fh: text = fh.read() except OSError as err: die(f"Can't read mapping file {path}: {err}") first_line = text.splitlines()[0] if text else "" delimiter = max(",;\t", key=first_line.count) if first_line else "," reader = csv.DictReader(text.splitlines(), delimiter=delimiter) columns = {name.strip().lower(): name for name in reader.fieldnames or []} def pick(*names): return next((columns[n] for n in names if n in columns), None) login_col = pick("github_login", "login", "github", "github_username", "username") email_col = pick("employee_email", "email", "upn", "work_email", "mail") status_col = pick("employment_status", "status", "account_status", "enabled", "accountenabled") if not login_col: die(f"{path} needs a github_login column (found: {', '.join(columns) or 'nothing'}).") mapping = {} for row in reader: login = (row.get(login_col) or "").strip().lstrip("@") if not login: continue raw_status = (row.get(status_col) or "").strip().lower() if status_col else "" status = "left" if raw_status in LEFT_VALUES else "active" if raw_status in ACTIVE_VALUES else "unknown" mapping[login.lower()] = {"email": (row.get(email_col) or "").strip() if email_col else "", "status": status} note(f"Loaded {len(mapping)} people from {path}") return mapping # ---------------------------------------------------------------- analysis def fix_remove_from_org(org, login, kind): if kind == "outside collaborator": return f"gh api -X DELETE /orgs/{org}/outside_collaborators/{login}" return f"gh api -X DELETE /orgs/{org}/memberships/{login}" def finding(severity, category, subject, title, detail, recommendation, fixes=()): return {"severity": severity, "category": category, "subject": subject, "title": title, "detail": detail, "recommendation": recommendation, "fixes": list(fixes)} def describe_via(via): labels = [] for source in via: label = "direct" if source["source"] == "direct" else f"team {source['team']}" labels.append(f"{label} ({PERMISSION_NAME.get(source['rank'], '?')})") return ", ".join(labels) def analyze(scan, mapping): org, now, since, days = scan["org"], scan["now"], scan["since"], scan["days"] owners, members, outside = scan["owners"], set(scan["members"]), set(scan["outside"]) repos = scan["repos"] base_name = scan["org_info"].get("default_repository_permission") base_rank = BASE_RANK.get(base_name or "", 0) findings = [] # Explicit grants: direct collaborators and team access. grants = {} def add_grant(login, full, source, rank, team=None): grant = grants.setdefault((login, full), {"rank": 0, "via": []}) grant["via"].append({"source": source, "team": team, "rank": rank}) grant["rank"] = max(grant["rank"], rank) for full, rows in scan["direct"].items(): for login, rank, _role in rows: add_grant(login, full, "direct", rank) for slug, team in scan["teams"].items(): for full, rank in team["repos"].items(): if full in repos: for login in team["members"]: add_grant(login, full, "team", rank, slug) # Activity per person. last = defaultdict(dict) pushes = defaultdict(int) write_used = defaultdict(set) for full, act in scan["activity"].items(): for bucket in ("push", "pr", "talk"): for login, ts in act[bucket].items(): last[login][bucket] = later(last[login].get(bucket), ts) if bucket != "talk": write_used[login].add(full) for login, count in act["push_count"].items(): pushes[login] += count for bucket, source in (("audit", scan["audit"]), ("git", scan["audit_git"]), ("copilot", scan["copilot"])): for login, ts in (source or {}).items(): if ts: last[login][bucket] = later(last[login].get(bucket), ts) for cred in scan["credentials"]: login, ts = cred.get("login"), parse_ts(cred.get("credential_accessed_at")) if login and ts: last[login]["credential"] = later(last[login].get("credential"), ts) for login, (ts, _what) in scan["search_hits"].items(): if ts: last[login]["search"] = later(last[login].get("search"), ts) grants_by_login = defaultdict(list) for (login, full), grant in grants.items(): grants_by_login[login].append((full, grant)) has_identity_source = bool(mapping) or scan["saml"] is not None people = [] for login in sorted(members | outside | {l for l, _ in grants}, key=str.lower): kind = ("owner" if login in owners else "member" if login in members else "outside collaborator" if login in outside else "collaborator") profile = scan["profiles"].get(login, {}) my_grants = sorted(grants_by_login.get(login, []), key=lambda g: (-g[1]["rank"], g[0])) signals = last.get(login, {}) last_seen = None for ts in signals.values(): last_seen = later(last_seen, ts) active_in_window = bool(last_seen and last_seen >= since) email, identity_source, status = "", "", "unknown" mapped = (mapping or {}).get(login.lower()) if mapped: email, status, identity_source = mapped["email"], mapped["status"], "mapping file" if not email and scan["saml"] and login in scan["saml"]: email, identity_source = scan["saml"][login], "SAML SSO" if not email and profile.get("verified_emails"): email, identity_source = profile["verified_emails"][0], "verified domain email" implicit = "" if kind == "owner": implicit = "admin on every repo (org owner)" elif kind == "member" and base_rank: implicit = f"{base_name} on every repo (org default)" # A member's grant only adds access if it beats the org's base permission. writable = [(full, g) for full, g in my_grants if g["rank"] >= 3 and not repos[full].get("archived") and (kind != "member" or g["rank"] > base_rank)] # Repos we only partly read can't prove access is unused. unused_write = [(full, g) for full, g in writable if full not in write_used.get(login, set()) and full not in scan["uncertain_repos"]] can_write = kind == "owner" or (kind == "member" and base_rank >= 3) or bool(writable) if status == "left": verdict = "departed" elif not scan["activity_scanned"] and not active_in_window: verdict = "unknown" elif not active_in_window: verdict = "dormant" elif unused_write and kind != "owner": verdict = "over-privileged" else: verdict = "active" tfa = ("off" if login in scan["tfa_disabled"] else "SMS" if login in scan["tfa_insecure"] else "on" if scan["tfa_checked"] else "unknown") person = { "login": login, "name": profile.get("name", ""), "kind": kind, "verdict": verdict, "employee_email": email, "employment_status": status, "identity_source": identity_source, "public_email": profile.get("email", ""), "two_factor": tfa, "last_seen": iso(last_seen), "active_in_window": active_in_window, "signals": {k: iso(v) for k, v in sorted(signals.items(), key=lambda kv: kv[1], reverse=True)}, "pushes_in_window": pushes.get(login, 0), "implicit_access": implicit, "admin_repos": sum(1 for _, g in my_grants if g["rank"] >= 5), "write_repos": sum(1 for _, g in my_grants if g["rank"] >= 3), "read_repos": sum(1 for _, g in my_grants if g["rank"] < 3), "unused_write_repos": [full for full, _ in unused_write], } people.append(person) last_seen_text = (f"Last seen {fmt_date(last_seen)} ({SIGNAL_LABELS.get(max(signals, key=signals.get), '')})." if last_seen else "No activity found at all.") remove = fix_remove_from_org(org, login, kind) if verdict == "departed": findings.append(finding( "critical", "departed", login, "Has left the company but still has access", f"The mapping file marks {email or login} as {status}. Access: {implicit or 'no org-wide access'}; " f"write or admin on {person['write_repos']} repositories. {last_seen_text}", "Remove them from the organization now, then check their deploy keys and tokens.", [remove])) elif verdict == "dormant": readers_note = ("Read-only use (cloning, browsing code) doesn't show up here" + ("" if scan["audit"] is not None else " without the Enterprise Cloud audit log") + ", so confirm with the person or their manager before removing.") findings.append(finding( "high" if can_write else "medium", "dormant", login, f"No activity in {days} days" + (" (org owner)" if kind == "owner" else ""), f"No pushes, pull requests, reviews, issues or comments since {fmt_date(since)}" + (", and GitHub search found none either. " if scan["search_checked"] else ". ") + f"{last_seen_text} " f"Access: {implicit or 'explicit grants only'}; write or admin on {person['write_repos']} repositories. " + readers_note, ("Demote to member or remove. Dormant owners are the most valuable account to take over." if kind == "owner" else "Confirm they still need access; otherwise remove them."), ([f"gh api -X PUT /orgs/{org}/memberships/{login} -f role=member"] if kind == "owner" else []) + [remove])) elif verdict == "over-privileged": shown = unused_write[:15] more = len(unused_write) - len(shown) listing = "; ".join(f"{full.split('/', 1)[1]} via {describe_via(g['via'])}" for full, g in shown) fixes = [] used_repos = write_used.get(login, set()) for full, grant in shown: for source in grant["via"]: if source["rank"] < 3: continue if source["source"] == "direct": fixes.append(f"gh api -X PUT /repos/{full}/collaborators/{login} -f permission=pull") elif not used_repos & set(scan["teams"].get(source["team"], {}).get("repos", {})): # Only suggest leaving a team when they use none of its repositories. fixes.append(f"gh api -X DELETE /orgs/{org}/teams/{source['team']}/memberships/{login}" f" # removes them from team {source['team']} and all its repos") findings.append(finding( "low", "unused-write", login, f"Write access to {len(unused_write)} of {len(writable)} repositories they haven't written to in {days} days", listing + (f"; and {more} more" if more > 0 else "") + ".", "Downgrade to read where they no longer push, review or merge. Team access changes for the whole " "team, so lower the team's permission on those repositories or move them to a narrower team.", list(dict.fromkeys(fixes)))) if kind == "outside collaborator": ranks = [(full, g) for full, g in my_grants] top = max((g["rank"] for _, g in ranks), default=0) listing = ", ".join(f"{full.split('/', 1)[1]} ({PERMISSION_NAME.get(g['rank'], '?')})" for full, g in ranks[:15]) findings.append(finding( "high" if top >= 5 else "medium", "outside-collaborator", login, "Outside collaborator" + (" with admin rights" if top >= 5 else ""), f"Not an org member, but has access to {len(ranks)} repositories: {listing or 'none listed'}.", "Make sure someone inside the company sponsors this access and that it has an end date.", [remove])) if tfa == "off": findings.append(finding( "high", "2fa", login, "Two-factor authentication is off", "A stolen password is enough to take over this account and everything it can reach.", "Ask them to turn on 2FA, then require 2FA for the whole org.")) elif tfa == "SMS": findings.append(finding( "medium", "2fa", login, "Uses SMS for two-factor authentication", "SMS codes can be intercepted through SIM swapping.", "Ask them to switch to an authenticator app, a passkey or a security key.")) if has_identity_source and not email and kind != "collaborator": findings.append(finding( "medium", "unmapped", login, "Not linked to an employee", "No match in the mapping file" + (" and no linked SAML SSO identity" if scan["saml"] is not None else "") + ".", "Find out who this account belongs to. If nobody claims it, remove it.")) # Organization-wide settings. if scan["tfa_required"] is False: missing = f" {len(scan['tfa_disabled'])} people currently have it turned off." if scan["tfa_checked"] else "" findings.append(finding( "high", "org-setting", org, "Two-factor authentication isn't required", f"The organization doesn't require 2FA.{missing}", "Ask everyone to turn on 2FA, then require it under Settings > Authentication security. " "Turning on the requirement removes everyone who still doesn't have 2FA.")) if base_rank >= 3: findings.append(finding( "high", "org-setting", org, f"Every member gets {base_name} access to every repository", f"The org's base permission is '{base_name}', so all {len(members)} members can change every repository.", "Set the base permission to read (or none) and grant write through teams.", [f"gh api -X PATCH /orgs/{org} -f default_repository_permission=read"])) owner_limit = max(3, math.ceil(len(members) * 0.1)) if len(owners) > owner_limit: findings.append(finding( "medium", "org-setting", org, f"{len(owners)} organization owners", f"Owners can delete repositories, change security settings and remove people: {', '.join(sorted(owners, key=str.lower))}.", f"Keep owners to a small group (two or three). Demote the rest to member.")) if not has_identity_source: findings.append(finding( "info", "identity", org, "Accounts aren't linked to employees", "Without a mapping file or SAML SSO, Ghostbuster can't tell who has left the company.", "Fill in employee_email and employment_status in people.csv and re-run with --mapping people.csv.")) # Invitations. for invite in scan["invitations"]: created = parse_ts(invite.get("created_at")) age = (now - created).days if created else None who = invite.get("login") or invite.get("email") or "unknown" admin = invite.get("role") == "admin" findings.append(finding( "medium" if admin else "low", "invitation", who, "Pending owner invitation" if admin else "Pending invitation", f"Invited by {login_of(invite.get('inviter')) or 'unknown'}" + (f" {age} days ago" if age is not None else "") + f" as {invite.get('role') or 'member'}.", "Cancel it if it's no longer expected.", [f"gh api -X DELETE /orgs/{org}/invitations/{invite.get('id')}"])) # Deploy keys. deploy_keys = [] for full, keys in scan["keys"].items(): for key in keys: created, used = parse_ts(key.get("created_at")), parse_ts(key.get("last_used")) can_push = not key.get("read_only", True) unused = (used is None and created and created < since) or (used is not None and used < since) deploy_keys.append({"repo": full, "id": key.get("id"), "title": key.get("title", ""), "can_push": can_push, "unused": bool(unused), "created_at": iso(created), "last_used": iso(used)}) if can_push or unused: problems = [] if can_push: problems.append("can push code") if unused: problems.append("never used" if used is None else f"last used {fmt_date(used)}") findings.append(finding( "medium" if can_push else "low", "deploy-key", full, f"Deploy key '{key.get('title', '')}' " + " and ".join(problems), f"Created {fmt_date(created) or 'unknown'}; last used {fmt_date(used) or 'never'}.", "Delete it if nothing uses it; otherwise replace it with a read-only key.", [f"gh api -X DELETE /repos/{full}/keys/{key.get('id')}"])) # SSO-authorized credentials (Enterprise Cloud with SAML). for cred in scan["credentials"]: accessed = parse_ts(cred.get("credential_accessed_at")) if accessed and accessed >= since: continue findings.append(finding( "low", "credential", cred.get("login") or "unknown", f"Unused {str(cred.get('credential_type') or 'credential').replace('_', ' ')}", f"Authorized for SSO on {fmt_date(parse_ts(cred.get('credential_authorized_at'))) or 'unknown'}; " f"last used {fmt_date(accessed) or 'never'}.", "Revoke its SSO authorization.", [f"gh api -X DELETE /orgs/{org}/credential-authorizations/{cred.get('credential_id')}"])) # Installed apps. apps = [] for install in scan["installs"]: perms = install.get("permissions") or {} write = sorted(k for k, v in perms.items() if v in ("write", "admin")) risky = sorted(set(write) & SENSITIVE_APP_PERMISSIONS) slug = install.get("app_slug") or str(install.get("app_id")) apps.append({"app": slug, "repositories": install.get("repository_selection"), "write_permissions": write, "created_at": install.get("created_at"), "suspended": bool(install.get("suspended_at"))}) if risky and install.get("repository_selection") == "all" and not install.get("suspended_at"): findings.append(finding( "medium", "app", slug, "App can change every repository", f"Installed on all repositories with write access to: {', '.join(risky)}.", "Check it's still used and limit it to the repositories it needs.")) # Access rows for the CSV. access = [] kinds = {p["login"]: p["kind"] for p in people} for (login, full), grant in sorted(grants.items(), key=lambda kv: (kv[0][1].lower(), kv[0][0].lower())): act = scan["activity"].get(full, {}) last_write = later((act.get("push") or {}).get(login), (act.get("pr") or {}).get(login)) access.append({ "repository": full, "github_login": login, "kind": kinds.get(login, ""), "permission": PERMISSION_NAME.get(grant["rank"], "?"), "via": describe_via(grant["via"]), "last_write_activity": iso(last_write), "archived": bool(repos[full].get("archived")), "visibility": repos[full].get("visibility") or ("private" if repos[full].get("private") else "public"), }) # Per-repository summary: explicit grants on top of org ownership. unused_by_repo = Counter(full for p in people for full in p["unused_write_repos"]) grants_by_repo = defaultdict(list) for (login, full), grant in grants.items(): if login not in owners: grants_by_repo[full].append((login, grant)) repositories = [] for full, repo in repos.items(): explicit = grants_by_repo.get(full, []) repositories.append({ "repository": full, "visibility": repo.get("visibility") or ("private" if repo.get("private") else "public"), "archived": bool(repo.get("archived")), "pushed_at": repo.get("pushed_at"), "admins": sorted((login for login, g in explicit if g["rank"] >= 5), key=str.lower), "writers": sum(1 for _, g in explicit if g["rank"] >= 3), "outside": sorted((login for login, _ in explicit if login in outside), key=str.lower), "unused_write": unused_by_repo.get(full, 0), "partly_read": full in scan["uncertain_repos"], }) repositories.sort(key=lambda r: (-len(r["outside"]), -len(r["admins"]), -r["writers"], r["repository"].lower())) findings.sort(key=lambda f: (SEVERITIES.index(f["severity"]), f["category"], f["subject"].lower())) people.sort(key=lambda p: (VERDICT_ORDER[p["verdict"]], p["login"].lower())) counts = {s: sum(1 for f in findings if f["severity"] == s) for s in SEVERITIES} return { "tool": {"name": "ghostbuster", "version": VERSION, "schema": 1}, "org": org, "plan": scan["plan"], "scanned_by": scan["viewer"], "scanned_as_owner": scan["is_owner"], "scanned_at": iso(now), "window_days": days, "window_start": iso(since), "settings": { "two_factor_required": scan["tfa_required"], "base_permission": base_name, }, "summary": { "people": len(people), "owners": len(owners), "members": len(members), "outside_collaborators": len(outside), "possible_ghosts": sum(1 for p in people if p["verdict"] in ("departed", "dormant")), "repositories": len(repos), "repositories_in_org": scan["total_repos"], "findings": counts, }, "owners": sorted(owners, key=str.lower), "people": people, "access": access, "findings": findings, "repositories": repositories, "invitations": [{"invitee": i.get("login") or i.get("email"), "role": i.get("role"), "invited_by": login_of(i.get("inviter")), "created_at": i.get("created_at")} for i in scan["invitations"]], "deploy_keys": deploy_keys, "apps": apps, "coverage": scan["coverage"], } # ---------------------------------------------------------------- output LIMITATIONS = [ "Read-only use (cloning, browsing code) is invisible to GitHub's API on Free and Team plans. " "On Enterprise Cloud, --audit-log adds git clone and fetch events, but GitHub keeps those for only 7 days.", "Activity means pushes, merges, branch changes, pull requests, reviews, issues and comments inside this organization.", "Fine-grained personal access tokens can only be listed by a GitHub App, not by a CLI login.", "Without a mapping file (or SAML SSO on Enterprise Cloud), Ghostbuster can't tell who has left the company.", ] def csv_safe(value): """Stops spreadsheet apps from running cell values (names come from GitHub users) as formulas.""" text = "" if value is None else str(value) return "'" + text if text[:1] in ("=", "+", "-", "@", "\t", "\r") else text def write_csv(path, rows, columns): with open(path, "w", newline="", encoding="utf-8-sig") as fh: writer = csv.writer(fh) writer.writerow(columns) for row in rows: writer.writerow([csv_safe(row.get(c)) for c in columns]) def render_fixes(rep): lines = [ "#!/usr/bin/env bash", f"# Ghostbuster suggested fixes for {rep['org']}, generated {rep['scanned_at'][:10]}", "#", "# NOTHING IN THIS FILE RUNS AS-IS: every command is commented out.", "# Review each one (ideally with the person's manager), remove the leading '# '", "# from the ones you agree with, then run the file or paste single lines into", "# any terminal where gh is signed in (bash, zsh or PowerShell).", "#", "# These commands change your organization, so your gh login needs write scopes:", "# gh auth refresh -h github.com -s admin:org,repo", "#", "# Removing someone from the org also removes their forks of private repositories.", "", ] for item in rep["findings"]: if not item["fixes"]: continue lines.append(f"# [{item['severity'].upper()}] {item['title']} - {item['subject']}") lines += [f"# {command}" for command in item["fixes"]] lines.append("") if len(lines) == 13: lines.append("# No fixes to suggest.") return "\n".join(lines) + "\n" def write_outputs(rep, out_dir): os.makedirs(out_dir, exist_ok=True) with open(os.path.join(out_dir, "report.json"), "w", encoding="utf-8") as fh: json.dump(rep, fh, indent=2, ensure_ascii=False) people_rows = [] for p in rep["people"]: row = dict(p) row["github_login"] = p["login"] row["signals"] = "; ".join(f"{k} {v[:10]}" for k, v in p["signals"].items()) row["unused_write_repos"] = " ".join(p["unused_write_repos"]) row["last_seen"] = (p["last_seen"] or "")[:10] people_rows.append(row) write_csv(os.path.join(out_dir, "people.csv"), people_rows, ["github_login", "employee_email", "employment_status", "name", "kind", "verdict", "last_seen", "signals", "pushes_in_window", "two_factor", "implicit_access", "admin_repos", "write_repos", "read_repos", "unused_write_repos", "identity_source", "public_email"]) write_csv(os.path.join(out_dir, "access.csv"), rep["access"], ["repository", "github_login", "kind", "permission", "via", "last_write_activity", "archived", "visibility"]) findings_rows = [dict(f, fix_commands=" ; ".join(f["fixes"])) for f in rep["findings"]] write_csv(os.path.join(out_dir, "findings.csv"), findings_rows, ["severity", "category", "subject", "title", "detail", "recommendation", "fix_commands"]) with open(os.path.join(out_dir, "suggested-fixes.sh"), "w", encoding="utf-8", newline="\n") as fh: fh.write(render_fixes(rep)) with open(os.path.join(out_dir, "report.html"), "w", encoding="utf-8") as fh: fh.write(render_html(rep)) LIGHT_TOKENS = ("--bg:#f5f6f3;--paper:#fff;--surface-2:#f1f3ef;--text:#1b1d1b;--muted:#5e625d;--border:#e1e4de;" "--accent:#365b3f;--accent-soft:#e6efe8;--brand-light:#74ad82;--slate:#3d5a80;--slate-soft:#e6edf6;" "--ok:#1b7745;--ok-soft:#e2f2e8;--warn:#985800;--warn-soft:#fbefd9;" "--bad:#b42318;--bad-soft:#fbe5e2;--crit:#6b0f0b;--neutral-soft:#ecedea;--code:#f1f3ef") DARK_TOKENS = ("--bg:#111311;--paper:#1a1c1a;--surface-2:#232622;--text:#eceeea;--muted:#a1a59f;--border:#323630;" "--accent:#74ad82;--accent-soft:#1f3326;--brand-light:#74ad82;--slate:#9db8dc;--slate-soft:#1e2a3a;" "--ok:#6fd39a;--ok-soft:#173325;--warn:#f0b85a;--warn-soft:#3a2c12;" "--bad:#ff8a80;--bad-soft:#3d1c19;--crit:#ffb4ab;--neutral-soft:#292b28;--code:#292b28") REPORT_CSS = """ :root{%LIGHT%} @media (prefers-color-scheme:dark){:root:not([data-theme="light"]){%DARK%}} :root[data-theme="dark"]{%DARK%} *{box-sizing:border-box} html{-webkit-text-size-adjust:100%} body{margin:0;background:var(--bg);color:var(--text);font:15px/1.55 Inter,-apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,"Helvetica Neue",Arial,sans-serif} .page{max-width:1080px;margin:24px auto 64px;padding:0 16px} .sheet{background:var(--paper);border:1px solid var(--border);border-radius:16px;padding:0 44px 40px;overflow:hidden} .brandbar{height:5px;margin:0 -44px 28px;background:linear-gradient(90deg,var(--accent),var(--brand-light))} .top{display:flex;justify-content:space-between;align-items:center;gap:16px;padding-bottom:18px;border-bottom:1px solid var(--border)} .top a{display:block;color:var(--accent)} .adcyma-logo{display:block;height:26px;width:auto;fill:currentColor} .eyebrow{display:flex;align-items:center;gap:8px;margin-top:26px;color:var(--accent);font-weight:650;font-size:.82rem;letter-spacing:.08em;text-transform:uppercase} .eyebrow svg{width:24px;height:24px} .eyebrow .sep{color:var(--muted);font-weight:400} .eyebrow .what{color:var(--muted);font-weight:600} h1{font-size:clamp(1.7rem,4.4vw,2.4rem);margin:6px 0 0;letter-spacing:-.02em;line-height:1.15;overflow-wrap:anywhere} .btn{font:inherit;font-weight:600;font-size:.88rem;padding:8px 16px;border-radius:10px;border:1px solid var(--accent);background:var(--accent);color:var(--paper);cursor:pointer;white-space:nowrap} .meta{display:grid;grid-template-columns:repeat(4,minmax(0,1fr));gap:12px;margin:22px 0 0;padding:16px 0;border-top:1px solid var(--border);border-bottom:1px solid var(--border)} .meta dt{font-size:.72rem;text-transform:uppercase;letter-spacing:.06em;color:var(--muted)} .meta dd{margin:2px 0 0;font-weight:600} nav.toc{display:flex;flex-wrap:wrap;gap:6px 16px;font-size:.86rem;margin:14px 0 0} nav.toc a{color:var(--muted);text-decoration:none} nav.toc a:hover{color:var(--accent);text-decoration:underline} section{margin-top:40px} .h{display:flex;align-items:baseline;gap:10px;border-bottom:2px solid var(--text);padding-bottom:6px;margin-bottom:12px} .h .num{font-weight:700;color:var(--accent);font-variant-numeric:tabular-nums;white-space:nowrap} h2{font-size:1.25rem;margin:0;line-height:1.3} .h .count{margin-left:auto;color:var(--muted);font-size:.85rem;font-variant-numeric:tabular-nums} h3{font-size:.98rem;margin:24px 0 8px} p{margin:0 0 12px} .intro{color:var(--muted);max-width:76ch} .lead{font-size:1.08rem;max-width:78ch} .tiles{display:grid;grid-template-columns:repeat(5,minmax(0,1fr));gap:10px;margin:18px 0 4px} .tile{border:1px solid var(--border);border-radius:12px;padding:12px 14px} .tile .n{font-size:1.7rem;font-weight:750;line-height:1.1;font-variant-numeric:tabular-nums} .tile .l{font-size:.8rem;color:var(--muted)} .tile.alert .n{color:var(--bad)} .bar{display:flex;height:14px;border-radius:7px;overflow:hidden;background:var(--surface-2);margin:8px 0} .seg{display:block;height:100%} .seg+.seg{border-left:2px solid var(--paper)} .legend{display:flex;flex-wrap:wrap;gap:6px 18px;font-size:.84rem;color:var(--muted)} .legend i{display:inline-block;width:10px;height:10px;border-radius:3px;margin-right:6px;vertical-align:-1px} .legend strong{color:var(--text);margin-left:4px} .v-departed{background:var(--crit)}.v-dormant{background:var(--bad)}.v-over-privileged{background:var(--slate)}.v-unknown{background:var(--muted)}.v-active{background:var(--ok)} ol.actions{list-style:none;padding:0;margin:8px 0 0;counter-reset:a;border:1px solid var(--border);border-radius:12px} ol.actions li{counter-increment:a;display:grid;grid-template-columns:24px 92px 1fr auto;gap:10px;align-items:center;padding:10px 14px;border-bottom:1px solid var(--border)} ol.actions li:last-child{border-bottom:none} ol.actions .pill{justify-self:start} ol.actions li::before{content:counter(a);font-weight:700;color:var(--muted);font-variant-numeric:tabular-nums} ol.actions a{font-size:.82rem;color:var(--accent);text-decoration:none;white-space:nowrap} .wrap{overflow-x:auto;border:1px solid var(--border);border-radius:12px} table{border-collapse:collapse;width:100%;font-size:.86rem} th,td{text-align:left;vertical-align:top;padding:8px 12px;border-bottom:1px solid var(--border)} th{background:var(--surface-2);color:var(--muted);font-size:.7rem;text-transform:uppercase;letter-spacing:.05em;white-space:nowrap} tbody tr:last-child td{border-bottom:none} td strong{font-weight:650} .sub{color:var(--muted);font-size:.78rem} .sub-inline{color:var(--muted);font-size:.78rem} .muted{color:var(--muted)} .empty{color:var(--muted);font-style:italic} .pill{display:inline-block;font-size:.7rem;font-weight:700;padding:2px 8px;border-radius:999px;white-space:nowrap;text-transform:uppercase;letter-spacing:.03em} .critical,.departed{background:var(--bad);color:var(--paper)} .high,.dormant,.off,.fix,.skipped{background:var(--bad-soft);color:var(--bad)} .medium,.sms,.partial{background:var(--warn-soft);color:var(--warn)} .low,.over-privileged{background:var(--slate-soft);color:var(--slate)} .info,.unknown,.not-available,.not-requested{background:var(--neutral-soft);color:var(--muted)} .active,.on,.good,.checked{background:var(--ok-soft);color:var(--ok)} .callout{border:1px solid var(--border);background:var(--surface-2);border-radius:12px;padding:14px 18px} ul.limits{margin:0;padding-left:20px} ul.limits li{margin:4px 0} details.fixes{margin-top:10px} details.fixes summary{cursor:pointer;color:var(--accent);font-size:.84rem} pre{font:12px/1.5 ui-monospace,SFMono-Regular,Consolas,monospace;background:var(--code);padding:10px 12px;border-radius:8px;overflow-x:auto;margin:8px 0} code{font:12px/1.4 ui-monospace,SFMono-Regular,Consolas,monospace} footer.about{margin-top:52px} .cta{display:grid;grid-template-columns:1fr auto;gap:20px 32px;align-items:center;padding:26px 28px;border-radius:14px;background:var(--accent-soft);border:1px solid var(--border)} .cta .adcyma-logo{height:24px;color:var(--accent);margin-bottom:14px} .cta h2{font-size:1.35rem;margin:0 0 6px} .cta p{margin:0;max-width:62ch;color:var(--muted)} .cta-btn{display:inline-block;text-decoration:none;font-size:.95rem;padding:11px 20px} .cta-btn span{margin-left:6px} .fineprint{margin:14px 4px 0;color:var(--muted);font-size:.8rem} .fineprint a{color:var(--accent)} @media screen and (max-width:760px){ .sheet{padding:0 16px 22px;border-radius:12px} .page{margin-top:12px} .cta{grid-template-columns:1fr;padding:22px 20px} .brandbar{margin:0 -16px 20px} .meta{grid-template-columns:1fr 1fr} .tiles{grid-template-columns:1fr 1fr} ol.actions li{display:flex;flex-wrap:wrap;gap:6px 10px} th,td{padding:7px 9px} } @page{size:A4;margin:14mm 12mm 16mm; @bottom-left{content:"Ghostbuster by Adcyma · adcyma.com/tools";font:8pt -apple-system,"Segoe UI",Roboto,Arial,sans-serif;color:#777} @bottom-right{content:"%ORG% · page " counter(page) " of " counter(pages);font:8pt -apple-system,"Segoe UI",Roboto,Arial,sans-serif;color:#777}} @media print{ :root,:root:not([data-theme="light"]),:root[data-theme="dark"]{%LIGHT%} *{-webkit-print-color-adjust:exact;print-color-adjust:exact} body{background:#fff;font-size:10pt} .page{max-width:none;margin:0;padding:0} .sheet{border:none;border-radius:0;padding:0;overflow:visible} .brandbar{margin:0 0 18px} footer.about{break-inside:avoid} .cta-btn span{display:none} .no-print{display:none!important} section{margin-top:24px} .h,.intro,h3{break-after:avoid} tr,.tile,ol.actions li,.callout,.meta{break-inside:avoid} thead{display:table-header-group} .wrap{overflow:visible} table{font-size:8.5pt} th,td{padding:5px 8px} .appendix-start{break-before:page} a{color:inherit;text-decoration:none} } """ # Adcyma logo from frontend/public/images, coloured through currentColor. ADCYMA_LOGO = ( '') GHOST_SVG = """""" VERDICT_LABELS = (("departed", "Left the company"), ("dormant", "No activity"), ("over-privileged", "Unused write access"), ("unknown", "Activity not checked"), ("active", "Active")) VERDICT_PILL = {"departed": "Left", "dormant": "Inactive", "over-privileged": "Unused access", "unknown": "Not checked", "active": "Active"} REPORT_FILES = ( ("report.html", "This report. Use Save as PDF to keep a copy."), ("people.csv", "One row per person. Fill in employee_email and employment_status, then re-run with --mapping."), ("access.csv", "Every explicit grant: who has which permission on which repository, and where it comes from."), ("findings.csv", "Every finding with its recommendation and fix command."), ("suggested-fixes.sh", "gh commands that fix the findings. All commented out: review, uncomment, run."), ("report.json", "Everything above, machine-readable."), ) def _plural(n, one, many): return f"{n} {one if n == 1 else many}" def render_html(rep): e = html.escape s = rep["summary"] org, days = rep["org"], rep["window_days"] plan = PLAN_LABEL.get(rep["plan"] or "", rep["plan"] or "Unknown") now = parse_ts(rep["scanned_at"]) date = rep["scanned_at"][:10] settings = rep["settings"] people = rep["people"] by_login = {p["login"]: p for p in people} access_by_login = defaultdict(list) for row in rep["access"]: access_by_login[row["github_login"]].append(row) access_index = {(r["github_login"], r["repository"]): r for r in rep["access"]} by_cat = defaultdict(list) for item in rep["findings"]: by_cat[item["category"]].append(item) coverage = {c["check"]: c for c in rep["coverage"]} activity_checked = (coverage.get(ACTIVITY_SOURCES[0][1]) or {}).get("status") != "not requested" verdicts = Counter(p["verdict"] for p in people) def pill(text, cls=None): return f'{e(text)}' def verdict_pill(p): return pill(VERDICT_PILL[p["verdict"]], p["verdict"]) def seen(value): ts = parse_ts(value) if not ts: return 'never' return f'{e(fmt_date(ts))}
{e(empty)}
' th = "".join(f"{e(chr(10).join(commands))}"
'The same commands are in suggested-fixes.sh. Review each one before you run it.
people.csv, fill in "
"employee_email and employment_status (active or left), then run again "
"with --mapping people.csv. On Enterprise Cloud with SAML single sign-on, accounts "
"link automatically.{linked} of {len(people)} accounts are linked to an employee.
" + table(["Account", "Type", "Last seen", "Access"], rows, "Every account is linked to an employee.") + fixes("unmapped")) count = len(unmapped) add("identity", "Linked to employees", "Whether each GitHub account can be tied to a person in your company.", body, count) # Apps, deploy keys and tokens risky_apps = {item["subject"] for item in by_cat.get("app", [])} app_rows = [tr(f"{e(a['app'])}" + (' suspended' if a["suspended"] else ""), e(a["repositories"] or ""), e(", ".join(a["write_permissions"]) or "read only"), e((a["created_at"] or "")[:10]), pill("review", "fix") if a["app"] in risky_apps else pill("ok", "good")) for a in rep["apps"]] key_rows = [] for key in rep["deploy_keys"]: flags = (["can push"] if key["can_push"] else []) + (["unused"] if key.get("unused") else []) key_rows.append(tr(e(key["repo"]), e(key["title"]), e((key["last_used"] or "never")[:10]), e((key["created_at"] or "")[:10]), pill(", ".join(flags) or "ok", "fix" if flags else "good"))) cred_rows = [tr(who(item["subject"]), e(item["title"]), e(item["detail"])) for item in by_cat.get("credential", [])] body = ("{e(name)}", e(what)) for name, what in REPORT_FILES]
add("method", "How this was checked",
f"Ghostbuster {VERSION} read this organization through GitHub's API with the account that ran it. "
"It changed nothing.",
table(["Check", "Status", "Details"], rows, "")
+ "Nothing to do.
' total = max(1, len(people)) segments = "".join(f'' for v, _ in VERDICT_LABELS if verdicts[v]) legend = "".join(f'{e(label)}{verdicts[v]}' for v, label in VERDICT_LABELS if verdicts[v]) serious = s["findings"]["critical"] + s["findings"]["high"] tiles = [ (s["people"], "people with access", False), (s["possible_ghosts"], "possible ghosts", s["possible_ghosts"] > 0), (s["outside_collaborators"], "outside collaborators", False), (s["owners"], "org owners", too_many_owners), (serious, "critical or high findings", serious > 0), ] tiles_html = "".join(f'{e(sec["intro"])}
{sec["body"]}{' '.join(sentences)}