#!/usr/bin/env python3 import json import os import random import re import struct import subprocess import sys import tempfile import urllib.parse import zlib from datetime import datetime, timedelta, timezone from http.server import ThreadingHTTPServer, BaseHTTPRequestHandler PORT = int(os.environ.get("PORT", "8000")) CLIENT_ID = os.environ.get("GITHUB_CLIENT_ID", "ghstub-client-id") CLIENT_SECRET = os.environ.get("GITHUB_CLIENT_SECRET", "ghstub-client-secret") APP_SLUG = os.environ.get("GITHUB_APP_SLUG", "tangled-dev-stub") OBJECT_FORMAT = os.environ.get("GITHUB_STUB_OBJECT_FORMAT", "sha1") REPOS_DIR = os.environ.get("REPOS_DIR", os.path.join(tempfile.gettempdir(), "github-stub-repos")) PUBLIC_ORIGIN = os.environ.get("GITHUB_PUBLIC_ORIGIN", "https://github.com").rstrip("/") API_ORIGIN = os.environ.get("GITHUB_API_ORIGIN", "https://api.github.com").rstrip("/") AVATARS_ORIGIN = os.environ.get("GITHUB_AVATARS_ORIGIN", "https://avatars.githubusercontent.com").rstrip("/") OWNER = { "login": "octostub", "id": 10001, "avatar_url": f"{AVATARS_ORIGIN}/u/10001", "bio": "stub bio for the import flow", "blog": "https://octostub.example.com", } # sha256 exercises knot2 object-format negotiation locally SEED = [ ( "hello", "Hello world stub repository", { "README.md": "# Hello\n\nHello world stub repository for Tangled import testing.\n", "hello.txt": "Hello from octostub/hello!\n", }, "sha1", ), ( "world", "World stub repository for bulk import", {"README.md": "# World\n\nWorld stub repository for bulk import testing.\n"}, "sha1", ), ( "tangled-demo", "Tangled demo repository", {"README.md": "# Tangled Demo\n\nDemo repository for tangled import.\n"}, "sha1", ), ( "fresh", "Fresh stub repository, never imported before", { "README.md": "# Fresh\n\nA repository nobody has imported yet.\n", "src/lib.rs": "pub fn fresh() -> &'static str {\n \"fresh\"\n}\n", }, "sha256", ), ( "second", "Second fresh stub repository, for bulk import", { "README.md": "# Second\n\nThe second repository in a bulk import.\n", "notes.md": "two\n", }, "sha1", ), ( "toolkit", "Small tools, for the bulk import", { "README.md": "# Toolkit\n\nTools nobody has imported yet.\n", "tools.md": "hammer\n", }, "sha1", ), ( "notes", "Scratch notes, also unimported", {"README.md": "# Notes\n\nNothing to see here yet.\n", "today.md": "one\n"}, "sha1", ), ] REPO_METADATA = { "hello": {"language": "TypeScript", "stars": 12, "issues": 3, "forks": 2, "size_kb": 21504, "days_ago": 2}, "world": {"language": "Rust", "stars": 340, "issues": 7, "forks": 41, "size_kb": 512, "days_ago": 5}, "tangled-demo": {"language": "Go", "stars": 88, "issues": 1, "forks": 9, "size_kb": 330, "hours_ago": 3}, "fresh": {"language": "Rust", "stars": 0, "issues": 0, "forks": 1, "size_kb": 64, "days_ago": 1}, "second": {"language": "Python", "stars": 2048, "issues": 22, "forks": 190, "size_kb": 20480, "days_ago": 200}, "toolkit": {"language": "Shell", "stars": 5, "issues": 0, "forks": 0, "size_kb": 88, "days_ago": 3}, "notes": {"language": None, "stars": 1, "issues": 0, "forks": 0, "size_kb": 16, "days_ago": 45}, } # filler so the import list has enough rows to page, scroll and truncate; seeded so it's stable def _filler(count: int): rng = random.Random(7) adjectives = ["tiny", "rusty", "quiet", "brisk", "lazy", "shiny", "fuzzy", "bold", "spare", "odd", "swift", "mellow", "plain", "wild", "sleepy", "eager", "hollow", "neon", "paper", "frozen"] nouns = ["parser", "kettle", "badger", "lantern", "compiler", "garden", "relay", "sketch", "atlas", "widget", "harbor", "beacon", "notebook", "cache", "spindle", "loom", "orbit", "pebble", "toolbox", "ledger", "router", "canvas", "sampler", "daemon", "shelf"] subjects = ["A small library", "Experimental tooling", "Personal dotfiles", "A toy interpreter", "Benchmarks and notes", "A static site", "Bindings", "A CLI", "Config and scripts", "Half-finished prototype"] purposes = ["for parsing odd file formats", "for syncing things between machines", "that nobody asked for", "for the homelab", "for learning the language", "kept around for reference", "for a weekend hackathon", "extracted from a bigger project", "for generating release notes", "for poking at network protocols"] tails = ["", "", "", " Mostly works, occasionally surprises. Contributions welcome but expect slow reviews.", " Includes a test suite, a fuzzing harness, a handful of examples and a README that is more" " aspirational than accurate."] languages = ["Rust", "Go", "TypeScript", "Python", "Shell", "C", "Haskell", "Nix", "Lua", "Zig", "OCaml", "Elixir", "JavaScript", "Svelte", None] names = set(REPO_METADATA) seeds = [] while len(seeds) < count: name = f"{rng.choice(adjectives)}-{rng.choice(nouns)}" if name in names: name = f"{name}-{rng.randint(2, 99)}" if name in names: continue names.add(name) description = "" if rng.random() < 0.1 else f"{rng.choice(subjects)} {rng.choice(purposes)}.{rng.choice(tails)}" seeds.append((name, description, {"README.md": f"# {name}\n\n{description}\n"}, "sha1")) REPO_METADATA[name] = { "language": rng.choice(languages), "stars": rng.choice([rng.randint(0, 40)] * 8 + [rng.randint(40, 2000)] * 3 + [rng.randint(2000, 25000)]), "issues": rng.randint(0, 40), "forks": rng.randint(0, 200), "size_kb": rng.randint(8, 50000), "days_ago": rng.randint(0, 1500), } return seeds SEED += _filler(93) def _iso_age(**kwargs) -> str: return (datetime.now(timezone.utc) - timedelta(**kwargs)).strftime("%Y-%m-%dT%H:%M:%SZ") def repository(name: str, description: str, files: dict, object_format: str = "sha1") -> dict: meta = REPO_METADATA[name] pushed_at = _iso_age(hours=meta.get("hours_ago", 0), days=meta.get("days_ago", 0)) return { "id": 101 + [seed[0] for seed in SEED].index(name), "name": name, "full_name": f"octostub/{name}", "owner": OWNER, "html_url": f"{PUBLIC_ORIGIN}/octostub/{name}", "description": description, "clone_url": f"{PUBLIC_ORIGIN}/octostub/{name}.git", "default_branch": "main", "visibility": "public", "private": False, "language": meta["language"], "stargazers_count": meta["stars"], "open_issues_count": meta["issues"], "forks_count": meta["forks"], "size": meta["size_kb"], "pushed_at": pushed_at, "updated_at": pushed_at, "files": files, "object_format": object_format, } REPOSITORIES = [repository(*seed) for seed in SEED] STUB_TOKEN_PREFIX = "ghstub-token-" def email_for(token: str) -> str: """The caller's address. Deliberi refuses one address for two accounts, so every authorization gets its own: the token is derived from the code minted per connect.""" match = re.fullmatch(STUB_TOKEN_PREFIX + r"([0-9a-f]{16})", token) suffix = f"+{match.group(1)[:8]}" if match else "" return f"octostub{suffix}@octostub.dev" GITHUB_EMAILS = [ {"email": "octostub@octostub.dev", "primary": True, "verified": True, "visibility": "public"}, { "email": "octostub-secondary@octostub.dev", "primary": False, "verified": False, "visibility": "public", }, { "email": "octostub-alt@octostub.dev", "primary": False, "verified": True, "visibility": "public", }, ] def _clean_repo(r: dict) -> dict: return {k: v for k, v in r.items() if k not in ("files", "object_format")} def make_png(width=16, height=16, color=(120, 80, 200)) -> bytes: png = b"\x89PNG\r\n\x1a\n" ihdr_data = struct.pack(">IIBBBBB", width, height, 8, 2, 0, 0, 0) ihdr_crc = struct.pack(">I", zlib.crc32(b"IHDR" + ihdr_data) & 0xffffffff) png += struct.pack(">I", len(ihdr_data)) + b"IHDR" + ihdr_data + ihdr_crc raw = bytearray() for _ in range(height): raw.append(0) for _ in range(width): raw.extend(color) compressed = zlib.compress(bytes(raw)) idat_crc = struct.pack(">I", zlib.crc32(b"IDAT" + compressed) & 0xffffffff) png += struct.pack(">I", len(compressed)) + b"IDAT" + compressed + idat_crc iend_crc = struct.pack(">I", zlib.crc32(b"IEND") & 0xffffffff) png += struct.pack(">I", 0) + b"IEND" + iend_crc return png AVATAR_PNG = make_png() def seed_repositories(): os.makedirs(REPOS_DIR, exist_ok=True) for repo_info in REPOSITORIES: full_name = repo_info["full_name"] bare_path = os.path.join(REPOS_DIR, f"{full_name}.git") if os.path.exists(bare_path): continue os.makedirs(os.path.dirname(bare_path), exist_ok=True) fmt = repo_info.get("object_format") or OBJECT_FORMAT with tempfile.TemporaryDirectory() as tmp_dir: subprocess.run( ["git", "init", f"--object-format={fmt}", tmp_dir], check=True, capture_output=True, ) subprocess.run(["git", "-C", tmp_dir, "checkout", "-B", "main"], check=True, capture_output=True) for fname, content in repo_info["files"].items(): fpath = os.path.join(tmp_dir, fname) os.makedirs(os.path.dirname(fpath), exist_ok=True) with open(fpath, "w") as f: f.write(content) subprocess.run(["git", "-C", tmp_dir, "config", "user.name", "Octo Stub"], check=True, capture_output=True) subprocess.run(["git", "-C", tmp_dir, "config", "user.email", "octostub@octostub.dev"], check=True, capture_output=True) subprocess.run(["git", "-C", tmp_dir, "add", "."], check=True, capture_output=True) subprocess.run(["git", "-C", tmp_dir, "commit", "-m", "Initial commit"], check=True, capture_output=True) subprocess.run(["git", "clone", "--bare", tmp_dir, bare_path], check=True, capture_output=True) subprocess.run(["git", "-C", bare_path, "symbolic-ref", "HEAD", "refs/heads/main"], check=True, capture_output=True) subprocess.run(["git", "-C", bare_path, "config", "http.receivepack", "false"], check=True, capture_output=True) desc_path = os.path.join(bare_path, "description") with open(desc_path, "w") as f: f.write(repo_info["description"] + "\n") print(f"Seeded bare repository ({fmt}): {bare_path}", flush=True) class GitHubStubHandler(BaseHTTPRequestHandler): protocol_version = "HTTP/1.1" def log_message(self, format, *args): sys.stderr.write(f"[{self.log_date_time_string()}] {self.address_string()} {format % args}\n") sys.stderr.flush() def read_body(self) -> bytes: if self.headers.get("Transfer-Encoding", "").lower() == "chunked": chunks = [] while True: line = self.rfile.readline().strip() if not line: break chunk_len = int(line.split(b";")[0], 16) if chunk_len == 0: while self.rfile.readline().strip(): pass break chunk = self.rfile.read(chunk_len) chunks.append(chunk) self.rfile.readline() return b"".join(chunks) length = int(self.headers.get("Content-Length", 0)) return self.rfile.read(length) if length > 0 else b"" def send_json(self, status: int, data: any, link: str = ""): body = json.dumps(data, indent=2).encode("utf-8") self.send_response(status) self.send_header("Content-Type", "application/json; charset=utf-8") self.send_header("Content-Length", str(len(body))) self.send_header("Cache-Control", "no-store") if link: self.send_header("Link", link) self.end_headers() self.wfile.write(body) def page_link(self, path: str, query: dict, page: int, per_page: int, total: int) -> str: if page * per_page >= total: return "" params = {k: v[0] for k, v in query.items()} params.update({"page": str(page + 1), "per_page": str(per_page)}) nxt = urllib.parse.urlencode(params) return f'<{API_ORIGIN}{path}?{nxt}>; rel="next"' def send_avatar(self): self.send_response(200) self.send_header("Content-Type", "image/png") self.send_header("Content-Length", str(len(AVATAR_PNG))) self.send_header("Cache-Control", "public, max-age=86400") self.end_headers() self.wfile.write(AVATAR_PNG) def do_GET(self): parsed = urllib.parse.urlparse(self.path) path = parsed.path.rstrip("/") if not path: path = "/" query = urllib.parse.parse_qs(parsed.query) if path.startswith("/u/") or path.startswith("/avatar") or path.startswith("/avatars/") or path == "/user/avatar": self.send_avatar() return git_refs_match = re.match(r"^/([^/]+)/([^/]+?)(?:\.git)?/info/refs$", parsed.path) if git_refs_match: owner, repo_name = git_refs_match.group(1), git_refs_match.group(2) service = query.get("service", [""])[0] if service != "git-upload-pack": self.send_error(400, "Unsupported git service") return repo_path = os.path.join(REPOS_DIR, owner, f"{repo_name}.git") if not os.path.exists(repo_path): self.send_error(404, "Repository not found") return env = os.environ.copy() proto = self.headers.get("Git-Protocol") if proto: env["GIT_PROTOCOL"] = proto env["GIT_CONFIG_NOSYSTEM"] = "1" proc = subprocess.run( ["git", "upload-pack", "--stateless-rpc", "--advertise-refs", repo_path], capture_output=True, env=env ) if proc.returncode != 0: self.send_error(500, f"git upload-pack failed: {proc.stderr.decode('utf-8', errors='replace')}") return body = b"001e# service=git-upload-pack\n0000" + proc.stdout self.send_response(200) self.send_header("Content-Type", "application/x-git-upload-pack-advertisement") self.send_header("Content-Length", str(len(body))) self.send_header("Cache-Control", "no-cache, no-store, must-revalidate") self.send_header("Pragma", "no-cache") self.end_headers() self.wfile.write(body) return if path == "/login/oauth/authorize": redirect_uri = query.get("redirect_uri", [""])[0] state = query.get("state", [""])[0] code = "ghstub-" + os.urandom(8).hex() if not redirect_uri: self.send_error(400, "Missing redirect_uri") return sep = "&" if "?" in redirect_uri else "?" dest = f"{redirect_uri}{sep}code={code}&state={urllib.parse.quote(state)}" self.send_response(302) self.send_header("Location", dest) self.send_header("Cache-Control", "no-store") self.send_header("Content-Length", "0") self.end_headers() return if re.match(r"^/apps/[^/]+/installations/new$", path): setup = os.environ.get( "GITHUB_APP_SETUP_URL", "http://127.0.0.1:5174/_internal/github/install/callback", ) sep = "&" if "?" in setup else "?" state = query.get("state", [""])[0] dest = f"{setup}{sep}installation_id=1&setup_action=install" if state: dest += f"&state={urllib.parse.quote(state)}" self.send_response(302) self.send_header("Location", dest) self.send_header("Cache-Control", "no-store") self.send_header("Content-Length", "0") self.end_headers() return if path == "/stub/installed": html = ( b"GitHub App Installed" b"

GitHub App Installed

" b"

The Tangled GitHub App has been installed on your stub account.

" b"" ) self.send_response(200) self.send_header("Content-Type", "text/html; charset=utf-8") self.send_header("Content-Length", str(len(html))) self.end_headers() self.wfile.write(html) return if path == "/user": token = self.headers.get("Authorization", "").removeprefix("Bearer ").strip() user_data = { "login": "octostub", "id": 10001, "node_id": "MDQ6VXNlcjEwMDAx", "avatar_url": f"{AVATARS_ORIGIN}/u/10001", "gravatar_id": "", "url": f"{API_ORIGIN}/users/octostub", "html_url": f"{PUBLIC_ORIGIN}/octostub", "type": "User", "site_admin": False, "name": "Octo Stub", "company": None, "blog": "https://octostub.example.com", "location": None, "email": email_for(token), "hireable": None, "bio": "Stub GitHub user for local development", "public_repos": len(REPOSITORIES), "public_gists": 0, "followers": 0, "following": 0, "created_at": "2024-01-01T00:00:00Z", "updated_at": "2024-01-01T00:00:00Z" } self.send_json(200, user_data) return if path == "/user/emails": token = self.headers.get("Authorization", "").removeprefix("Bearer ").strip() address = email_for(token) local_suffix = address[len("octostub") : address.index("@")] self.send_json( 200, [ {**row, "email": row["email"].replace("@octostub.dev", "").replace("octostub", "octostub" + local_suffix) + "@octostub.dev"} for row in GITHUB_EMAILS ], ) return if path == "/user/keys": keys = [ { "id": 1, "key": "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIOr0s3gZf7W/8vV/EXAMPLEEXAMPLEEXAMPLEEXAMPLE octostub@octostub.dev", "title": "Stub Key" } ] self.send_json(200, keys) return if path == "/user/installations": installations_data = { "total_count": 1, "installations": [ { "id": 1, "account": { "login": "octostub", "id": 10001, "type": "User", "avatar_url": f"{AVATARS_ORIGIN}/u/10001" }, "app_id": 1, "app_slug": APP_SLUG, "target_id": 10001, "target_type": "User", "permissions": { "metadata": "read", "contents": "read" }, "events": [], "repository_selection": "all" } ] } self.send_json(200, installations_data) return inst_repos_match = re.match(r"^/user/installations/(\d+)/repositories$", path) if inst_repos_match: clean_repos = [_clean_repo(r) for r in REPOSITORIES] page = max(1, int(query.get("page", ["1"])[0])) per_page = max(1, min(100, int(query.get("per_page", ["30"])[0]))) self.send_json(200, { "total_count": len(clean_repos), "repositories": clean_repos[(page - 1) * per_page : page * per_page] }, self.page_link(path, query, page, per_page, len(clean_repos))) return if path == "/user/repos": visibility = query.get("visibility", [""])[0] affiliation = query.get("affiliation", ["owner,collaborator,organization_member"])[0] page = max(1, int(query.get("page", ["1"])[0])) per_page = max(1, min(100, int(query.get("per_page", ["30"])[0]))) owned = "owner" in {a for a in affiliation.split(",") if a} rows = [] if not owned else [ _clean_repo(r) for r in REPOSITORIES if not visibility or r.get("visibility") == visibility ] self.send_json(200, rows[(page - 1) * per_page : page * per_page], self.page_link(path, query, page, per_page, len(rows))) return users_repos_match = re.match(r"^/users/([^/]+)/repos$", path) if users_repos_match: login = users_repos_match.group(1) page = max(1, int(query.get("page", ["1"])[0])) per_page = max(1, min(100, int(query.get("per_page", ["30"])[0]))) rows = [ _clean_repo(r) for r in REPOSITORIES if r.get("visibility") == "public" and r.get("full_name", "").startswith(f"{login}/") ] self.send_json(200, rows[(page - 1) * per_page : page * per_page], self.page_link(path, query, page, per_page, len(rows))) return repo_match = re.match(r"^/repos/([^/]+)/([^/]+)$", path) if repo_match: owner, name = repo_match.group(1), repo_match.group(2) full_name = f"{owner}/{name}" for r in REPOSITORIES: if r["full_name"] == full_name: self.send_json(200, _clean_repo(r)) return self.send_json(404, {"message": "Not Found"}) return if path == "/": self.send_json(200, { "status": "ok", "service": "github-stub", "client_id": CLIENT_ID, "app_slug": APP_SLUG, "repositories": [r["full_name"] for r in REPOSITORIES] }) return self.send_json(404, {"message": f"Not Found: {path}"}) def do_POST(self): parsed = urllib.parse.urlparse(self.path) path = parsed.path.rstrip("/") if not path: path = "/" git_pack_match = re.match(r"^/([^/]+)/([^/]+?)(?:\.git)?/git-upload-pack$", parsed.path) if git_pack_match: owner, repo_name = git_pack_match.group(1), git_pack_match.group(2) repo_path = os.path.join(REPOS_DIR, owner, f"{repo_name}.git") if not os.path.exists(repo_path): self.send_error(404, "Repository not found") return body = self.read_body() content_encoding = self.headers.get("Content-Encoding", "").lower() if content_encoding == "gzip": body = zlib.decompress(body, 16 + zlib.MAX_WBITS) elif content_encoding == "deflate": body = zlib.decompress(body) env = os.environ.copy() proto = self.headers.get("Git-Protocol") if proto: env["GIT_PROTOCOL"] = proto env["GIT_CONFIG_NOSYSTEM"] = "1" proc = subprocess.Popen( ["git", "upload-pack", "--stateless-rpc", repo_path], stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, env=env ) stdout, stderr = proc.communicate(input=body) if proc.returncode != 0: self.send_error(500, f"git upload-pack failed: {stderr.decode('utf-8', errors='replace')}") return self.send_response(200) self.send_header("Content-Type", "application/x-git-upload-pack-result") self.send_header("Content-Length", str(len(stdout))) self.send_header("Cache-Control", "no-cache, no-store, must-revalidate") self.send_header("Pragma", "no-cache") self.end_headers() self.wfile.write(stdout) return if path == "/login/oauth/access_token": raw_body = self.read_body() content_type = self.headers.get("Content-Type", "").lower() params = {} if "application/json" in content_type: try: params = json.loads(raw_body.decode("utf-8")) except Exception: pass else: try: parsed_qs = urllib.parse.parse_qs(raw_body.decode("utf-8")) params = {k: v[0] for k, v in parsed_qs.items()} except Exception: pass token_data = { "access_token": STUB_TOKEN_PREFIX + (str(params.get("code", "")).removeprefix("ghstub-") or "0123456789abcdef")[:16], "token_type": "bearer", "scope": "repo,read:user,user:email" } accept = self.headers.get("Accept", "").lower() if "application/json" in accept or "application/json" in content_type: self.send_json(200, token_data) else: body = urllib.parse.urlencode(token_data).encode("utf-8") self.send_response(200) self.send_header("Content-Type", "application/x-www-form-urlencoded") self.send_header("Content-Length", str(len(body))) self.send_header("Cache-Control", "no-store") self.end_headers() self.wfile.write(body) return self.send_json(404, {"message": f"Not Found: {path}"}) def run(): print(f"Starting GitHub Stub on port {PORT}...", flush=True) seed_repositories() server = ThreadingHTTPServer(("0.0.0.0", PORT), GitHubStubHandler) print(f"GitHub Stub listening on 0.0.0.0:{PORT}", flush=True) try: server.serve_forever() except KeyboardInterrupt: pass finally: server.server_close() if __name__ == "__main__": run()