Something went wrong. Try again.
we (web engine): Experimental web browser project to understand the limits of Claude
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675#!/usr/bin/env -S uv run --script# /// script# requires-python = ">=3.11"# ///"""Autonomous implementation loop for the `we` browser engine.Runs an implementation agent in a loop, each iteration picking up and completingone task. Uses Claude while usage is under cap, and falls back to Codex whenClaude usage is over cap."""import jsonimport osimport subprocessimport sysimport timefrom datetime import datetime, timedelta, timezoneGIT_ROOT = "/Users/piefev/misc/we"LOG_PATH = "/Users/piefev/misc/we/implementor.log"RATE_LIMITS_PATH = "/Users/piefev/misc/we/ratelimit.json"# Fallback poll interval used only when no reset time is known (usage# unavailable, or both runners disabled).PAUSE_SECONDS_WHEN_OVER_CAPS = 300# Extra slack added after a window's reset so refreshed usage reflects it.RESET_BUFFER_SECONDS = 30# Never sleep less than this, even if a reset time is in the near past.MIN_SLEEP_SECONDS = 60DEFAULT_CAPS = { "claude": {"enabled": True, "five_hour": 100, "weekly": 100}, "codex": {"enabled": True, "five_hour": 100, "weekly": 100},}_log_file = Noneclass Tee: """Write to multiple streams; used to mirror stdout/stderr into a log file.""" def __init__(self, *streams): self.streams = streams def write(self, data): for stream in self.streams: stream.write(data) stream.flush() return len(data) def flush(self): for stream in self.streams: stream.flush() def isatty(self): return FalsePROMPT = r"""You are an autonomous implementation agent for the `we` browser engine project.The git root is at /Users/piefev/misc/we. Always start by reading /Users/piefev/misc/we/CLAUDE.md and /Users/piefev/misc/we/PLAN.md.The project is hosted on Tangled, but issue and project tracking lives in thelocal `isu` tracker. `isu` is the only issue/project tracking tool that shallbe used. All work merges directly to `main`; this project does not use pullrequests.The `.isu/` directory is issue/project tracker state. Whenever `isu` creates,updates, labels, or closes an issue, commit and push the resulting `.isu/`changes. Never leave `.isu/` changes local-only.Your job is to pick ONE task, complete it fully, and then exit. Follow this priority order:---## Workspace (shared by every priority)All work happens in ONE long-lived worktree at /Users/piefev/misc/we-implementor.Do NOT create a fresh per-task worktree, and do NOT remove the worktree when youfinish — reuse the same worktree every iteration.At the start of every iteration, set the worktree up:1. Create the worktree if it does not already exist. The `|| true` swallows the "already exists" error on later iterations: ``` cd /Users/piefev/misc/we git worktree add --detach /Users/piefev/misc/we-implementor 2>/dev/null || true ```2. Refresh main in the primary checkout, then reset the worktree onto the latest main via a fresh `implementor` branch. This discards any leftover state from the previous iteration, which is safe because that work was already merged: ``` cd /Users/piefev/misc/we git checkout main && git pull cd /Users/piefev/misc/we-implementor git fetch origin git checkout -B implementor origin/main ```Do all implementation work in /Users/piefev/misc/we-implementor on the`implementor` branch. When the task is complete and committed there, merge itinto main from the primary checkout and push: ``` cd /Users/piefev/misc/we git checkout main git merge implementor git push origin main ```Leave the worktree and the `implementor` branch in place for the next iteration.NEVER run `git worktree remove` and never delete the `implementor` branch.---## Priority 1: Open IssuesCheck for open issues:```isu issue list --state open --format json```If there are open issues, pick the FIRST issue in the JSON array and implement it:1. Read the issue details: ``` isu issue show <id> ```2. Understand the task described in the issue body.3. Set up the workspace as described in the "Workspace" section above: create /Users/piefev/misc/we-implementor if needed, then run `git checkout -B implementor origin/main` inside it.4. Implement the issue in /Users/piefev/misc/we-implementor on the `implementor` branch. Follow all conventions from CLAUDE.md: - Zero external crate dependencies - `unsafe` only in platform, crypto, js - Write tests5. Run the full completion checklist: ``` cd /Users/piefev/misc/we-implementor cargo fmt --all cargo clippy --workspace -- -D warnings cargo test --workspace ``` If the issue is about rendering, layout, styling, the UA stylesheet, forms, scripting, SVG, canvas, or anything user-visible, also run the e2e smoke suite and visually inspect the screenshots: ``` cargo run -p we-e2e -- --scenario crates/e2e/scenarios/smoke.we --out-dir crates/e2e/artifacts ``` When fixing a bug the harness exposed, extend `crates/e2e/scenarios/` with an assertion that would have caught the bug.6. Close the issue locally so the `.isu/` tracker state is included in the implementation commit: ``` isu issue close <id> ```7. Commit all code changes and `.isu/` tracker changes with a descriptive message that references `isu issue <id>`.8. Merge into main and push from the primary checkout: ``` cd /Users/piefev/misc/we git checkout main git merge implementor git push origin main ``` Do NOT remove the worktree or delete the `implementor` branch.After merging to main, exit.---## Priority 2: Real-Web Soak Iteration (Phase 23)If there are no open issues, run one Phase 23 real-web soak iteration. Phase 23is the final phase in PLAN.md and is open-ended by design: each empty-queueiteration picks a fresh random safe popular site, snapshots it, writes ascenario, and files an `isu` issue for every defect surfaced. The canonicalworkflow lives in `crates/e2e/real-web/README.md` under "Soak iterationworkflow"; the steps below are the autonomous-agent version.Do NOT break a hypothetical Phase 24 down into issues. There is no Phase 24.1. Pick a random safe popular URL: ``` python3 /Users/piefev/misc/we/tests/popular-sites/pick.py ``` If the picker exits non-zero (network down, all feeds unreachable, …), print a single line containing the literal string `OUT_OF_JOB: picker unavailable` so the wrapping loop backs off, then exit without making any changes.2. Derive `<domain>` from the URL (host with no scheme, no trailing dot).3. Set up the workspace as described in the "Workspace" section above: create /Users/piefev/misc/we-implementor if needed, then run `git checkout -B implementor origin/main` inside it and `cd` there. All soak work happens in that single reused worktree on the `implementor` branch.4. If `crates/e2e/scenarios/real-web/<domain>.we` already exists, re-baseline that scenario instead of writing a new one: ``` cargo run -p we-e2e -- --real-web --real-web-online \ --out-dir crates/e2e/artifacts ``` Inspect the snapshot diff and the screenshot diff under `crates/e2e/artifacts/real-web/runs/<domain>/`. If `we`'s render now diverges from the committed Chromium golden, file an `isu` issue with the `real-web` label including the artifact paths and exit. Otherwise update the snapshot (and Chromium golden if upstream visibly drifted) in their own commit, merge, push, and exit.5. Otherwise (new domain), follow the full soak iteration workflow per `crates/e2e/real-web/README.md`: - Snapshot the site under `crates/e2e/real-web/snapshots/<domain>/` (HTML plus any subresources required for the assertion). Fetch once over the network and commit the bytes. - Write `crates/e2e/scenarios/real-web/<domain>.we`: `network offline`, `goto_as <url> <snapshot>`, `screenshot`, `dump_dom`, `dump_console`, and at least one `assert_dom_contains` for the page's primary content. - Capture the Chromium golden: ``` python3 tests/popular-sites/compare.py --write-golden --browser chromium \ crates/e2e/scenarios/real-web/<domain>.we ``` and assert against it in the scenario: `assert_screenshot_matches real-web/<domain>/desktop.png <domain>.desktop.chromium.expected.png`. - Add at least one interactivity assertion (`click`, `type`, focus, …) per the Phase 23 interactivity policy. If the DSL doesn't yet support the interaction the page needs, mark the scenario `# xfail: see isu issue <id>` pointing at the tracking issue. - Run the scenario: ``` cargo run -p we-e2e -- \ --scenario crates/e2e/scenarios/real-web/<domain>.we \ --out-dir crates/e2e/artifacts ``` - For every defect surfaced (panic, layout glitch, missing glyph, JS exception, network misbehaviour, perf cliff, Chromium-parity gap), file an `isu` issue with the `real-web` label, the scenario path, and pointers to the captured artifacts under `crates/e2e/artifacts/real-web/runs/<domain>/`.6. Commit the scenario, snapshot, Chromium golden, and any `.isu/` defect issues — even if the scenario does not yet pass — so the next iteration picks the defect issues up via the normal Priority 1 path. Reference any `isu` issue numbers in the commit message.7. Merge into main and push from the primary checkout: ``` cd /Users/piefev/misc/we git checkout main git merge implementor git push origin main ``` Do NOT remove the worktree or delete the `implementor` branch.After committing the soak iteration to main, exit.---## Important Rules- Only do ONE task per invocation, then exit.- Merge issue implementations directly to main; this project does not use pull requests.- Always work in the single reused /Users/piefev/misc/we-implementor worktree on the `implementor` branch, never directly on main. Never create per-task worktrees and never remove the worktree.- Always run the completion checklist before pushing.- `isu` is the only issue/project tracker. Never use Tangled issue commands.- Every `.isu/` change must be committed and pushed, including issue creation, labels, updates, and close state.- If something fails and you cannot fix it, create a new `isu` issue: ``` isu issue create --title "..." --body "..." ``` Then exit.- Be thorough but focused. Do not add scope beyond what the issue requires.- Use `isu issue close <id>` after the completion checklist passes, and include the `.isu/` close state in the implementation commit that gets pushed.- When implementing, read PLAN.md to understand the phased roadmap and where the current issue fits.""".strip()def claude_command() -> tuple[list[str], dict[str, str]]: return ( [ "claude", "--dangerously-skip-permissions", "-p", PROMPT, "--output-format", "text", "--max-turns", "200", "--verbose", ], {**os.environ, "ANTHROPIC_MODEL": "best"}, )def codex_command() -> tuple[list[str], dict[str, str]]: return (["codex", "--yolo", "exec", PROMPT], os.environ.copy())def run_iteration(iteration: int, runner: str) -> tuple[str, float]: """Run one implementation-agent invocation.""" if runner == "claude": cmd, env = claude_command() elif runner == "codex": cmd, env = codex_command() else: raise ValueError(f"unknown runner: {runner}") start = time.monotonic() try: result = subprocess.run( cmd, capture_output=True, text=True, cwd=GIT_ROOT, env=env, ) except Exception as exc: elapsed = time.monotonic() - start print(f"\nIteration {iteration} ({runner}) failed after {elapsed:.1f}s: {exc}") return str(exc), elapsed elapsed = time.monotonic() - start output = result.stdout + result.stderr if _log_file is not None: _log_file.write( f"\n----- Iteration {iteration} ({runner}) full output begin -----\n" ) _log_file.write(output) if not output.endswith("\n"): _log_file.write("\n") _log_file.write( f"----- Iteration {iteration} ({runner}) full output end -----\n" ) _log_file.flush() print(f"\n{'=' * 60}") print( f"Iteration {iteration} ({runner}) completed in {elapsed:.1f}s " f"[{datetime.now(timezone.utc).isoformat()}]" ) print(f"{'=' * 60}") lines = output.strip().split("\n") for line in lines[-40:]: print(f" {line}") print() return output, elapseddef load_usage_json(command: str) -> tuple[dict | None, str | None]: try: result = subprocess.run( [command, "--json"], capture_output=True, text=True, timeout=15, ) except (FileNotFoundError, subprocess.TimeoutExpired) as exc: return None, str(exc) if result.returncode != 0: stderr = result.stderr.strip() or result.stdout.strip() return None, f"rc={result.returncode}: {stderr}" try: return json.loads(result.stdout), None except json.JSONDecodeError: return None, "returned non-JSON"def format_percent(value: float | int | None) -> str: if value is None: return "unavailable" return f"{float(value):.0f}%"def reset_datetime(reset_at: object) -> datetime | None: """Parse a `resets_at` value (epoch seconds or ISO-8601) to an aware UTC dt.""" if reset_at in (None, ""): return None try: if isinstance(reset_at, (int, float)): return datetime.fromtimestamp(reset_at, timezone.utc) dt = datetime.fromisoformat(str(reset_at).replace("Z", "+00:00")) if dt.tzinfo is None: dt = dt.replace(tzinfo=timezone.utc) return dt except (OSError, TypeError, ValueError): return Nonedef format_reset(reset_at: object) -> str: reset_dt = reset_datetime(reset_at) if reset_dt is None: return "not shown" return reset_dt.astimezone().strftime("%Y-%m-%d %H:%M %Z")def usage_line(provider: str, label: str, used: object, reset_at: object) -> str: try: used_percent = float(used) if used is not None else None except (TypeError, ValueError): used_percent = None if used_percent is None: return f" {provider} {label}: remaining unavailable (used unavailable)" remaining = max(0.0, 100.0 - used_percent) return ( f" {provider} {label}: {format_percent(remaining)} remaining " f"({format_percent(used_percent)} used, resets {format_reset(reset_at)})" )def print_usage_report( claude_data: dict | None, claude_error: str | None, codex_data: dict | None, codex_error: str | None, claude_enabled: bool = True, codex_enabled: bool = True,) -> None: if not claude_enabled: print(" Claude disabled; skipping") elif claude_data is None: print(f" Claude usage unavailable ({claude_error})") else: for key, label in (("five_hour", "5h"), ("seven_day", "weekly")): block = claude_data.get(key) or {} print(usage_line("Claude", label, block.get("utilization"), block.get("resets_at"))) if not codex_enabled: print(" Codex disabled; skipping") elif codex_data is None: print(f" Codex usage unavailable ({codex_error})") else: rate_limits = codex_data.get("rate_limits") or {} for key, label in (("primary", "5h"), ("secondary", "weekly")): block = rate_limits.get(key) or {} print(usage_line("Codex", label, block.get("used_percent"), block.get("resets_at")))def load_rate_limits() -> dict: """Load per-runner per-window usage caps (percent) from RATE_LIMITS_PATH. Missing or malformed → warn and fall back to DEFAULT_CAPS (100% = no cap). """ try: with open(RATE_LIMITS_PATH) as f: data = json.load(f) except FileNotFoundError: print(f" rate-limits file {RATE_LIMITS_PATH} not found; using defaults") return {k: dict(v) for k, v in DEFAULT_CAPS.items()} except (OSError, json.JSONDecodeError) as exc: print(f" rate-limits file unreadable ({exc}); using defaults") return {k: dict(v) for k, v in DEFAULT_CAPS.items()} caps = {k: dict(v) for k, v in DEFAULT_CAPS.items()} for runner in ("claude", "codex"): block = data.get(runner) or {} enabled = block.get("enabled") if isinstance(enabled, bool): caps[runner]["enabled"] = enabled for window in ("five_hour", "weekly"): value = block.get(window) if isinstance(value, (int, float)) and 0 < value <= 100: caps[runner][window] = float(value) return capsdef runner_windows(runner: str, data: dict) -> list[tuple[str, str, object, object]]: """Return (cap_key, label, utilization_percent, resets_at) for each window.""" if runner == "claude": five = data.get("five_hour") or {} seven = data.get("seven_day") or {} return [ ("five_hour", "5h", five.get("utilization"), five.get("resets_at")), ("weekly", "weekly", seven.get("utilization"), seven.get("resets_at")), ] rate_limits = data.get("rate_limits") or {} primary = rate_limits.get("primary") or {} secondary = rate_limits.get("secondary") or {} return [ ("five_hour", "5h", primary.get("used_percent"), primary.get("resets_at")), ("weekly", "weekly", secondary.get("used_percent"), secondary.get("resets_at")), ]def runner_blocking_windows( runner: str, caps: dict, data: dict | None,) -> list[tuple[str, float, float, object]] | None: """Return the windows that are at/over their cap for `runner`. `[]` means the runner has capacity. `None` means usage was unavailable. Each entry is `(label, util_percent, cap_percent, resets_at)`. """ if data is None: return None runner_caps = caps.get(runner, {}) over = [] for cap_key, label, util, reset_at in runner_windows(runner, data): cap = runner_caps.get(cap_key) if cap is None or util is None: continue try: util_f = float(util) except (TypeError, ValueError): continue if util_f >= float(cap): over.append((label, util_f, float(cap), reset_at)) return overdef runner_has_capacity( runner: str, caps: dict, data: dict | None, error: str | None,) -> bool: """Return whether all of `runner`'s usage windows are under their caps.""" over = runner_blocking_windows(runner, caps, data) if over is None: print(f" {runner} usage unavailable ({error})") return False if not over: return True for label, util_f, cap_f, reset_at in over: print( f" {runner} {label}: {util_f:.0f}% >= cap {cap_f:.0f}% " f"(resets {format_reset(reset_at)})" ) return Falsedef earliest_capacity_time( runners: list[str], caps: dict, datas: dict[str, dict | None],) -> datetime | None: """When does the soonest enabled runner regain capacity? A runner is free again only once *every* one of its over-cap windows has reset, so its ready time is the max reset among its blocking windows. The soonest any runner frees up is the min of those per-runner times. Returns None when no reset time is known (usage unavailable, or a blocking window has no parseable reset), so the caller can fall back to a fixed poll. """ per_runner = [] for runner in runners: over = runner_blocking_windows(runner, caps, datas.get(runner)) if not over: # None (unavailable) or [] (has capacity) continue resets = [reset_datetime(reset_at) for _, _, _, reset_at in over] if any(r is None for r in resets): continue # can't bound this runner's recovery per_runner.append(max(resets)) if not per_runner: return None return min(per_runner)def main() -> None: global _log_file _log_file = open(LOG_PATH, "a", buffering=1) sys.stdout = Tee(sys.__stdout__, _log_file) sys.stderr = Tee(sys.__stderr__, _log_file) print( f"\n========== implementor started {datetime.now(timezone.utc).isoformat()} ==========" ) print("we - autonomous implementation loop") print(f"Rate limits: {RATE_LIMITS_PATH}") print(f"Logging to {LOG_PATH}") print("Press Ctrl+C to stop\n") iteration = 0 try: while True: caps = load_rate_limits() claude_enabled = caps["claude"]["enabled"] codex_enabled = caps["codex"]["enabled"] print( f"\n--- Checking usage (caps " f"claude {'5h=%.0f%% / weekly=%.0f%%' % (caps['claude']['five_hour'], caps['claude']['weekly']) if claude_enabled else 'disabled'}, " f"codex {'5h=%.0f%% / weekly=%.0f%%' % (caps['codex']['five_hour'], caps['codex']['weekly']) if codex_enabled else 'disabled'}) ---" ) # Only query usage for enabled runners; disabled ones are skipped # entirely without even checking their usage. claude_data = claude_error = None codex_data = codex_error = None if claude_enabled: claude_data, claude_error = load_usage_json("claude-usage") if codex_enabled: codex_data, codex_error = load_usage_json("codex-usage") print_usage_report( claude_data, claude_error, codex_data, codex_error, claude_enabled, codex_enabled, ) if claude_enabled and runner_has_capacity( "claude", caps, claude_data, claude_error ): runner = "claude" elif codex_enabled and runner_has_capacity( "codex", caps, codex_data, codex_error ): runner = "codex" else: if not claude_enabled and not codex_enabled: print( f"\nBoth claude and codex are disabled; " f"sleeping {PAUSE_SECONDS_WHEN_OVER_CAPS}s before re-checking." ) time.sleep(PAUSE_SECONDS_WHEN_OVER_CAPS) continue enabled_runners = [ r for r, on in (("claude", claude_enabled), ("codex", codex_enabled)) if on ] datas = {"claude": claude_data, "codex": codex_data} ready_at = earliest_capacity_time(enabled_runners, caps, datas) if ready_at is None: print( f"\nNo enabled runner has capacity and no reset time is known; " f"sleeping {PAUSE_SECONDS_WHEN_OVER_CAPS}s before re-checking." ) time.sleep(PAUSE_SECONDS_WHEN_OVER_CAPS) continue now = datetime.now(timezone.utc) sleep_s = max( (ready_at - now).total_seconds() + RESET_BUFFER_SECONDS, MIN_SLEEP_SECONDS, ) wake = now + timedelta(seconds=sleep_s) print( f"\nNo enabled runner has capacity; sleeping " f"{sleep_s / 60:.0f}m until the earliest limit resets " f"(~{wake.astimezone():%Y-%m-%d %H:%M %Z})." ) time.sleep(sleep_s) continue iteration += 1 print( f"\n--- Starting iteration {iteration} with {runner} " f"[{datetime.now(timezone.utc).isoformat()}] ---" ) output, _ = run_iteration(iteration, runner) if "OUT_OF_JOB" in output: print("Nothing to do. Sleeping 60s...") time.sleep(60) except KeyboardInterrupt: print(f"\n\nStopped after {iteration} iteration(s).")if __name__ == "__main__": main()