Something went wrong. Try again.
watches turns and automatically creates and resolves tasks that get pinned into main agent context
Something went wrong. Try again.
6.7 kB · 176 lines
Python
at main
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177
import os, sys, json, time, pathlib, httpx
ROOT = pathlib.Path('/Users/dawn/proj/prime-agent/.prime/agent/session-artifacts/01a06166-382f-703f-8f17-bc939488de93/observer-bench')AUTH = json.loads((pathlib.Path.home() / '.prime/agent/auth.json').read_text())KEY = AUTH['google']['key']
MODEL = "gemma-4-31b-it"URL = f"https://generativelanguage.googleapis.com/v1beta/models/{MODEL}:generateContent"
PROMPT = pathlib.Path('/Users/dawn/proj/prime-commitment-observer/src/prompt.ts').read_text()# Extract prompt string between backticksprompt_start = PROMPT.find('`') + 1prompt_end = PROMPT.rfind('`')SYSTEM_PROMPT = PROMPT[prompt_start:prompt_end]
STATUSES = ["open", "active", "blocked", "waiting", "completion_candidate", "cancelled", "superseded"]
ITEM_SCHEMA = { "type": "object", "properties": { "id": {"type": "string"}, "canonicalRequest": {"type": "string"}, "requestedProperties": {"type": "array", "items": {"type": "string"}}, "implementedProperties": {"type": "array", "items": {"type": "string"}}, "contradictions": {"type": "array", "items": {"type": "string"}}, "evidenceActionIds": {"type": "array", "items": {"type": "string"}}, "status": {"type": "string", "enum": STATUSES} }, "required": [ "canonicalRequest", "requestedProperties", "implementedProperties", "contradictions", "evidenceActionIds", "status" ], "additionalProperties": False}
SCHEMA = { "type": "object", "properties": { "existing": {"type": "array", "items": ITEM_SCHEMA}, "additions": {"type": "array", "items": ITEM_SCHEMA} }, "required": ["existing", "additions"], "additionalProperties": False}
def call_observer(packet_dict, timeout=120): body = { "contents": [{"role": "user", "parts": [{"text": json.dumps(packet_dict)}]}], "systemInstruction": {"parts": [{"text": SYSTEM_PROMPT}]}, "generationConfig": { "temperature": 0, "maxOutputTokens": 4000, "thinkingConfig": {"thinkingLevel": "MINIMAL"}, "responseMimeType": "application/json", "responseJsonSchema": SCHEMA } } t0 = time.time() try: r = httpx.post(URL, headers={"x-goog-api-key": KEY, "content-type": "application/json"}, json=body, timeout=timeout) duration_ms = int((time.time() - t0) * 1000) if r.status_code != 200: return {"error": f"HTTP {r.status_code}: {r.text[:300]}", "ms": duration_ms} res = r.json() cand = res["candidates"][0] text = "".join(p.get("text", "") for p in cand.get("content", {}).get("parts", [])) return { "output": text, "ms": duration_ms, "finishReason": cand.get("finishReason"), "usage": res.get("usageMetadata") } except Exception as e: return {"error": f"{type(e).__name__}: {str(e)[:200]}", "ms": int((time.time() - t0) * 1000)}
def run_track(name, input_file, packet_builder, out_file): print(f"\n================================================") print(f"Running Track: {name} ({input_file.name})") print(f"================================================") items = json.loads(input_file.read_text()) results = [] with out_file.open('w') as fh: for idx, item in enumerate(items): packet = packet_builder(item) res = call_observer(packet) row = { "track": name, "index": idx, "itemId": item.get("id") or item.get("sessionId") or str(idx), "gold": item, "packet": packet, "result": res } results.append(row) fh.write(json.dumps(row) + "\n") fh.flush() status_char = "✓" if "output" in res else "✗" ms = res.get("ms", 0) print(f"[{idx+1}/{len(items)}] {status_char} ({ms}ms) ID={row['itemId']}") return results
if __name__ == "__main__": track_name = sys.argv[1] if len(sys.argv) > 1 else "all"
# Track 1: Multi-Intent if track_name in ("all", "multi-intent"): def build_mi_packet(c): return { "threads": [], "precedingAssistant": "", "userMessages": [c["raw_utterance"]], "completedAssistantTurn": "I will handle those requests for you.", "actions": [] } run_track( "multi-intent", ROOT / "bench-multi-intent-50.json", build_mi_packet, ROOT / "results-multi-intent.jsonl" )
# Track 2: Supersession if track_name in ("all", "supersession"): def build_sup_packet(p): threads = [] for t in p.get("threads", []): threads.append({ "id": t.get("id"), "canonicalRequest": t.get("canonicalRequest") or t.get("sourceUserText", ""), "status": "active" }) actions = [] for idx, a in enumerate(p.get("actions", [])[:10]): actions.append({ "id": f"a{idx+1}", "label": f"{a.get('act', 'action')}: {a.get('slot', '')} {','.join(a.get('values', []))}".strip(), "required": False, "ok": a.get("ok", True) }) return { "threads": threads, "precedingAssistant": p.get("precedingAssistant", ""), "userMessages": p.get("userMessages", []), "completedAssistantTurn": p.get("completedAssistant", ""), "actions": actions } run_track( "supersession", ROOT / "bench-supersession-30.json", build_sup_packet, ROOT / "results-supersession.jsonl" )
# Track 3: Real Candidates if track_name in ("all", "real"): def build_real_packet(p): return { "threads": p.get("threads", []), "precedingAssistant": p.get("precedingAssistant", ""), "userMessages": p.get("userMessages", []), "completedAssistantTurn": p.get("completedAssistantTurn", ""), "actions": p.get("actions", []) } run_track( "real-candidates", ROOT / "bench-real-candidates-20.json", build_real_packet, ROOT / "results-real-candidates.jsonl" )