diff --git a/tools/caretaker-agent/evals/triage/.gitignore b/tools/caretaker-agent/evals/triage/.gitignore
new file mode 100644
index 0000000000..010a321be8
--- /dev/null
+++ b/tools/caretaker-agent/evals/triage/.gitignore
@@ -0,0 +1,10 @@
+# Python bytecode
+__pycache__/
+*.pyc
+
+# Dynamic git worktrees and cloned target repository
+target_repo/
+worktrees/
+
+# Evaluation run output logs
+results/
diff --git a/tools/caretaker-agent/evals/triage/__init__.py b/tools/caretaker-agent/evals/triage/__init__.py
new file mode 100644
index 0000000000..50b00ea7eb
--- /dev/null
+++ b/tools/caretaker-agent/evals/triage/__init__.py
@@ -0,0 +1 @@
+"""Triage evaluation benchmark runner and judge suite."""
diff --git a/tools/caretaker-agent/evals/triage/helpers/__init__.py b/tools/caretaker-agent/evals/triage/helpers/__init__.py
new file mode 100644
index 0000000000..552a4419ae
--- /dev/null
+++ b/tools/caretaker-agent/evals/triage/helpers/__init__.py
@@ -0,0 +1 @@
+"""Internal helper modules for dataset loading, GitHub API, and summary reports."""
diff --git a/tools/caretaker-agent/evals/triage/helpers/dataset.py b/tools/caretaker-agent/evals/triage/helpers/dataset.py
new file mode 100644
index 0000000000..a9c31085ef
--- /dev/null
+++ b/tools/caretaker-agent/evals/triage/helpers/dataset.py
@@ -0,0 +1,62 @@
+"""Firestore Golden Dataset Streaming"""
+
+import os
+from typing import Dict, List, Any, Optional
+from dotenv import load_dotenv
+from google.cloud import firestore
+
+load_dotenv()
+
+
+def get_env_var(name: str) -> str:
+ """Helper that loads an environment variable and fails fast if missing."""
+ val = os.environ.get(name)
+ if not val:
+ raise RuntimeError(
+ f"Missing required environment variable '{name}'. "
+ f"Please ensure your .env file or environment is properly configured."
+ )
+ return val
+
+
+def load_issues(filter_issues: Optional[List[int]] = None) -> List[Dict[str, Any]]:
+ """Loads golden issue test cases directly from Firestore into memory."""
+ project_id = get_env_var("PROJECT_ID")
+ db_id = get_env_var("FIRESTORE_DATABASE")
+ collection_name = get_env_var("FIRESTORE_EVAL_COLLECTION")
+
+ db = firestore.Client(project=project_id, database=db_id)
+ docs = db.collection(collection_name).stream()
+
+ issues = []
+ for doc in docs:
+ data = doc.to_dict()
+ issue_num = data.get("issue_number")
+ if issue_num is None:
+ print(f"ā ļø Warning: Firestore document '{doc.id}' missing 'issue_number'. Skipping.")
+ continue
+ data["issue_number"] = int(issue_num)
+ if filter_issues and data["issue_number"] not in filter_issues:
+ continue
+ issues.append(data)
+
+ issues.sort(key=lambda x: x["issue_number"])
+ return issues
+
+
+def prep_payload(item: Dict[str, Any]) -> Dict[str, Any]:
+ """Preprocesses and wraps title & body to simulate production Ingestion Layer safety encapsulation."""
+ raw_body = item.get("issue_body") or ""
+ escaped_body = raw_body.replace("", "\\")
+ sanitized_body = f"\n{escaped_body}\n"
+
+ raw_title = item.get("issue_title") or ""
+ escaped_title = raw_title.replace("", "\\")
+ sanitized_title = f"\n{escaped_title}\n"
+
+ return {
+ "issue_number": item.get("issue_number"),
+ "title": sanitized_title,
+ "body": sanitized_body,
+ "repository": f"{item.get('owner', 'google-gemini')}/{item.get('repo', 'gemini-cli')}"
+ }
diff --git a/tools/caretaker-agent/evals/triage/helpers/summary.py b/tools/caretaker-agent/evals/triage/helpers/summary.py
new file mode 100644
index 0000000000..d4dfa82208
--- /dev/null
+++ b/tools/caretaker-agent/evals/triage/helpers/summary.py
@@ -0,0 +1,340 @@
+"""Run Evaluation Summary Calculator & Markdown Report Generator."""
+
+import json
+import datetime
+from os import environ
+from pathlib import Path
+from typing import Dict, List, Any, Optional
+
+BASE_DIR = Path(__file__).resolve().parent.parent
+PROJECT_ROOT = BASE_DIR.parent.parent
+RESULTS_DIR = BASE_DIR / "results"
+
+
+class MarkdownBuilder:
+ """Helper class for constructing safe, formatted Markdown reports."""
+
+ def __init__(self):
+ self.lines: List[str] = []
+
+ def h3(self, text: str):
+ self.lines.append(f"### {text}\n")
+
+ def text(self, text: str):
+ self.lines.append(f"{text}\n")
+
+ def table(self, headers: List[str], rows: List[List[Any]]):
+ self.lines.append("| " + " | ".join(headers) + " |")
+ self.lines.append("| " + " | ".join([":---"] * len(headers)) + " |")
+ for row in rows:
+ escaped = [str(cell).replace("|", "\\|").replace("\n", " ") for cell in row]
+ self.lines.append("| " + " | ".join(escaped) + " |")
+ self.lines.append("")
+
+ def details(self, summary_text: str, content: str):
+ self.lines.append(f"\n{summary_text}
\n\n{content}\n\n \n")
+
+ def render(self) -> str:
+ return "\n".join(self.lines)
+
+
+def init_dir(save: bool = True) -> str:
+ """Creates run output directory and sets up logging environment variables."""
+ if save:
+ timestamp_str = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")
+ run_dir = RESULTS_DIR / "runs" / f"run_{timestamp_str}"
+ else:
+ run_dir = RESULTS_DIR / "runs" / "run_temp"
+ if run_dir.exists():
+ import shutil
+ shutil.rmtree(run_dir)
+
+ issues_dir = run_dir / "issues"
+ issues_dir.mkdir(parents=True, exist_ok=True)
+ environ["GCS_LOGGING"] = "LOCAL"
+ environ["LOCAL_LOG_DIR"] = str(issues_dir)
+ return str(run_dir)
+
+
+def save_issue_result(issues_dir: Path, issue_num: int, record: Dict[str, Any]) -> None:
+ """Saves individual issue evaluation result JSON file to disk."""
+ file_path = Path(issues_dir) / f"gemini_cli_{issue_num}.json"
+ file_path.write_text(json.dumps(record, indent=2), encoding="utf-8")
+
+
+def _save_run_summary(run_summary: Dict[str, Any], run_dir: str) -> None:
+ """Saves structured suite summary evaluation result to run_dir/summary.json."""
+ (Path(run_dir) / "summary.json").write_text(json.dumps(run_summary, indent=2), encoding="utf-8")
+
+
+def _write_markdown(run_summary: Dict[str, Any], results: List[Dict[str, Any]], filepath: str) -> None:
+ """Writes formatted markdown summary report using MarkdownBuilder helper."""
+ doc = MarkdownBuilder()
+ doc.h3("š Triage Evaluation Summary")
+
+ note = run_summary.get("note")
+ if note:
+ doc.text(f"**Run Note:** {note}")
+
+ total_tested = run_summary.get("total_tested", 0)
+ total_attempted = run_summary.get("total_attempted", 0)
+ total_failed = run_summary.get("total_failed", 0)
+ doc.text(f"**Run Stats:** {total_tested}/{total_attempted} passed, {total_failed} failed/crashed.")
+
+ quality_match_pct = run_summary.get("quality_categorization_rate", 0) * 100
+ effort_match_pct = run_summary.get("effort_categorization_rate", 0) * 100
+
+ autoclose_recall_pct = run_summary.get("autoclose_recall_rate", 0) * 100
+ autoclose_correct_count = run_summary.get("correct_autoclose_count", 0)
+ autoclose_expected_count = run_summary.get("expected_autoclose_count", 0)
+
+ valid_kept_open_pct = run_summary.get("valid_kept_open_rate", 0) * 100
+ valid_kept_open_count = run_summary.get("valid_kept_open_count", 0)
+ valid_kept_open_expected = run_summary.get("expected_active_count", 0)
+
+ human_pr_match_count = run_summary.get("human_pr_match_count", 0)
+ human_pr_match_total = run_summary.get("human_pr_match_total", 0)
+ human_pr_match_rate_pct = run_summary.get("human_pr_match_rate_pct", 0.0)
+
+ workable_spec_count = run_summary.get("workable_spec_count", 0)
+ workable_spec_pass_rate = run_summary.get("avg_workable_spec_pass_rate_pct", 0)
+ avg_execution_time_seconds = run_summary.get("avg_execution_time_seconds", 0)
+
+ summary_rows = [
+ [
+ "**Quality Categorization Match**",
+ f"{int(total_tested * quality_match_pct / 100)}/{total_tested}",
+ f"**{quality_match_pct:.1f}%**"
+ ],
+ [
+ "**Effort Categorization Match**",
+ f"{int(total_tested * effort_match_pct / 100)}/{total_tested}",
+ f"**{effort_match_pct:.1f}%**"
+ ],
+ ]
+ if autoclose_expected_count > 0:
+ summary_rows.append([
+ "**Auto-Close Match (Recall)**",
+ f"{autoclose_correct_count}/{autoclose_expected_count}",
+ f"**{autoclose_recall_pct:.1f}%**"
+ ])
+ if valid_kept_open_expected > 0:
+ summary_rows.append([
+ "**Valid Issues Kept Open**",
+ f"{valid_kept_open_count}/{valid_kept_open_expected}",
+ f"**{valid_kept_open_pct:.1f}%**"
+ ])
+ if human_pr_match_total > 0:
+ summary_rows.append([
+ "**Human PR Match Rate**",
+ f"{human_pr_match_count}/{human_pr_match_total}",
+ f"**{human_pr_match_rate_pct:.1f}%**"
+ ])
+ if workable_spec_count > 0:
+ summary_rows.append([
+ "**Workable Spec Quality Score**",
+ f"{workable_spec_count} specs evaluated",
+ f"**{workable_spec_pass_rate:.1f}%**"
+ ])
+ summary_rows.append([
+ "**Avg Execution Time**",
+ "-",
+ f"**{avg_execution_time_seconds:.2f}s**"
+ ])
+
+ doc.table(["Metric", "Result", "Score"], summary_rows)
+
+ failures = run_summary.get("failures", [])
+ if failures:
+ doc.h3("ā Failed / Crashed Issues")
+ fail_rows = [
+ [f"#{f['issue_number']}", f"`{' '.join(str(f.get('error', '')).split())[:80]}`"]
+ for f in failures
+ ]
+ doc.table(["Issue", "Error Message"], fail_rows)
+ failed_ids_str = ",".join(str(f['issue_number']) for f in failures)
+ doc.text(f"**š Copy-paste to retry failed issues (paste into `issues` input):**\n```text\n{failed_ids_str}\n```")
+
+ if results:
+ doc.h3("š Detailed Issue Evaluation Results")
+ table_builder = MarkdownBuilder()
+ detail_rows = []
+
+ for r in results:
+ issue_num = r.get("issue_number")
+ title = (r.get("title") or "")[:45]
+ t_ver = str(r.get("target_version", "N/A"))[:7]
+ a_ver = str(r.get("actual_version", "N/A"))[:7]
+ ver_str = f"{t_ver} ā {a_ver}" if t_ver == a_ver else f"{t_ver} ā {a_ver} ā"
+
+ if "error" in r:
+ clean_err = " ".join(str(r.get("error", "")).split())[:35]
+ detail_rows.append([f"#{issue_num}", title, ver_str, f"CRASHED ({clean_err}...)", "-", "-", "-", "-"])
+ continue
+
+ cat_eval = r.get("categorization", {})
+ spec_grade = r.get("judge_evaluation", {})
+
+ exp_q = r.get("expected", {}).get("quality", "")
+ pred_q = cat_eval.get("predicted_quality", "")
+ q_icon = "" if cat_eval.get("quality_match") else " ā"
+ quality_str = f"{exp_q} ā {pred_q}{q_icon}"
+
+ exp_e = r.get("expected", {}).get("effort", "")
+ pred_e = cat_eval.get("predicted_effort", "")
+ effort_str = f"{exp_e} ā {pred_e}" + ("" if cat_eval.get("effort_match") else " ā") if exp_q == "OK" else "-"
+
+ hpm_val = spec_grade.get("human_pr_match")
+ if hpm_val == 1:
+ pr_match_str = "ā
"
+ elif hpm_val == 0 and exp_q == "OK":
+ pr_match_str = "ā"
+ else:
+ pr_match_str = "-"
+
+ spec_score_val = spec_grade.get("spec_score_pct", "")
+ spec_score_str = f"{spec_score_val}%" if spec_score_val != "" else "-"
+
+ reasons = spec_grade.get("reasoning", {})
+ if isinstance(reasons, dict) and reasons:
+ lines = []
+ for k, v in reasons.items():
+ val_str = str(v).replace('|', '\\|').replace('\n', ' ')
+ lines.append(f"{k}: {val_str}")
+ critique = f"{'
'.join(lines)}"
+ else:
+ critique = "-"
+
+ detail_rows.append([f"#{issue_num}", title, ver_str, quality_str, effort_str, pr_match_str, spec_score_str, critique])
+
+ table_headers = ["Issue", "Title", "Version (Target ā Actual)", "Quality (Exp ā Pred)", "Effort (Exp ā Pred)", "PR Match", "Spec Score", "Judge Critique"]
+ table_builder.table(table_headers, detail_rows)
+ doc.details("š Click to expand detailed issue-by-issue results", table_builder.render())
+
+ doc.text("---\n*Generated by Triage Eval Runner.*")
+
+ target_path = Path(filepath)
+ target_path.parent.mkdir(parents=True, exist_ok=True)
+ target_path.write_text(doc.render(), encoding="utf-8")
+
+
+def calc_summary(
+ run_dir: str,
+ note: Optional[str],
+ start_timestamp: str,
+ end_timestamp: str
+) -> Dict[str, Any]:
+ """Calculates evaluation metrics from results persisted in run_dir/issues/, prints summary report, and saves it."""
+ issues_dir = Path(run_dir) / "issues"
+ results = []
+
+ if not issues_dir.exists():
+ print(f"ā Run issues directory not found: {issues_dir}")
+ return {}
+
+ issue_files = [f for f in sorted(issues_dir.glob("gemini_cli_*.json")) if "debug" not in f.name]
+ for file_path in issue_files:
+ try:
+ results.append(json.loads(file_path.read_text(encoding="utf-8")))
+ except Exception as e:
+ print(f"ā Error reading {file_path} during summary generation: {e}")
+
+ successful_results = [r for r in results if "error" not in r]
+ failed_results = [r for r in results if "error" in r]
+
+ total_attempted = len(results)
+ total_tested = len(successful_results)
+ total_failed = len(failed_results)
+
+ AUTOCLOSE_TYPES = {"SPAM", "EMPTY", "FEATURE"}
+ total_quality_matches = 0
+ total_effort_matches = 0
+ total_expected_autoclose = 0
+ correct_autoclose = 0
+ predicted_autoclose = 0
+ human_pr_match_count = 0
+ human_pr_match_total = 0
+
+ for r in successful_results:
+ cat = r.get("categorization", {})
+ expected = r.get("expected", {})
+
+ if cat.get("quality_match"):
+ total_quality_matches += 1
+ if cat.get("effort_match"):
+ total_effort_matches += 1
+
+ exp_quality = expected.get("quality")
+ pred_quality = cat.get("predicted_quality")
+
+ if exp_quality in AUTOCLOSE_TYPES:
+ total_expected_autoclose += 1
+ if pred_quality in AUTOCLOSE_TYPES:
+ correct_autoclose += 1
+ if pred_quality in AUTOCLOSE_TYPES:
+ predicted_autoclose += 1
+
+ judge = r.get("judge_evaluation", {})
+ if isinstance(judge, dict) and "human_pr_match" in judge:
+ human_pr_match_count += int(judge.get("human_pr_match", 0))
+ human_pr_match_total += 1
+
+ total_expected_active = total_tested - total_expected_autoclose
+ false_autoclose = predicted_autoclose - correct_autoclose
+ valid_kept_open = total_expected_active - false_autoclose
+
+ spec_pass_rates = [
+ r.get("judge_evaluation", {}).get("spec_score_pct")
+ for r in successful_results
+ if r.get("judge_evaluation") and "spec_score_pct" in r.get("judge_evaluation", {})
+ ]
+ execution_times = [r.get("execution_time_seconds", 0.0) for r in successful_results]
+
+ avg_spec_pass_rate = round(sum(spec_pass_rates) / len(spec_pass_rates), 1) if spec_pass_rates else 0.0
+ avg_exec_time = round(sum(execution_times) / len(execution_times), 2) if execution_times else 0.0
+
+ run_summary = {
+ "start_timestamp": start_timestamp,
+ "end_timestamp": end_timestamp,
+ "note": note or "",
+ "total_attempted": total_attempted,
+ "total_tested": total_tested,
+ "total_failed": total_failed,
+ "failures": [
+ {"issue_number": r.get("issue_number"), "error": r.get("error")}
+ for r in failed_results
+ ],
+ "workable_spec_count": len(spec_pass_rates),
+ "quality_categorization_rate": total_quality_matches / total_tested if total_tested else 0,
+ "effort_categorization_rate": total_effort_matches / total_tested if total_tested else 0,
+ "expected_autoclose_count": total_expected_autoclose,
+ "correct_autoclose_count": correct_autoclose,
+ "autoclose_recall_rate": correct_autoclose / total_expected_autoclose if total_expected_autoclose else 0,
+ "expected_active_count": total_expected_active,
+ "valid_kept_open_count": valid_kept_open,
+ "valid_kept_open_rate": valid_kept_open / total_expected_active if total_expected_active else 0,
+ "human_pr_match_count": human_pr_match_count,
+ "human_pr_match_total": human_pr_match_total,
+ "human_pr_match_rate_pct": round((human_pr_match_count / human_pr_match_total) * 100.0, 1) if human_pr_match_total else 0.0,
+ "avg_workable_spec_pass_rate_pct": avg_spec_pass_rate,
+ "avg_execution_time_seconds": avg_exec_time
+ }
+
+ if total_failed > 0:
+ failed_ids_str = ",".join(str(r.get("issue_number")) for r in failed_results if r.get("issue_number") is not None)
+ print(f"\nā ļø Evaluation completed with {total_failed} execution error(s) ({total_tested}/{total_attempted} executed successfully).")
+ print(f"Failed Issue IDs to Retry: {failed_ids_str}")
+ else:
+ print(f"\nā
Evaluation execution completed successfully! ({total_tested}/{total_attempted} executed without error)")
+
+ _save_run_summary(run_summary, run_dir)
+ print(f"š Saved structured run results to: {run_dir}/\n")
+
+ # Write markdown summary report to run_dir/summary.md and latest_summary.md
+ md_filepath = Path(run_dir) / "summary.md"
+ _write_markdown(run_summary, results, str(md_filepath))
+
+ latest_md_filepath = PROJECT_ROOT / "evals" / "triage" / "results" / "latest_summary.md"
+ _write_markdown(run_summary, results, str(latest_md_filepath))
+
+ return run_summary
diff --git a/tools/caretaker-agent/evals/triage/helpers/worktrees.py b/tools/caretaker-agent/evals/triage/helpers/worktrees.py
new file mode 100644
index 0000000000..0feeccc1e3
--- /dev/null
+++ b/tools/caretaker-agent/evals/triage/helpers/worktrees.py
@@ -0,0 +1,46 @@
+"""Git Repository Cloning & Isolated Worktree Lifecycle Manager."""
+
+import subprocess
+from pathlib import Path
+from typing import Tuple
+
+BASE_DIR = Path(__file__).resolve().parent.parent
+TARGET_REPO_DIR = str(BASE_DIR / "target_repo")
+WORKTREES_DIR = str(BASE_DIR / "worktrees")
+
+
+def get_repo() -> str:
+ """Ensures base target repository google-gemini/gemini-cli is cloned and fetched once upfront."""
+ if not Path(TARGET_REPO_DIR).exists():
+ print(f"[EVAL] Target repository missing at {TARGET_REPO_DIR}. Cloning google-gemini/gemini-cli...")
+ subprocess.run(["git", "clone", "https://github.com/google-gemini/gemini-cli.git", TARGET_REPO_DIR], check=True, timeout=120)
+ else:
+ try:
+ subprocess.run(["git", "fetch", "--all", "--tags"], cwd=TARGET_REPO_DIR, capture_output=True, timeout=60)
+ except subprocess.TimeoutExpired:
+ print(" ā ļø [EVAL WARNING] 'git fetch' timed out after 60s. Continuing with cached repository state.")
+ return TARGET_REPO_DIR
+
+
+def add_worktree(worker_id: int, version: str) -> Tuple[str, str]:
+ """Creates an isolated, lightweight Git Worktree for a worker slot in ~10ms. Returns (worktree_dir, actual_version)."""
+ worktree_dir = str(Path(WORKTREES_DIR) / f"worker_{worker_id}")
+ Path(WORKTREES_DIR).mkdir(parents=True, exist_ok=True)
+
+ # Clean up any stale worktree for this worker slot
+ subprocess.run(["git", "worktree", "remove", "--force", worktree_dir], cwd=TARGET_REPO_DIR, capture_output=True)
+
+ actual_version = version
+ res = subprocess.run(["git", "worktree", "add", "-f", worktree_dir, version], cwd=TARGET_REPO_DIR, capture_output=True, text=True)
+ if res.returncode != 0:
+ print(f" [EVAL] Warning: Could not checkout commit '{version[:10]}' for worker {worker_id}. Falling back to 'main'.")
+ subprocess.run(["git", "worktree", "add", "-f", worktree_dir, "main"], cwd=TARGET_REPO_DIR, capture_output=True)
+ actual_version = "main"
+
+ return worktree_dir, actual_version
+
+
+def remove_worktree(worker_id: int) -> None:
+ """Removes a worker's temporary Git Worktree cleanly."""
+ worktree_dir = str(Path(WORKTREES_DIR) / f"worker_{worker_id}")
+ subprocess.run(["git", "worktree", "remove", "--force", worktree_dir], cwd=TARGET_REPO_DIR, capture_output=True)
diff --git a/tools/caretaker-agent/evals/triage/judge.md b/tools/caretaker-agent/evals/triage/judge.md
new file mode 100644
index 0000000000..ebdc8c3a05
--- /dev/null
+++ b/tools/caretaker-agent/evals/triage/judge.md
@@ -0,0 +1,71 @@
+You are an impartial AI evaluation judge. Your task is to evaluate a candidate
+Workable Spec produced by an automated triage bot by comparing it against a
+ground-truth Golden Workable Spec using a 4-criterion Rubric rated on a 0 to 2
+scale.
+
+SCALE DEFINITIONS:
+
+- 0 (Not Met / Inaccurate / Missing): The candidate spec misses key target
+ files, proposes an incorrect or hand-wavy solution (e.g., "explore index.ts"),
+ or completely fails to match the Golden Spec.
+- 1 (Partially Met / High-Level): The candidate spec identifies the correct
+ general files and general solution, but lacks specific steps, clarity, or
+ alignment present in the Golden Spec.
+- 2 (Fully Met / Excellent Match): The candidate spec accurately identifies the
+ target files, aligns closely with the root cause and step-by-step
+ implementation plan in the Golden Spec, and provides clear, actionable
+ instructions.
+
+GENERIC FAIRNESS RULE: Human PRs often include additional refactoring or
+un-reported edge-case fixes. Do NOT penalize a candidate spec for omitting extra
+refactoring that goes beyond the reported issue scope. Evaluate based on whether
+the candidate correctly solves the reported issue problem and matches the Golden
+Spec's core targets.
+
+STRICT GROUND-TRUTH RULE: You do NOT have access to the codebase. Evaluate the
+candidate spec STRICTLY by comparing its contents against the Golden Spec
+target.
+
+EVALUATE ACROSS THESE 4 GOLDEN-SPEC MATCH CRITERIA (Score 0, 1, or 2 for each):
+
+1. target_files_score (0-2): Evaluate how well the candidate's target files
+ match the Golden Spec:
+ - Score 2 (Full Credit): The candidate accurately identifies all primary
+ target files (or valid alternative target files in parenthetical format).
+ - Score 1 (Partial Credit): The candidate correctly identifies at least one
+ primary target file (or a closely related parent/child file in the same
+ call chain), but misses some key files or includes extra non-essential
+ files.
+ - Score 0 (No Credit): The candidate completely misses all target files or
+ only includes completely irrelevant files.
+2. root_cause_and_summary_score (0-2): Does the candidate's problem statement
+ and root cause analysis accurately identify the underlying defect or error?
+ (Focus strictly on diagnostic accuracy independently of target files, not fix
+ design or file path matching).
+3. implementation_plan_score (0-2): Does the step-by-step implementation plan
+ outline clear, actionable steps that align with the solution strategy in the
+ Golden Spec?
+4. testing_strategy_score (0-2): Does the testing strategy match the test file,
+ expected behavior, and verification steps in the Golden Spec (or correctly
+ identify that no automated test file is needed if the Golden Spec specifies
+ N/A)?
+
+FINAL OVERALL ASSESSMENT STEP: 5. human_pr_match: High-level evaluation
+measuring practical agent triage effectiveness.
+
+- 1 (Match): The candidate spec accurately diagnoses the defect and proposes an
+ effective, actionable fix matching the core intent of the human PR. (Award a
+ Match if the spec provides an effective solution, even if implementation steps
+ or target file paths vary slightly).
+- 0 (No Match): The candidate spec fails to address the underlying bug, proposes
+ an ineffective or unworkable fix strategy, or targets completely irrelevant
+ files.
+
+Output ONLY a raw JSON object with concise explanations per criterion: {
+"target_files_score": <0|1|2>, "root_cause_and_summary_score": <0|1|2>,
+"implementation_plan_score": <0|1|2>, "testing_strategy_score": <0|1|2>,
+"human_pr_match": <0|1>, "reasoning": { "target_files": "", "root_cause": "", "implementation_plan": "", "testing_strategy":
+"" } }
diff --git a/tools/caretaker-agent/evals/triage/judge.py b/tools/caretaker-agent/evals/triage/judge.py
new file mode 100644
index 0000000000..c30df8a3eb
--- /dev/null
+++ b/tools/caretaker-agent/evals/triage/judge.py
@@ -0,0 +1,188 @@
+"""
+Evaluation Judge Module for Gemini CLI Triage Worker.
+
+Provides evaluation functions:
+1. evaluate_categorization: Exact match string evaluation for quality & effort.
+2. judge_workable_spec: LLM-as-a-Judge grading for Workable Specs matching Golden Spec fidelity (0-2 Rubric Scale) via Gemini API.
+"""
+
+import os
+import json
+from pathlib import Path
+from typing import Any, Dict
+from dotenv import load_dotenv
+
+load_dotenv()
+
+from google import genai
+
+PROMPT_FILE = Path(__file__).parent / "judge.md"
+if not PROMPT_FILE.exists():
+ raise FileNotFoundError(f"Required judge.md prompt file missing from {PROMPT_FILE.parent}")
+
+with open(PROMPT_FILE, "r", encoding="utf-8") as f:
+ JUDGE_PROMPT = f.read()
+
+_CLIENT: Any = None
+
+
+def _get_client() -> genai.Client:
+ """Returns thread-safe cached Gemini API client instance."""
+ global _CLIENT
+ if _CLIENT is None:
+ api_key = os.environ.get("GEMINI_API_KEY")
+ _CLIENT = genai.Client(api_key=api_key)
+ return _CLIENT
+
+
+def evaluate_categorization(predicted: Dict[str, Any], expected: Dict[str, Any]) -> Dict[str, Any]:
+ """
+ Evaluates quality and effort categorization match against expected values.
+
+ Rules:
+ - Quality: Exact match between predicted quality and expected quality.
+ - Effort: If expected quality is OK, predicted effort must match expected effort.
+ If expected quality is non-OK (SPAM, NEEDS_INFO, FEATURE), predicted effort must be empty ("").
+ """
+ pred_quality = predicted.get("quality")
+ exp_quality = expected.get("expected_quality")
+
+ # 1. Quality match check
+ quality_match = (pred_quality == exp_quality)
+
+ # 2. Effort match check
+ pred_effort = predicted.get("effort_estimate")
+ exp_effort = expected.get("expected_effort")
+
+ if exp_quality == "OK":
+ effort_match = (pred_effort == exp_effort)
+ else:
+ effort_match = (pred_effort == "")
+
+ return {
+ "quality_match": quality_match,
+ "predicted_quality": pred_quality,
+ "expected_quality": exp_quality,
+ "effort_match": effort_match,
+ "predicted_effort": pred_effort,
+ "expected_effort": exp_effort,
+ }
+
+
+def judge_workable_spec(predicted_spec: Dict[str, Any], golden_spec: Dict[str, Any]) -> Dict[str, Any]:
+ """
+ Uses direct Gemini API (gemini-flash-latest) to evaluate a candidate Workable Spec
+ against a ground-truth Golden Workable Spec using a 4-criterion 0-2 Rubric measuring Golden Spec alignment.
+ """
+ default_reasoning = {
+ "target_files": "Missing predicted or golden workable spec.",
+ "root_cause": "Missing predicted or golden workable spec.",
+ "implementation_plan": "Missing predicted or golden workable spec.",
+ "testing_strategy": "Missing predicted or golden workable spec."
+ }
+
+ if not predicted_spec or not golden_spec:
+ return {
+ "target_files_score": 0,
+ "root_cause_and_summary_score": 0,
+ "implementation_plan_score": 0,
+ "testing_strategy_score": 0,
+ "human_pr_match": 0,
+ "total_points": 0,
+ "max_points": 8,
+ "spec_score_pct": 0.0,
+ "reasoning": default_reasoning
+ }
+
+ system_instruction = JUDGE_PROMPT
+
+ prompt = f"""Golden Spec Target:
+{json.dumps(golden_spec, indent=2)}
+
+Predicted Candidate Spec:
+{json.dumps(predicted_spec, indent=2)}"""
+
+ try:
+ client = _get_client()
+
+ response_schema = {
+ "type": "OBJECT",
+ "properties": {
+ "target_files_score": {"type": "INTEGER"},
+ "root_cause_and_summary_score": {"type": "INTEGER"},
+ "implementation_plan_score": {"type": "INTEGER"},
+ "testing_strategy_score": {"type": "INTEGER"},
+ "human_pr_match": {"type": "INTEGER"},
+ "reasoning": {
+ "type": "OBJECT",
+ "properties": {
+ "target_files": {"type": "STRING"},
+ "root_cause": {"type": "STRING"},
+ "implementation_plan": {"type": "STRING"},
+ "testing_strategy": {"type": "STRING"},
+ },
+ "required": ["target_files", "root_cause", "implementation_plan", "testing_strategy"],
+ },
+ },
+ "required": [
+ "target_files_score",
+ "root_cause_and_summary_score",
+ "implementation_plan_score",
+ "testing_strategy_score",
+ "human_pr_match",
+ "reasoning",
+ ],
+ }
+
+ response = client.models.generate_content(
+ model="gemini-flash-latest",
+ contents=prompt,
+ config={
+ "system_instruction": system_instruction,
+ "response_mime_type": "application/json",
+ "response_schema": response_schema
+ }
+ )
+
+ res = json.loads(response.text.strip())
+
+ tfs = int(res.get("target_files_score", 0))
+ rcs = int(res.get("root_cause_and_summary_score", 0))
+ ips = int(res.get("implementation_plan_score", 0))
+ tss = int(res.get("testing_strategy_score", 0))
+ hpm = int(res.get("human_pr_match", 0))
+
+ total_points = tfs + rcs + ips + tss
+ max_points = 8
+ score_pct = round((total_points / float(max_points)) * 100.0, 1)
+
+ reasoning = res.get("reasoning", {})
+ if not isinstance(reasoning, dict):
+ reasoning = {"summary": str(reasoning)}
+
+ res["target_files_score"] = tfs
+ res["root_cause_and_summary_score"] = rcs
+ res["implementation_plan_score"] = ips
+ res["testing_strategy_score"] = tss
+ res["human_pr_match"] = hpm
+ res["total_points"] = total_points
+ res["max_points"] = max_points
+ res["spec_score_pct"] = score_pct
+ res["reasoning"] = reasoning
+ return res
+
+ except Exception as e:
+ print(f" ā [JUDGE ERROR] {e}")
+ return {
+ "target_files_score": 0,
+ "root_cause_and_summary_score": 0,
+ "implementation_plan_score": 0,
+ "testing_strategy_score": 0,
+ "human_pr_match": 0,
+ "total_points": 0,
+ "max_points": 8,
+ "spec_score_pct": 0.0,
+ "reasoning": {
+ "error": f"Judge execution error: {e}"
+ }
+ }
diff --git a/tools/caretaker-agent/evals/triage/requirements.txt b/tools/caretaker-agent/evals/triage/requirements.txt
new file mode 100644
index 0000000000..59cf226512
--- /dev/null
+++ b/tools/caretaker-agent/evals/triage/requirements.txt
@@ -0,0 +1,4 @@
+google-cloud-firestore>=2.15.0
+google-antigravity>=0.1.0
+python-dotenv
+requests
diff --git a/tools/caretaker-agent/evals/triage/runner.py b/tools/caretaker-agent/evals/triage/runner.py
new file mode 100644
index 0000000000..bc939c1e18
--- /dev/null
+++ b/tools/caretaker-agent/evals/triage/runner.py
@@ -0,0 +1,209 @@
+"""
+Evaluation Benchmark Runner for Gemini CLI Triage Worker.
+
+Executes parallel LLM unit evaluations against curated golden issues,
+checks categorization match, evaluates Workable Specs,
+and persists structured results under evals/triage/results/.
+
+Uses Git Worktrees for 100% thread-safe parallel checkouts across different commit SHAs.
+
+CLI Usage:
+ python3 -m evals.triage.runner --issues 1,2,3 --concurrency 5 --note "test run" --no-save
+"""
+
+import os
+import sys
+import json
+import time
+import argparse
+import datetime
+from pathlib import Path
+from os.path import abspath, dirname
+from typing import Any, Dict, List, Optional
+from concurrent.futures import ProcessPoolExecutor, as_completed
+
+from dotenv import load_dotenv
+
+# Ensure repository root and cloudrun/triage-worker are in sys.path
+CARETAKER_DIR = abspath(os.path.join(dirname(__file__), "..", ".."))
+TRIAGE_WORKER_DIR = os.path.join(CARETAKER_DIR, "cloudrun", "triage-worker")
+
+if CARETAKER_DIR not in sys.path:
+ sys.path.insert(0, CARETAKER_DIR)
+if TRIAGE_WORKER_DIR not in sys.path:
+ sys.path.insert(0, TRIAGE_WORKER_DIR)
+
+load_dotenv()
+
+from triage_orchestrator import process_issue_triage
+from evals.triage.judge import evaluate_categorization, judge_workable_spec
+from evals.triage.helpers.worktrees import get_repo, add_worktree, remove_worktree
+from evals.triage.helpers.dataset import load_issues, prep_payload
+from evals.triage.helpers.summary import init_dir, save_issue_result, calc_summary
+
+
+def eval_issue(golden_issue: Dict[str, Any], worker_id: int) -> Dict[str, Any]:
+ """Evaluates a single issue under ThreadPoolExecutor using an isolated Git Worktree."""
+ issue_num = golden_issue.get("issue_number")
+ title = golden_issue.get("issue_title")
+ target_version = golden_issue.get("target_version", "main")
+ actual_version = target_version
+
+ payload = prep_payload(golden_issue)
+
+ try:
+ worktree_dir, actual_version = add_worktree(worker_id, target_version)
+ print(f"[TEST START] Issue #{issue_num} (Version: {actual_version[:10]})")
+
+ start_time = time.time()
+ success, raw_output = process_issue_triage(payload, target_cwd=worktree_dir)
+ execution_time_seconds = round(time.time() - start_time, 2)
+
+ if not success:
+ raise RuntimeError(f"Triage execution failed: {raw_output}")
+
+ try:
+ result = json.loads(raw_output)
+ except Exception:
+ cleaned_output = raw_output.replace("\\'", "'")
+ result = json.loads(cleaned_output)
+
+ metadata = result.get("triage_metadata", {})
+ predicted_spec = result.get("workable_spec", {})
+
+ cat_eval = evaluate_categorization(metadata, golden_issue)
+
+ golden_spec = golden_issue.get("expected_workable_spec", {})
+ spec_grade = {}
+ if golden_issue.get("expected_quality") == "OK" and golden_spec:
+ spec_grade = judge_workable_spec(predicted_spec, golden_spec)
+
+ record = {
+ "issue_number": issue_num,
+ "title": title,
+ "target_version": target_version,
+ "actual_version": actual_version,
+ "execution_time_seconds": execution_time_seconds,
+ "categorization": cat_eval,
+ "predicted": {"metadata": metadata, "workable_spec": predicted_spec},
+ "expected": {
+ "quality": golden_issue.get("expected_quality"),
+ "effort": golden_issue.get("expected_effort"),
+ "workable_spec": golden_issue.get("expected_workable_spec", {})
+ },
+ "judge_evaluation": spec_grade
+ }
+ if os.environ.get("LOCAL_LOG_DIR"):
+ issues_dir = Path(os.environ["LOCAL_LOG_DIR"])
+ save_issue_result(issues_dir, issue_num, record)
+
+ print(f"[TEST FINISHED] Issue #{issue_num}")
+
+ return {
+ "success": True,
+ "issue_number": issue_num,
+ "golden_issue": golden_issue,
+ "execution_time_seconds": execution_time_seconds,
+ "predicted_metadata": metadata,
+ "predicted_spec": predicted_spec,
+ "cat_eval": cat_eval,
+ "spec_grade": spec_grade
+ }
+ except Exception as e:
+ err_msg = f"{e}"
+ print(f" ā [Issue #{issue_num}] Worker execution failed: {err_msg}")
+
+ err_record = {
+ "issue_number": issue_num,
+ "title": title,
+ "target_version": target_version,
+ "actual_version": actual_version,
+ "error": err_msg,
+ "judge_evaluation": {
+ "reasoning": {"error": f"Worker execution error: {err_msg}"}
+ }
+ }
+ if os.environ.get("LOCAL_LOG_DIR"):
+ issues_dir = Path(os.environ["LOCAL_LOG_DIR"])
+ save_issue_result(issues_dir, issue_num, err_record)
+
+ return {"success": False, "issue_number": issue_num, "error": err_msg}
+ finally:
+ remove_worktree(worker_id)
+
+
+def run_suite(
+ filter_issues: Optional[List[int]] = None,
+ concurrency: int = 5,
+ note: Optional[str] = None,
+ save: bool = True
+) -> None:
+ """Runs evaluation suite using Git Worktrees."""
+ issues = load_issues(filter_issues=filter_issues)
+ if not issues:
+ print("ā No golden issues matched the specified issue filter.")
+ return
+
+ get_repo()
+ run_dir = init_dir(save)
+
+ print(f"\n========================================================")
+ print(f" Gemini CLI Triage Worker Benchmark Suite (Git Worktrees)")
+ print(f"========================================================")
+ print(f"[EVAL] Loaded {len(issues)} golden issue(s).")
+ if filter_issues:
+ print(f"[EVAL] Filtered Issues: {filter_issues}")
+ if note:
+ print(f"[EVAL] Run Note: '{note}'")
+ print(f"[EVAL] Parallel Workers: {concurrency}.")
+ print(f"[EVAL] Save Results: {save}.")
+ if run_dir:
+ print(f"[EVAL] Run Output Folder: {run_dir}/\n")
+ else:
+ print(f"[EVAL] [--no-save] Skipping disk persistence.\n")
+
+ start_timestamp = datetime.datetime.now().isoformat()
+
+ results = []
+
+ with ProcessPoolExecutor(max_workers=concurrency) as executor:
+ future_to_issue = {
+ executor.submit(eval_issue, item, worker_id=i % concurrency): item
+ for i, item in enumerate(issues)
+ }
+ for future in as_completed(future_to_issue):
+ results.append(future.result())
+
+ end_timestamp = datetime.datetime.now().isoformat()
+
+ calc_summary(
+ run_dir=run_dir,
+ note=note,
+ start_timestamp=start_timestamp,
+ end_timestamp=end_timestamp
+ )
+
+
+def main() -> None:
+ parser = argparse.ArgumentParser(description="Run parallel evaluation suite over golden issue dataset using Git Worktrees.")
+ parser.add_argument("--issues", type=str, default=None, help="Comma-separated issue numbers to test (e.g. --issues 28052,25693)")
+ parser.add_argument("--concurrency", type=int, default=5, help="Number of parallel workers (default: 5)")
+ parser.add_argument("--note", type=str, default=None, help="Optional description note for this evaluation run (saved in summary.json)")
+ parser.add_argument("--save", action=argparse.BooleanOptionalAction, default=True, help="Persist structured evaluation run results to disk under evals/triage/results/ (default: True, use --no-save to skip)")
+
+ args = parser.parse_args()
+
+ filter_issues = None
+ if args.issues:
+ filter_issues = [int(x.strip()) for x in args.issues.split(",") if x.strip().isdigit()]
+
+ run_suite(
+ filter_issues=filter_issues,
+ concurrency=args.concurrency,
+ note=args.note,
+ save=args.save
+ )
+
+
+if __name__ == "__main__":
+ main()