Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
24 commits
Select commit Hold shift + click to select a range
b0b88e8
fix(harbor): support Claude Code on Vertex AI and harden GKE execution
kweinmeister Sep 13, 2026
fea99b9
fix(harbor): add active 1-token probe for Vertex AI OpenAPI endpoints
kweinmeister Sep 13, 2026
789eecc
fix(harbor): decouple agent provider routing and harden GKE environment
kweinmeister Sep 14, 2026
fb46b86
Merge branch 'main' into fix/harbor-claude-vertex-routing
kweinmeister Sep 16, 2026
70206f7
Merge branch 'main' into fix/harbor-claude-vertex-routing
rng1995 Sep 16, 2026
c06cfb3
fix(harbor): address review comments on GKE execution and credentials…
kweinmeister Sep 16, 2026
18780b0
Merge branch 'main' into fix/harbor-claude-vertex-routing
rng1995 Sep 16, 2026
d83b700
Merge branch 'main' into fix/harbor-claude-vertex-routing
rng1995 Sep 17, 2026
34701fd
Merge branch 'main' into fix/harbor-claude-vertex-routing
rng1995 Sep 17, 2026
d2116a5
Merge branch 'main' into fix/harbor-claude-vertex-routing
kweinmeister Sep 18, 2026
f0e70cf
fix(harbor): harden GKE Workload Identity, MCP secret boundaries, and…
kweinmeister Sep 18, 2026
9e80980
Merge branch 'main' into fix/harbor-claude-vertex-routing
rng1995 Sep 22, 2026
dfa6c4f
fix(harbor): address PR #140 review on ADC token refresh overwrite an…
kweinmeister Sep 24, 2026
79310a1
Merge branch 'main' into fix/harbor-claude-vertex-routing
kweinmeister Sep 24, 2026
f8fadaf
Merge branch 'main' into fix/harbor-claude-vertex-routing
rng1995 Sep 26, 2026
cd2d599
fix(tier3): enforce GKE pod SA isolation, ADC provenance gating, and …
kweinmeister Sep 26, 2026
ad10234
Merge branch 'main' into fix/harbor-claude-vertex-routing
kweinmeister Sep 27, 2026
a1202f0
fix(tier3): harden GKE kubelet exec readiness, retry, redaction, and …
kweinmeister Sep 26, 2026
cd9405b
Merge remote-tracking branch 'origin/main' into fix/harbor-claude-ver…
kweinmeister Sep 28, 2026
b229925
fix(tier3): resolve GKE env kwargs in workflow preflight and harden A…
kweinmeister Sep 28, 2026
2b0df44
Merge remote-tracking branch 'origin/main' into fix/harbor-claude-ver…
kweinmeister Sep 29, 2026
76963c1
fix(harbor): enforce GKE metadata isolation and fix MCP placeholder s…
kweinmeister Sep 29, 2026
8e89753
Merge branch 'main' into fix/harbor-claude-vertex-routing
kweinmeister Sep 30, 2026
41cb00f
Merge branch 'main' into fix/harbor-claude-vertex-routing
kweinmeister Oct 1, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 19 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,25 @@ All notable changes to SkillEvaluator are documented in this file.

## Unreleased

### Added

- Tier 3 support for Harbor GKE execution mode (`--env-mode gke`) and `--ek`
argument forwarding. API keys, bearer tokens, and passwords passed in `--ek`
keys or values are rejected to keep credentials out of process listings, while
configuration such as rate limits and token counts is preserved. Cluster
infrastructure settings in skill configs are rejected to enforce security
boundaries in favor of host environment variables and CLI flags.
- Claude Code live agent routing for Google Cloud Vertex AI
(`CLAUDE_CODE_USE_VERTEX=1`, `ANTHROPIC_VERTEX_PROJECT_ID`, `CLOUD_ML_REGION`),
with redirect-blocking preflight probes, case-insensitive model alias
resolution, and explicit trusted-skill opt-in (`SKILLEVALUATOR_GKE_ALLOW_WORKLOAD_IDENTITY=1`
or `--ek allow_workload_identity=true`) when using single-pod GKE Workload Identity.
- Fail-closed validation for skill-authored MCP server configurations
(`evals/environment/mcp_servers.toml` and `mcp_servers.json`) that blocks
operator/provider credential references and literal secrets, while allowing
operator-approved MCP endpoints and non-LLM secrets via
`SKILLEVALUATOR_ALLOWED_MCP_HOSTS` and `SKILLEVALUATOR_ALLOWED_MCP_SECRETS`.

## 0.4.0 - 2026-09-30

### Fixed
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -61,7 +61,7 @@ tier2 = [
# Tier 3 live agent evaluation through Harbor's native environments.
tier3 = [
"skillevaluator[llm]",
"harbor==0.13.2",
"harbor[gke]==0.13.2",
# CVE-2026-59950, CVE-2026-52870, and CVE-2026-52869 fixed floor.
"mcp>=1.28.1,<2",
# CVE-2026-48522 through CVE-2026-48526 fixed floor.
Expand Down
73 changes: 71 additions & 2 deletions src/skillevaluator/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -2293,6 +2293,16 @@ def dedup_scan(
raise click.ClickException("dedup scan failed")


def _parse_environment_kwargs_cli(raw_kwargs: tuple[str, ...]) -> dict[str, str]:
"""Parse and validate environment kwargs for CLI commands, raising Click BadParameter on errors."""
from skillevaluator.tier3.commands import parse_environment_kwargs

try:
return parse_environment_kwargs(raw_kwargs)
except ValueError as exc:
raise click.BadParameter(str(exc), param_hint=["--ek"]) from exc


def _workflow_report_options(func):
func = _report_options(func)
for param in func.__click_params__:
Expand Down Expand Up @@ -2479,6 +2489,13 @@ def _tier2_workflow(
show_default=True,
help="Tier 3 progress presentation (auto uses Rich on a TTY and plain lines otherwise).",
)
@click.option(
"--ek",
"--environment-kwarg",
"environment_kwargs",
multiple=True,
help="Environment kwarg override in KEY=VALUE form (repeatable).",
)
def evaluate(
skill_path: Path,
agents: str | None,
Expand Down Expand Up @@ -2508,6 +2525,7 @@ def evaluate(
evaluated_source_revision: str | None,
evaluator_container_revision: str | None,
progress: str,
environment_kwargs: tuple[str, ...] = (),
) -> None:
"""Run Tier 3 live agent evaluation."""
from skillevaluator.evaluation import EvaluationOptions, EvaluationService
Expand All @@ -2524,6 +2542,8 @@ def evaluate(
if autopilot:
_ensure_autopilot_dataset(skill_path, progress=progress)

parsed_environment_kwargs = _parse_environment_kwargs_cli(environment_kwargs)

options = EvaluationOptions(
skill_path=skill_path,
agents=agents,
Expand All @@ -2549,6 +2569,7 @@ def evaluate(
override_memory_mb=override_memory_mb,
override_storage_mb=override_storage_mb,
evaluated_source=evaluated_source,
environment_kwargs=parsed_environment_kwargs,
)
try:
if env_mode == "local":
Expand All @@ -2567,6 +2588,31 @@ def evaluate(
padding=(0, 1),
)
)
elif env_mode == "gke":
from skillevaluator.tier3.commands import parse_agents
from skillevaluator.tier3.harbor.runner import (
is_gke_vertex_workload_identity_active,
is_gke_workload_identity_allowed,
)

if is_gke_vertex_workload_identity_active(
env_mode, parse_agents(agents)
) and is_gke_workload_identity_allowed(options.environment_kwargs):
from rich.panel import Panel
from rich.text import Text

console.print(
Panel(
Text(
"Intended for trusted skills and least-privilege service accounts. Harbor 0.13.2 single-pod "
"GKE execution shares the pod service account and metadata server with evaluated skill commands.",
style="yellow",
),
title=Text("GKE Workload Identity · Trusted Skills Only", style="bold cyan"),
border_style="yellow",
padding=(0, 1),
)
)
progress_reporter = create_progress_reporter(progress, stream=click.get_text_stream("stderr"))
engine_result = service.evaluate(options, progress_reporter=progress_reporter)
failure = service.failure_reason(engine_result)
Expand Down Expand Up @@ -2746,16 +2792,31 @@ def models_command(limit: int, as_json: bool) -> None:
is_flag=True,
help="Check resolved agent-model catalog reachability with a live credential-bearing request.",
)
def doctor(agents: str | None, env_mode: str, agent_model: tuple[str, ...], verify_models: bool) -> None:
@click.option(
"--ek",
"--environment-kwarg",
"environment_kwargs",
multiple=True,
help="Environment kwarg override in KEY=VALUE form (repeatable).",
)
def doctor(
agents: str | None,
env_mode: str,
agent_model: tuple[str, ...],
verify_models: bool,
environment_kwargs: tuple[str, ...] = (),
) -> None:
"""Check live-evaluation runtime readiness."""
from skillevaluator.tier3.commands import doctor as tier3_doctor

parsed_environment_kwargs = _parse_environment_kwargs_cli(environment_kwargs)
raise SystemExit(
tier3_doctor(
agents=agents,
env_mode=env_mode,
verify_models=verify_models,
agent_model=agent_model,
environment_kwargs=parsed_environment_kwargs,
)
)

Expand All @@ -2772,7 +2833,15 @@ def health_check(agents: str | None, env_mode: str) -> None:
"""Quick readiness check for the CLI and selected live-eval backend."""
from skillevaluator.tier3.commands import doctor as tier3_doctor

raise SystemExit(tier3_doctor(agents=agents, env_mode=env_mode, verify_models=False, agent_model=()))
raise SystemExit(
tier3_doctor(
agents=agents,
env_mode=env_mode,
verify_models=False,
agent_model=(),
environment_kwargs={},
)
)


@tier3.command("validate")
Expand Down
3 changes: 2 additions & 1 deletion src/skillevaluator/evaluation/options.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,7 @@

from __future__ import annotations

from dataclasses import dataclass, fields
from dataclasses import dataclass, field, fields
from pathlib import Path
from typing import Any

Expand Down Expand Up @@ -49,6 +49,7 @@ class EvaluationOptions:
# Supplied by the orchestration input, never inferred from repository state:
# the tree that runs the evaluator is not the tree being evaluated.
evaluated_source: dict[str, str] | None = None
environment_kwargs: dict[str, str] = field(default_factory=dict)

def engine_kwargs(self) -> dict[str, Any]:
"""Return keyword arguments (excluding ``skill_path``) for the engine."""
Expand Down
25 changes: 24 additions & 1 deletion src/skillevaluator/inference/client.py
Original file line number Diff line number Diff line change
Expand Up @@ -288,7 +288,30 @@ def completions(self, system_prompt: str, user_prompt: str) -> str:
**_token_limit_kwargs(config, self._max_tokens),
}

response = client.chat.completions.create(**call_kwargs)
try:
response = client.chat.completions.create(**call_kwargs)
except Exception as exc:
from skillevaluator.provider_config import _get_google_access_token

if config.credential_env == "ADC":
from openai import AuthenticationError

if isinstance(exc, AuthenticationError) or getattr(exc, "status_code", None) == 401:
new_token = _get_google_access_token()
if new_token:
import dataclasses

client.api_key = new_token
self._provider_config = dataclasses.replace(config, api_key=new_token)
if self._api_key is not None:
self._api_key = new_token
response = client.chat.completions.create(**call_kwargs)
else:
raise
else:
raise
else:
raise
content = response.choices[0].message.content
if not content:
raise EmptyLLMResponseError("LLM returned empty response content")
Expand Down
Loading