diff --git a/.env.example b/.env.example index 79af59da..10b98f56 100644 --- a/.env.example +++ b/.env.example @@ -51,6 +51,8 @@ FCC_SMOKE_MODEL_KIMI= FCC_SMOKE_MODEL_WAFER= FCC_SMOKE_NIM_MODELS= FCC_SMOKE_NIM_EXTRA_MODELS= +FCC_SMOKE_OPENROUTER_FREE_MODELS= +FCC_SMOKE_OPENROUTER_FREE_EXTRA_MODELS= # Thinking output diff --git a/smoke/README.md b/smoke/README.md index e0208b4d..2eb36cb8 100644 --- a/smoke/README.md +++ b/smoke/README.md @@ -63,6 +63,7 @@ Heavy/side-effectful targets are opt-in: | Target | Product scenarios | Required environment | | --- | --- | --- | | `nvidia_nim_cli` | Claude Code CLI feature matrix across NIM models | `NVIDIA_NIM_API_KEY`, Claude CLI | +| `openrouter_free_cli` | Claude Code CLI feature matrix across OpenRouter free models | `OPENROUTER_API_KEY`, Claude CLI | | `telegram` | getMe, send, edit, delete, optional manual inbound | token and chat/user ID | | `discord` | channel access, send, edit, delete, optional manual inbound | token and channel ID | | `voice` | generated WAV through local Whisper or NVIDIA NIM transcription | `VOICE_NOTE_ENABLED=true`, `FCC_SMOKE_RUN_VOICE=1` | @@ -96,6 +97,13 @@ $env:FCC_SMOKE_NIM_MODELS = "z-ai/glm-5.1,moonshotai/kimi-k2.6,minimaxai/minimax uv run pytest smoke/product -n 0 -s --tb=short ``` +```powershell +$env:FCC_LIVE_SMOKE = "1" +$env:FCC_SMOKE_TARGETS = "openrouter_free_cli" +$env:FCC_SMOKE_OPENROUTER_FREE_MODELS = "nvidia/nemotron-3-super-120b-a12b:free,openai/gpt-oss-120b:free,minimax/minimax-m2.5:free,inclusionai/ring-2.6-1t:free,poolside/laguna-m.1:free" +uv run pytest smoke/product -n 0 -s --tb=short +``` + ```powershell $env:FCC_LIVE_SMOKE = "1" $env:FCC_SMOKE_TARGETS = "messaging,config,extensibility" @@ -118,6 +126,10 @@ uv run pytest smoke/product -n 0 -s --tb=short that replace the default characterization set. - `FCC_SMOKE_NIM_EXTRA_MODELS`: optional comma-separated NVIDIA NIM CLI matrix models appended to the default or replacement set. +- `FCC_SMOKE_OPENROUTER_FREE_MODELS`: optional comma-separated OpenRouter free + CLI matrix models that replace the default characterization set. +- `FCC_SMOKE_OPENROUTER_FREE_EXTRA_MODELS`: optional comma-separated OpenRouter + free CLI matrix models appended to the default or replacement set. - `FCC_SMOKE_TIMEOUT_S`: per-request/subprocess timeout, default `45`. - `FCC_SMOKE_CLAUDE_BIN`: Claude CLI executable name, default `claude`. - `FCC_SMOKE_TELEGRAM_CHAT_ID`: Telegram chat/user ID for send/edit/delete. diff --git a/smoke/capabilities.py b/smoke/capabilities.py index 2a3efbe1..52ced768 100644 --- a/smoke/capabilities.py +++ b/smoke/capabilities.py @@ -411,7 +411,11 @@ CAPABILITY_CONTRACTS: tuple[CapabilityContract, ...] = ( "stream-json events and session id mapping", "stderr/error event and process cleanup", ("tests/cli/test_cli.py",), - ("test_claude_cli_prompt_when_available", "test_nvidia_nim_cli_matrix_e2e"), + ( + "test_claude_cli_prompt_when_available", + "test_nvidia_nim_cli_matrix_e2e", + "test_openrouter_free_cli_matrix_e2e", + ), ), CapabilityContract( "extensibility", diff --git a/smoke/features.py b/smoke/features.py index 4bb95faa..494fcb94 100644 --- a/smoke/features.py +++ b/smoke/features.py @@ -73,11 +73,17 @@ FEATURE_INVENTORY: tuple[FeatureCoverage, ...] = ( "test_api_basic_conversation_e2e", "test_claude_cli_adaptive_thinking_e2e", "test_nvidia_nim_cli_matrix_e2e", + "test_openrouter_free_cli_matrix_e2e", "test_vscode_protocol_e2e", "test_jetbrains_protocol_e2e", ), - ("api", "cli", "clients", "nvidia_nim_cli"), - ("configured provider", "FCC_SMOKE_CLAUDE_BIN for real Claude CLI"), + ("api", "cli", "clients", "nvidia_nim_cli", "openrouter_free_cli"), + ( + "configured provider", + "FCC_SMOKE_CLAUDE_BIN for real Claude CLI", + "NVIDIA_NIM_API_KEY", + "OPENROUTER_API_KEY", + ), "skip real CLI when binary is absent; configured providers must pass", ), FeatureCoverage( @@ -386,9 +392,15 @@ FEATURE_INVENTORY: tuple[FeatureCoverage, ...] = ( "test_claude_cli_adaptive_thinking_e2e", "test_claude_cli_multiturn_tool_protocol_e2e", "test_nvidia_nim_cli_matrix_e2e", + "test_openrouter_free_cli_matrix_e2e", + ), + ("cli", "nvidia_nim_cli", "openrouter_free_cli"), + ( + "FCC_SMOKE_CLAUDE_BIN", + "configured provider", + "NVIDIA_NIM_API_KEY", + "OPENROUTER_API_KEY", ), - ("cli", "nvidia_nim_cli"), - ("FCC_SMOKE_CLAUDE_BIN", "configured provider", "NVIDIA_NIM_API_KEY"), "skip only when Claude CLI binary is absent", ), FeatureCoverage( diff --git a/smoke/lib/claude_cli_matrix.py b/smoke/lib/claude_cli_matrix.py new file mode 100644 index 00000000..6a27b22f --- /dev/null +++ b/smoke/lib/claude_cli_matrix.py @@ -0,0 +1,774 @@ +"""Claude Code CLI characterization helpers for provider smoke matrices.""" + +from __future__ import annotations + +import json +import os +import re +import subprocess +import time +import uuid +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any + +from smoke.lib.config import ProviderModel, SmokeConfig, redacted +from smoke.lib.server import RunningServer + +REGRESSION_CLASSIFICATIONS = frozenset({"harness_bug", "product_failure"}) + +_HTTP_REGRESSION_PATTERNS = ( + r'POST /v1/messages[^"\n]* HTTP/1\.1" 4(?!01|03|04|08|09)\d\d', + r'POST /v1/messages[^"\n]* HTTP/1\.1" 5\d\d', +) +_UPSTREAM_UNAVAILABLE_MARKERS = ( + "upstream_unavailable", + "readtimeout", + "connecterror", + "connection refused", + "timed out", + "rate limit", + "overloaded", + "capacity", + "upstream provider", +) +_HTTP_429_PATTERNS = ( + r'HTTP/1\.[01]" 429\b', + r"\bHTTP/1\.[01] 429\b", + r"\bstatus_code=429\b", + r"\bstatus[=:]\s*429\b", + r"\b429 Too Many Requests\b", +) +_MISSING_ENV_MARKERS = ( + "api key", + "not logged in", + "authentication", + "permission denied", +) +_EMPTY_MCP_CONFIG = '{"mcpServers":{}}' +_SUBAGENT_SYSTEM_PROMPT = ( + "You are a deterministic smoke-test coordinator. Use Agent when asked to " + "use a subagent." +) + + +@dataclass(frozen=True, slots=True) +class ClaudeCliRun: + command: tuple[str, ...] + returncode: int | None + stdout: str + stderr: str + duration_s: float + timed_out: bool = False + + @property + def combined_output(self) -> str: + return f"{self.stdout}\n{self.stderr}" + + +@dataclass(frozen=True, slots=True) +class CliMatrixOutcome: + model: str + full_model: str + source: str + feature: str + outcome: str + classification: str + duration_s: float + cli_returncode: int | None + token_evidence: dict[str, Any] + request_count: int + log_path: str + stdout_excerpt: str + stderr_excerpt: str + log_excerpt: str + + +def run_claude_cli( + *, + claude_bin: str, + server: RunningServer, + config: SmokeConfig, + cwd: Path, + prompt: str, + tools: str | None, + bare: bool = True, + pre_tool_args: tuple[str, ...] = (), + extra_args: tuple[str, ...] = (), + session_id: str | None = None, + resume_session_id: str | None = None, + no_session_persistence: bool = True, +) -> ClaudeCliRun: + """Run Claude Code CLI against the local smoke proxy.""" + cwd.mkdir(parents=True, exist_ok=True) + + cmd = list( + _build_claude_cli_command( + claude_bin=claude_bin, + prompt=prompt, + tools=tools, + bare=bare, + pre_tool_args=pre_tool_args, + extra_args=extra_args, + session_id=session_id, + resume_session_id=resume_session_id, + no_session_persistence=no_session_persistence, + ) + ) + + env = os.environ.copy() + env["ANTHROPIC_BASE_URL"] = server.base_url + env["ANTHROPIC_API_URL"] = f"{server.base_url}/v1" + env.setdefault("ANTHROPIC_API_KEY", "sk-smoke-proxy") + if config.settings.anthropic_auth_token: + env["ANTHROPIC_AUTH_TOKEN"] = config.settings.anthropic_auth_token + env["TERM"] = "dumb" + env["NO_COLOR"] = "1" + env["PYTHONIOENCODING"] = "utf-8" + + started = time.monotonic() + try: + result = subprocess.run( + cmd, + cwd=cwd, + env=env, + capture_output=True, + text=True, + timeout=config.timeout_s, + check=False, + ) + except subprocess.TimeoutExpired as exc: + return ClaudeCliRun( + command=tuple(cmd), + returncode=None, + stdout=_coerce_timeout_text(exc.stdout), + stderr=_coerce_timeout_text(exc.stderr), + duration_s=time.monotonic() - started, + timed_out=True, + ) + + return ClaudeCliRun( + command=tuple(cmd), + returncode=result.returncode, + stdout=result.stdout, + stderr=result.stderr, + duration_s=time.monotonic() - started, + ) + + +def _build_claude_cli_command( + *, + claude_bin: str, + prompt: str, + tools: str | None, + bare: bool = True, + pre_tool_args: tuple[str, ...] = (), + extra_args: tuple[str, ...] = (), + session_id: str | None = None, + resume_session_id: str | None = None, + no_session_persistence: bool = True, +) -> tuple[str, ...]: + cmd: list[str] = [claude_bin] + if bare: + cmd.append("--bare") + if resume_session_id: + cmd.extend(["--resume", resume_session_id]) + if session_id: + cmd.extend(["--session-id", session_id]) + cmd.extend( + [ + "--output-format", + "stream-json", + "--include-partial-messages", + "--verbose", + "--permission-mode", + "bypassPermissions", + "--dangerously-skip-permissions", + "--model", + "sonnet", + ] + ) + if no_session_persistence: + cmd.append("--no-session-persistence") + cmd.extend(pre_tool_args) + if tools is not None: + cmd.extend(["--tools", tools]) + if tools: + cmd.extend(["--allowedTools", tools]) + cmd.extend(extra_args) + cmd.extend(["-p", prompt]) + return tuple(cmd) + + +def run_cli_feature_probes( + *, + claude_bin: str, + server: RunningServer, + smoke_config: SmokeConfig, + provider_model: ProviderModel, + model_dir: Path, + marker_prefix: str, +) -> list[CliMatrixOutcome]: + return [ + _basic_text( + claude_bin, server, smoke_config, provider_model, model_dir, marker_prefix + ), + _thinking( + claude_bin, server, smoke_config, provider_model, model_dir, marker_prefix + ), + _tool_use_roundtrip( + claude_bin, server, smoke_config, provider_model, model_dir, marker_prefix + ), + _interleaved_thinking_tool( + claude_bin, server, smoke_config, provider_model, model_dir, marker_prefix + ), + _subagent_task( + claude_bin, server, smoke_config, provider_model, model_dir, marker_prefix + ), + _compact_command( + claude_bin, server, smoke_config, provider_model, model_dir, marker_prefix + ), + ] + + +def read_log_offset(log_path: Path) -> int: + """Return the current text length of a smoke server log.""" + if not log_path.is_file(): + return 0 + return len(log_path.read_text(encoding="utf-8", errors="replace")) + + +def read_log_delta(log_path: Path, offset: int) -> str: + """Return smoke server log text written after ``offset``.""" + if not log_path.is_file(): + return "" + text = log_path.read_text(encoding="utf-8", errors="replace") + return text[offset:] + + +def token_evidence( + *, + feature: str, + marker: str, + run: ClaudeCliRun, + log_delta: str, +) -> dict[str, Any]: + """Collect compact evidence for a CLI feature probe.""" + combined = f"{run.combined_output}\n{log_delta}" + lower = combined.lower() + return { + "feature": feature, + "marker_present": bool(marker and marker in combined), + "thinking_delta_count": combined.count("thinking_delta"), + "tool_use_count": combined.count('"tool_use"'), + "tool_result_count": combined.count('"tool_result"'), + "agent_catalog_present": _tool_catalog_has(log_delta, "Agent"), + "agent_tool_count": _agent_tool_count(combined), + "agent_result_count": _agent_result_count(combined), + "task_tool_count": combined.count('"name": "Task"') + + combined.count('"name":"Task"'), + "run_in_background_false": "run_in_background" in combined and "false" in lower, + "compact_boundary": "compact_boundary" in combined, + "compact_metadata": "compact_metadata" in combined, + "http_422": 'HTTP/1.1" 422' in combined, + "http_500": bool(re.search(r'HTTP/1\.1" 5\d\d', combined)), + "timed_out": run.timed_out, + } + + +def classify_probe( + *, + run: ClaudeCliRun, + log_delta: str, + marker: str, + requires_tool_result: bool = False, + requires_agent: bool = False, + requires_task: bool = False, + requires_compact: bool = False, +) -> tuple[str, str]: + """Classify a probe without failing compatibility characterization failures.""" + combined = f"{run.combined_output}\n{log_delta}" + lower = combined.lower() + + if _has_proxy_regression(log_delta): + return "failed", "product_failure" + if run.returncode != 0 and any( + marker_text in lower for marker_text in _MISSING_ENV_MARKERS + ): + return "skipped", "missing_env" + if run.timed_out: + return "failed", "probe_timeout" + if requires_agent and not _tool_catalog_has(log_delta, "Agent"): + return "failed", "harness_bug" + + marker_ok = not marker or marker in combined + tool_ok = not requires_tool_result or '"tool_result"' in combined + agent_ok = not requires_agent or ( + _agent_tool_count(combined) > 0 and _agent_result_count(combined) > 0 + ) + task_ok = not requires_task or ( + ('"name": "Task"' in combined or '"name":"Task"' in combined) + and "run_in_background" in combined + and "false" in lower + ) + compact_ok = not requires_compact or ( + "compact_boundary" in combined + or "compact_metadata" in combined + or "/compact" in combined + or "compact" in lower + ) + cli_ok = run.returncode == 0 + + if cli_ok and marker_ok and tool_ok and agent_ok and task_ok and compact_ok: + return "passed", "passed" + if _has_upstream_unavailable_text(combined): + return "failed", "upstream_unavailable" + if not _has_proxy_request(log_delta): + return "failed", "harness_bug" + return "failed", "model_feature_failure" + + +def make_outcome( + *, + model: str, + full_model: str, + source: str, + feature: str, + marker: str, + run: ClaudeCliRun, + log_delta: str, + log_path: Path, + requires_tool_result: bool = False, + requires_agent: bool = False, + requires_task: bool = False, + requires_compact: bool = False, +) -> CliMatrixOutcome: + """Build one report outcome from a CLI run and its server log delta.""" + outcome, classification = classify_probe( + run=run, + log_delta=log_delta, + marker=marker, + requires_tool_result=requires_tool_result, + requires_agent=requires_agent, + requires_task=requires_task, + requires_compact=requires_compact, + ) + evidence = token_evidence( + feature=feature, + marker=marker, + run=run, + log_delta=log_delta, + ) + return CliMatrixOutcome( + model=model, + full_model=full_model, + source=source, + feature=feature, + outcome=outcome, + classification=classification, + duration_s=round(run.duration_s, 3), + cli_returncode=run.returncode, + token_evidence=evidence, + request_count=_request_count(log_delta), + log_path=str(log_path), + stdout_excerpt=_excerpt(run.stdout), + stderr_excerpt=_excerpt(run.stderr), + log_excerpt=_excerpt(log_delta), + ) + + +def write_matrix_report( + config: SmokeConfig, + outcomes: list[CliMatrixOutcome], + *, + target: str, + filename_prefix: str, +) -> Path: + """Write a Claude CLI compatibility matrix report.""" + config.results_dir.mkdir(parents=True, exist_ok=True) + path = ( + config.results_dir + / f"{filename_prefix}-matrix-{config.worker_id}-{int(time.time())}.json" + ) + payload = { + "started_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), + "worker_id": config.worker_id, + "target": target, + "models": sorted({outcome.full_model for outcome in outcomes}), + "outcomes": [asdict(outcome) for outcome in outcomes], + } + path.write_text(json.dumps(payload, indent=2, sort_keys=True), encoding="utf-8") + return path + + +def regression_failures(outcomes: list[CliMatrixOutcome]) -> list[str]: + """Return report lines for classifications that should fail pytest.""" + return [ + f"{outcome.full_model} {outcome.feature}: {outcome.classification}" + for outcome in outcomes + if outcome.classification in REGRESSION_CLASSIFICATIONS + ] + + +def _basic_text( + claude_bin: str, + server: RunningServer, + smoke_config: SmokeConfig, + provider_model: ProviderModel, + model_dir: Path, + marker_prefix: str, +) -> CliMatrixOutcome: + marker = _marker(marker_prefix, "BASIC") + return _run_probe( + claude_bin=claude_bin, + server=server, + smoke_config=smoke_config, + provider_model=provider_model, + workspace=model_dir / "basic_text", + feature="basic_text", + marker=marker, + prompt=f"Reply with exactly {marker} and no other text.", + tools="", + ) + + +def _thinking( + claude_bin: str, + server: RunningServer, + smoke_config: SmokeConfig, + provider_model: ProviderModel, + model_dir: Path, + marker_prefix: str, +) -> CliMatrixOutcome: + marker = _marker(marker_prefix, "THINK") + return _run_probe( + claude_bin=claude_bin, + server=server, + smoke_config=smoke_config, + provider_model=provider_model, + workspace=model_dir / "thinking", + feature="thinking", + marker=marker, + prompt=( + "Think privately about the request, then reply with exactly " + f"{marker} and no other text." + ), + tools="", + extra_args=("--effort", "high"), + ) + + +def _tool_use_roundtrip( + claude_bin: str, + server: RunningServer, + smoke_config: SmokeConfig, + provider_model: ProviderModel, + model_dir: Path, + marker_prefix: str, +) -> CliMatrixOutcome: + marker = _marker(marker_prefix, "TOOL") + workspace = model_dir / "tool_use_roundtrip" + (workspace / "smoke-read.txt").parent.mkdir(parents=True, exist_ok=True) + (workspace / "smoke-read.txt").write_text(marker, encoding="utf-8") + return _run_probe( + claude_bin=claude_bin, + server=server, + smoke_config=smoke_config, + provider_model=provider_model, + workspace=workspace, + feature="tool_use_roundtrip", + marker=marker, + prompt=( + "Use the Read tool to read smoke-read.txt. Reply with exactly the " + "secret token from that file and no other text." + ), + tools="Read", + requires_tool_result=True, + ) + + +def _interleaved_thinking_tool( + claude_bin: str, + server: RunningServer, + smoke_config: SmokeConfig, + provider_model: ProviderModel, + model_dir: Path, + marker_prefix: str, +) -> CliMatrixOutcome: + marker = _marker(marker_prefix, "INTERLEAVED") + workspace = model_dir / "interleaved_thinking_tool" + (workspace / "smoke-interleaved.txt").parent.mkdir(parents=True, exist_ok=True) + (workspace / "smoke-interleaved.txt").write_text(marker, encoding="utf-8") + return _run_probe( + claude_bin=claude_bin, + server=server, + smoke_config=smoke_config, + provider_model=provider_model, + workspace=workspace, + feature="interleaved_thinking_tool", + marker=marker, + prompt=( + "Think privately, use Read on smoke-interleaved.txt, then reply with " + "exactly the secret token from that file and no other text." + ), + tools="Read", + extra_args=("--effort", "high"), + requires_tool_result=True, + ) + + +def _subagent_task( + claude_bin: str, + server: RunningServer, + smoke_config: SmokeConfig, + provider_model: ProviderModel, + model_dir: Path, + marker_prefix: str, +) -> CliMatrixOutcome: + marker = _marker(marker_prefix, "TASK") + workspace = model_dir / "subagent_task" + (workspace / "smoke-subagent.txt").parent.mkdir(parents=True, exist_ok=True) + (workspace / "smoke-subagent.txt").write_text(marker, encoding="utf-8") + agents = json.dumps( + { + "smoke_reader": { + "description": "Reads one requested file and returns its token.", + "prompt": ( + "Read the requested file with Read and return only the token " + "inside it." + ), + "tools": ["Read"], + "permissionMode": "bypassPermissions", + "background": False, + } + } + ) + bare, tools, pre_tool_args, extra_args = _subagent_probe_options(agents) + return _run_probe( + claude_bin=claude_bin, + server=server, + smoke_config=smoke_config, + provider_model=provider_model, + workspace=workspace, + feature="subagent_task", + marker=marker, + prompt=( + "Use the smoke_reader subagent to read smoke-subagent.txt. After the " + "first agent result, reply with exactly the token and stop. Do not " + "call any other tools." + ), + tools=tools, + bare=bare, + pre_tool_args=pre_tool_args, + extra_args=extra_args, + requires_tool_result=True, + requires_agent=True, + ) + + +def _subagent_probe_options( + agents: str, +) -> tuple[bool, str, tuple[str, ...], tuple[str, ...]]: + return ( + False, + "Agent,Read", + ( + "--setting-sources", + "local", + "--strict-mcp-config", + "--mcp-config", + _EMPTY_MCP_CONFIG, + "--system-prompt", + _SUBAGENT_SYSTEM_PROMPT, + ), + ("--agents", agents), + ) + + +def _compact_command( + claude_bin: str, + server: RunningServer, + smoke_config: SmokeConfig, + provider_model: ProviderModel, + model_dir: Path, + marker_prefix: str, +) -> CliMatrixOutcome: + marker = _marker(marker_prefix, "COMPACT") + workspace = model_dir / "compact_command" + session_id = str(uuid.uuid4()) + offset = read_log_offset(server.log_path) + first = run_claude_cli( + claude_bin=claude_bin, + server=server, + config=smoke_config, + cwd=workspace, + prompt=f"Remember this smoke token: {marker}. Reply with exactly {marker}.", + tools="", + session_id=session_id, + no_session_persistence=False, + ) + second = run_claude_cli( + claude_bin=claude_bin, + server=server, + config=smoke_config, + cwd=workspace, + prompt=f"/compact preserve {marker}", + tools="", + resume_session_id=session_id, + no_session_persistence=False, + ) + log_delta = read_log_delta(server.log_path, offset) + run = ClaudeCliRun( + command=(*first.command, "&&", *second.command), + returncode=second.returncode if first.returncode == 0 else first.returncode, + stdout=f"{first.stdout}\n{second.stdout}", + stderr=f"{first.stderr}\n{second.stderr}", + duration_s=first.duration_s + second.duration_s, + timed_out=first.timed_out or second.timed_out, + ) + return make_outcome( + model=provider_model.model_name, + full_model=provider_model.full_model, + source=provider_model.source, + feature="compact_command", + marker="", + run=run, + log_delta=log_delta, + log_path=server.log_path, + requires_compact=True, + ) + + +def _run_probe( + *, + claude_bin: str, + server: RunningServer, + smoke_config: SmokeConfig, + provider_model: ProviderModel, + workspace: Path, + feature: str, + marker: str, + prompt: str, + tools: str | None, + bare: bool = True, + pre_tool_args: tuple[str, ...] = (), + extra_args: tuple[str, ...] = (), + requires_tool_result: bool = False, + requires_agent: bool = False, + requires_task: bool = False, +) -> CliMatrixOutcome: + offset = read_log_offset(server.log_path) + run = run_claude_cli( + claude_bin=claude_bin, + server=server, + config=smoke_config, + cwd=workspace, + prompt=prompt, + tools=tools, + bare=bare, + pre_tool_args=pre_tool_args, + extra_args=extra_args, + ) + log_delta = read_log_delta(server.log_path, offset) + return make_outcome( + model=provider_model.model_name, + full_model=provider_model.full_model, + source=provider_model.source, + feature=feature, + marker=marker, + run=run, + log_delta=log_delta, + log_path=server.log_path, + requires_tool_result=requires_tool_result, + requires_agent=requires_agent, + requires_task=requires_task, + ) + + +def _has_proxy_regression(log_delta: str) -> bool: + if "CREATE_MESSAGE_ERROR" in log_delta: + return True + return any(re.search(pattern, log_delta) for pattern in _HTTP_REGRESSION_PATTERNS) + + +def _has_proxy_request(log_delta: str) -> bool: + return "POST /v1/messages" in log_delta or "API_REQUEST:" in log_delta + + +def _tool_catalog_has(log_delta: str, tool_name: str) -> bool: + catalog = _first_tool_catalog(log_delta) + return ( + f"'name': '{tool_name}'" in catalog + or f'"name": "{tool_name}"' in catalog + or f'"name":"{tool_name}"' in catalog + ) + + +def _first_tool_catalog(log_delta: str) -> str: + for line in log_delta.splitlines(): + if "FULL_PAYLOAD" not in line: + continue + single_index = line.find("'tools': [") + double_index = line.find('"tools": [') + if single_index == -1 and double_index == -1: + continue + start = single_index if single_index != -1 else double_index + end_candidates = [ + index + for marker in ("'tool_choice'", '"tool_choice"', "'thinking'", '"thinking"') + if (index := line.find(marker, start)) != -1 + ] + end = min(end_candidates) if end_candidates else len(line) + return line[start:end] + return "" + + +def _agent_tool_count(text: str) -> int: + return ( + text.count('"name": "Agent"') + + text.count('"name":"Agent"') + + len( + re.findall( + r"'type': 'tool_use'[^}\n]+?'name': 'Agent'", + text, + flags=re.DOTALL, + ) + ) + ) + + +def _agent_result_count(text: str) -> int: + return text.count("agentId:") + text.count('"agentId"') + text.count("'agentId'") + + +def _has_upstream_unavailable_text(text: str) -> bool: + lower = text.lower() + if any(marker_text in lower for marker_text in _UPSTREAM_UNAVAILABLE_MARKERS): + return True + return any( + re.search(pattern, text, flags=re.IGNORECASE) for pattern in _HTTP_429_PATTERNS + ) + + +def _request_count(log_delta: str) -> int: + access_log_count = log_delta.count("POST /v1/messages") + service_log_count = log_delta.count("API_REQUEST:") + return max(access_log_count, service_log_count) + + +def _marker(scope: str, prefix: str) -> str: + return f"FCC_{scope}_{prefix}_{uuid.uuid4().hex[:8].upper()}" + + +def _excerpt(value: str, *, max_chars: int = 2400) -> str: + if len(value) <= max_chars: + return redacted(value) + return redacted(value[-max_chars:]) + + +def _coerce_timeout_text(value: str | bytes | None) -> str: + if value is None: + return "" + if isinstance(value, bytes): + return value.decode("utf-8", errors="replace") + return value diff --git a/smoke/lib/config.py b/smoke/lib/config.py index a5947cb0..f23f28c0 100644 --- a/smoke/lib/config.py +++ b/smoke/lib/config.py @@ -28,11 +28,13 @@ DEFAULT_TARGETS = frozenset( } ) SIDE_EFFECT_TARGETS = frozenset({"discord", "telegram", "voice"}) -OPT_IN_TARGETS = frozenset({"nvidia_nim_cli"}) +OPT_IN_TARGETS = frozenset({"nvidia_nim_cli", "openrouter_free_cli"}) ALL_TARGETS = DEFAULT_TARGETS | SIDE_EFFECT_TARGETS | OPT_IN_TARGETS TARGET_ALIASES = { "contract": "api", "nim_cli": "nvidia_nim_cli", + "openrouter_cli": "openrouter_free_cli", + "openrouter_free": "openrouter_free_cli", "optimizations": "api", "thinking": "providers", "vscode": "clients", @@ -58,6 +60,14 @@ NVIDIA_NIM_CLI_DEFAULT_MODELS: tuple[str, ...] = ( "deepseek-ai/deepseek-v4-flash", ) +OPENROUTER_FREE_CLI_DEFAULT_MODELS: tuple[str, ...] = ( + "nvidia/nemotron-3-super-120b-a12b:free", + "openai/gpt-oss-120b:free", + "minimax/minimax-m2.5:free", + "inclusionai/ring-2.6-1t:free", + "poolside/laguna-m.1:free", +) + TARGET_REQUIRED_ENV: dict[str, tuple[str, ...]] = { "api": (), @@ -77,6 +87,10 @@ TARGET_REQUIRED_ENV: dict[str, tuple[str, ...]] = { "NVIDIA_NIM_API_KEY", "FCC_SMOKE_CLAUDE_BIN or claude on PATH", ), + "openrouter_free_cli": ( + "OPENROUTER_API_KEY", + "FCC_SMOKE_CLAUDE_BIN or claude on PATH", + ), "telegram": ( "TELEGRAM_BOT_TOKEN", "ALLOWED_TELEGRAM_USER_ID or FCC_SMOKE_TELEGRAM_CHAT_ID", @@ -183,6 +197,13 @@ class SmokeConfig: for full_model, source in nvidia_nim_cli_model_refs().items() ] + def openrouter_free_cli_models(self) -> list[ProviderModel]: + """Return OpenRouter free models for Claude Code CLI characterization.""" + return [ + ProviderModel(provider="open_router", full_model=full_model, source=source) + for full_model, source in openrouter_free_cli_model_refs().items() + ] + def _include_provider_in_smoke( self, provider: str, mapped_providers: set[str] ) -> bool: @@ -295,6 +316,40 @@ def nvidia_nim_cli_model_refs( return normalized +def openrouter_free_cli_model_refs( + env: Mapping[str, str] | None = None, +) -> dict[str, str]: + """Return normalized OpenRouter free CLI matrix model refs in deterministic order.""" + source = env if env is not None else os.environ + explicit_models = _parse_csv_ordered(source.get("FCC_SMOKE_OPENROUTER_FREE_MODELS")) + extra_models = _parse_csv_ordered( + source.get("FCC_SMOKE_OPENROUTER_FREE_EXTRA_MODELS") + ) + + if "FCC_SMOKE_OPENROUTER_FREE_MODELS" in source and not explicit_models: + raise ValueError( + "FCC_SMOKE_OPENROUTER_FREE_MODELS must list at least one model" + ) + + models: list[tuple[str, str]] = [] + base_models = explicit_models or OPENROUTER_FREE_CLI_DEFAULT_MODELS + base_source = ( + "FCC_SMOKE_OPENROUTER_FREE_MODELS" + if explicit_models + else "openrouter_free_cli_default" + ) + models.extend((model, base_source) for model in base_models) + models.extend( + (model, "FCC_SMOKE_OPENROUTER_FREE_EXTRA_MODELS") for model in extra_models + ) + + normalized: dict[str, str] = {} + for raw_model, model_source in models: + full_model = _normalize_provider_model("open_router", raw_model) + normalized.setdefault(full_model, model_source) + return normalized + + def auth_headers(token: str | None = None) -> dict[str, str]: settings = get_settings() resolved = token if token is not None else settings.anthropic_auth_token diff --git a/smoke/lib/nvidia_nim_cli.py b/smoke/lib/nvidia_nim_cli.py deleted file mode 100644 index 30a85a54..00000000 --- a/smoke/lib/nvidia_nim_cli.py +++ /dev/null @@ -1,350 +0,0 @@ -"""Claude Code CLI characterization helpers for NVIDIA NIM smoke tests.""" - -from __future__ import annotations - -import json -import os -import re -import subprocess -import time -from dataclasses import asdict, dataclass -from pathlib import Path -from typing import Any - -from smoke.lib.config import SmokeConfig, redacted -from smoke.lib.server import RunningServer - -REGRESSION_CLASSIFICATIONS = frozenset({"harness_bug", "product_failure"}) - -_HTTP_REGRESSION_PATTERNS = ( - r'POST /v1/messages[^"\n]* HTTP/1\.1" 4(?!01|03|04|08|09)\d\d', - r'POST /v1/messages[^"\n]* HTTP/1\.1" 5\d\d', -) -_UPSTREAM_UNAVAILABLE_MARKERS = ( - "upstream_unavailable", - "readtimeout", - "connecterror", - "connection refused", - "timed out", - "rate limit", - "429", - "overloaded", - "capacity", - "upstream provider", -) -_MISSING_ENV_MARKERS = ( - "api key", - "not logged in", - "authentication", - "permission denied", -) - - -@dataclass(frozen=True, slots=True) -class ClaudeCliRun: - command: tuple[str, ...] - returncode: int | None - stdout: str - stderr: str - duration_s: float - timed_out: bool = False - - @property - def combined_output(self) -> str: - return f"{self.stdout}\n{self.stderr}" - - -@dataclass(frozen=True, slots=True) -class NimCliMatrixOutcome: - model: str - full_model: str - source: str - feature: str - outcome: str - classification: str - duration_s: float - cli_returncode: int | None - token_evidence: dict[str, Any] - request_count: int - log_path: str - stdout_excerpt: str - stderr_excerpt: str - log_excerpt: str - - -def run_claude_cli( - *, - claude_bin: str, - server: RunningServer, - config: SmokeConfig, - cwd: Path, - prompt: str, - tools: str | None, - extra_args: tuple[str, ...] = (), - session_id: str | None = None, - resume_session_id: str | None = None, - no_session_persistence: bool = True, -) -> ClaudeCliRun: - """Run Claude Code CLI against the local smoke proxy.""" - cwd.mkdir(parents=True, exist_ok=True) - - cmd: list[str] = [claude_bin, "--bare"] - if resume_session_id: - cmd.extend(["--resume", resume_session_id]) - if session_id: - cmd.extend(["--session-id", session_id]) - cmd.extend( - [ - "--output-format", - "stream-json", - "--include-partial-messages", - "--verbose", - "--permission-mode", - "bypassPermissions", - "--dangerously-skip-permissions", - "--model", - "sonnet", - ] - ) - if no_session_persistence: - cmd.append("--no-session-persistence") - if tools is not None: - cmd.extend(["--tools", tools]) - if tools: - cmd.extend(["--allowedTools", tools]) - cmd.extend(extra_args) - cmd.extend(["-p", prompt]) - - env = os.environ.copy() - env["ANTHROPIC_BASE_URL"] = server.base_url - env["ANTHROPIC_API_URL"] = f"{server.base_url}/v1" - env.setdefault("ANTHROPIC_API_KEY", "sk-smoke-proxy") - if config.settings.anthropic_auth_token: - env["ANTHROPIC_AUTH_TOKEN"] = config.settings.anthropic_auth_token - env["TERM"] = "dumb" - env["NO_COLOR"] = "1" - env["PYTHONIOENCODING"] = "utf-8" - - started = time.monotonic() - try: - result = subprocess.run( - cmd, - cwd=cwd, - env=env, - capture_output=True, - text=True, - timeout=config.timeout_s, - check=False, - ) - except subprocess.TimeoutExpired as exc: - return ClaudeCliRun( - command=tuple(cmd), - returncode=None, - stdout=_coerce_timeout_text(exc.stdout), - stderr=_coerce_timeout_text(exc.stderr), - duration_s=time.monotonic() - started, - timed_out=True, - ) - - return ClaudeCliRun( - command=tuple(cmd), - returncode=result.returncode, - stdout=result.stdout, - stderr=result.stderr, - duration_s=time.monotonic() - started, - ) - - -def read_log_offset(log_path: Path) -> int: - """Return the current text length of a smoke server log.""" - if not log_path.is_file(): - return 0 - return len(log_path.read_text(encoding="utf-8", errors="replace")) - - -def read_log_delta(log_path: Path, offset: int) -> str: - """Return smoke server log text written after ``offset``.""" - if not log_path.is_file(): - return "" - text = log_path.read_text(encoding="utf-8", errors="replace") - return text[offset:] - - -def token_evidence( - *, - feature: str, - marker: str, - run: ClaudeCliRun, - log_delta: str, -) -> dict[str, Any]: - """Collect compact evidence for a CLI feature probe.""" - combined = f"{run.combined_output}\n{log_delta}" - lower = combined.lower() - return { - "feature": feature, - "marker_present": bool(marker and marker in combined), - "thinking_delta_count": combined.count("thinking_delta"), - "tool_use_count": combined.count('"tool_use"'), - "tool_result_count": combined.count('"tool_result"'), - "task_tool_count": combined.count('"name": "Task"') - + combined.count('"name":"Task"'), - "run_in_background_false": "run_in_background" in combined and "false" in lower, - "compact_boundary": "compact_boundary" in combined, - "compact_metadata": "compact_metadata" in combined, - "http_422": 'HTTP/1.1" 422' in combined, - "http_500": bool(re.search(r'HTTP/1\.1" 5\d\d', combined)), - "timed_out": run.timed_out, - } - - -def classify_probe( - *, - run: ClaudeCliRun, - log_delta: str, - marker: str, - requires_tool_result: bool = False, - requires_task: bool = False, - requires_compact: bool = False, -) -> tuple[str, str]: - """Classify a probe without failing compatibility characterization failures.""" - combined = f"{run.combined_output}\n{log_delta}" - lower = combined.lower() - - if _has_proxy_regression(log_delta): - return "failed", "product_failure" - if run.returncode != 0 and any( - marker_text in lower for marker_text in _MISSING_ENV_MARKERS - ): - return "skipped", "missing_env" - if run.timed_out: - return "failed", "probe_timeout" - - marker_ok = not marker or marker in combined - tool_ok = not requires_tool_result or '"tool_result"' in combined - task_ok = not requires_task or ( - ('"name": "Task"' in combined or '"name":"Task"' in combined) - and "run_in_background" in combined - and "false" in lower - ) - compact_ok = not requires_compact or ( - "compact_boundary" in combined - or "compact_metadata" in combined - or "/compact" in combined - or "compact" in lower - ) - cli_ok = run.returncode == 0 - - if cli_ok and marker_ok and tool_ok and task_ok and compact_ok: - return "passed", "passed" - if any(marker_text in lower for marker_text in _UPSTREAM_UNAVAILABLE_MARKERS): - return "failed", "upstream_unavailable" - if not _has_proxy_request(log_delta): - return "failed", "harness_bug" - return "failed", "model_feature_failure" - - -def make_outcome( - *, - model: str, - full_model: str, - source: str, - feature: str, - marker: str, - run: ClaudeCliRun, - log_delta: str, - log_path: Path, - requires_tool_result: bool = False, - requires_task: bool = False, - requires_compact: bool = False, -) -> NimCliMatrixOutcome: - """Build one report outcome from a CLI run and its server log delta.""" - outcome, classification = classify_probe( - run=run, - log_delta=log_delta, - marker=marker, - requires_tool_result=requires_tool_result, - requires_task=requires_task, - requires_compact=requires_compact, - ) - evidence = token_evidence( - feature=feature, - marker=marker, - run=run, - log_delta=log_delta, - ) - return NimCliMatrixOutcome( - model=model, - full_model=full_model, - source=source, - feature=feature, - outcome=outcome, - classification=classification, - duration_s=round(run.duration_s, 3), - cli_returncode=run.returncode, - token_evidence=evidence, - request_count=_request_count(log_delta), - log_path=str(log_path), - stdout_excerpt=_excerpt(run.stdout), - stderr_excerpt=_excerpt(run.stderr), - log_excerpt=_excerpt(log_delta), - ) - - -def write_matrix_report( - config: SmokeConfig, - outcomes: list[NimCliMatrixOutcome], -) -> Path: - """Write the NVIDIA NIM CLI compatibility matrix report.""" - config.results_dir.mkdir(parents=True, exist_ok=True) - path = ( - config.results_dir - / f"nvidia-nim-cli-matrix-{config.worker_id}-{int(time.time())}.json" - ) - payload = { - "started_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), - "worker_id": config.worker_id, - "target": "nvidia_nim_cli", - "models": sorted({outcome.full_model for outcome in outcomes}), - "outcomes": [asdict(outcome) for outcome in outcomes], - } - path.write_text(json.dumps(payload, indent=2, sort_keys=True), encoding="utf-8") - return path - - -def regression_failures(outcomes: list[NimCliMatrixOutcome]) -> list[str]: - """Return report lines for classifications that should fail pytest.""" - return [ - f"{outcome.full_model} {outcome.feature}: {outcome.classification}" - for outcome in outcomes - if outcome.classification in REGRESSION_CLASSIFICATIONS - ] - - -def _has_proxy_regression(log_delta: str) -> bool: - if "CREATE_MESSAGE_ERROR" in log_delta: - return True - return any(re.search(pattern, log_delta) for pattern in _HTTP_REGRESSION_PATTERNS) - - -def _has_proxy_request(log_delta: str) -> bool: - return "POST /v1/messages" in log_delta or "API_REQUEST:" in log_delta - - -def _request_count(log_delta: str) -> int: - access_log_count = log_delta.count("POST /v1/messages") - service_log_count = log_delta.count("API_REQUEST:") - return max(access_log_count, service_log_count) - - -def _excerpt(value: str, *, max_chars: int = 2400) -> str: - if len(value) <= max_chars: - return redacted(value) - return redacted(value[-max_chars:]) - - -def _coerce_timeout_text(value: str | bytes | None) -> str: - if value is None: - return "" - if isinstance(value, bytes): - return value.decode("utf-8", errors="replace") - return value diff --git a/smoke/product/test_nvidia_nim_cli_product_live.py b/smoke/product/test_nvidia_nim_cli_product_live.py index f7e3e391..674db1be 100644 --- a/smoke/product/test_nvidia_nim_cli_product_live.py +++ b/smoke/product/test_nvidia_nim_cli_product_live.py @@ -1,25 +1,18 @@ from __future__ import annotations -import json import shutil -import uuid from pathlib import Path import pytest -from smoke.lib.config import ProviderModel, SmokeConfig -from smoke.lib.e2e import SmokeServerDriver -from smoke.lib.nvidia_nim_cli import ( - ClaudeCliRun, - NimCliMatrixOutcome, - make_outcome, - read_log_delta, - read_log_offset, +from smoke.lib.claude_cli_matrix import ( + CliMatrixOutcome, regression_failures, - run_claude_cli, + run_cli_feature_probes, write_matrix_report, ) -from smoke.lib.server import RunningServer +from smoke.lib.config import SmokeConfig +from smoke.lib.e2e import SmokeServerDriver pytestmark = [pytest.mark.live, pytest.mark.smoke_target("nvidia_nim_cli")] @@ -36,7 +29,7 @@ def test_nvidia_nim_cli_matrix_e2e(smoke_config: SmokeConfig, tmp_path: Path) -> if not provider_models: pytest.skip("missing_env: no NVIDIA NIM CLI smoke models configured") - outcomes: list[NimCliMatrixOutcome] = [] + outcomes: list[CliMatrixOutcome] = [] for provider_model in provider_models: with SmokeServerDriver( smoke_config, @@ -49,31 +42,23 @@ def test_nvidia_nim_cli_matrix_e2e(smoke_config: SmokeConfig, tmp_path: Path) -> "LOG_RAW_SSE_EVENTS": "true", }, ).run() as server: - model_dir = tmp_path / _slug(provider_model.model_name) outcomes.extend( - [ - _basic_text( - claude_bin, server, smoke_config, provider_model, model_dir - ), - _thinking( - claude_bin, server, smoke_config, provider_model, model_dir - ), - _tool_use_roundtrip( - claude_bin, server, smoke_config, provider_model, model_dir - ), - _interleaved_thinking_tool( - claude_bin, server, smoke_config, provider_model, model_dir - ), - _subagent_task( - claude_bin, server, smoke_config, provider_model, model_dir - ), - _compact_command( - claude_bin, server, smoke_config, provider_model, model_dir - ), - ] + run_cli_feature_probes( + claude_bin=claude_bin, + server=server, + smoke_config=smoke_config, + provider_model=provider_model, + model_dir=tmp_path / _slug(provider_model.model_name), + marker_prefix="NIM", + ) ) - report_path = write_matrix_report(smoke_config, outcomes) + report_path = write_matrix_report( + smoke_config, + outcomes, + target="nvidia_nim_cli", + filename_prefix="nvidia-nim-cli", + ) failures = regression_failures(outcomes) assert not failures, ( f"NVIDIA NIM CLI matrix regressions written to {report_path}:\n" @@ -81,245 +66,5 @@ def test_nvidia_nim_cli_matrix_e2e(smoke_config: SmokeConfig, tmp_path: Path) -> ) -def _basic_text( - claude_bin: str, - server: RunningServer, - smoke_config: SmokeConfig, - provider_model: ProviderModel, - model_dir: Path, -) -> NimCliMatrixOutcome: - marker = _marker("BASIC") - return _run_probe( - claude_bin=claude_bin, - server=server, - smoke_config=smoke_config, - provider_model=provider_model, - workspace=model_dir / "basic_text", - feature="basic_text", - marker=marker, - prompt=f"Reply with exactly {marker} and no other text.", - tools="", - ) - - -def _thinking( - claude_bin: str, - server: RunningServer, - smoke_config: SmokeConfig, - provider_model: ProviderModel, - model_dir: Path, -) -> NimCliMatrixOutcome: - marker = _marker("THINK") - return _run_probe( - claude_bin=claude_bin, - server=server, - smoke_config=smoke_config, - provider_model=provider_model, - workspace=model_dir / "thinking", - feature="thinking", - marker=marker, - prompt=( - "Think privately about the request, then reply with exactly " - f"{marker} and no other text." - ), - tools="", - extra_args=("--effort", "high"), - ) - - -def _tool_use_roundtrip( - claude_bin: str, - server: RunningServer, - smoke_config: SmokeConfig, - provider_model: ProviderModel, - model_dir: Path, -) -> NimCliMatrixOutcome: - marker = _marker("TOOL") - workspace = model_dir / "tool_use_roundtrip" - (workspace / "smoke-read.txt").parent.mkdir(parents=True, exist_ok=True) - (workspace / "smoke-read.txt").write_text(marker, encoding="utf-8") - return _run_probe( - claude_bin=claude_bin, - server=server, - smoke_config=smoke_config, - provider_model=provider_model, - workspace=workspace, - feature="tool_use_roundtrip", - marker=marker, - prompt=( - "Use the Read tool to read smoke-read.txt. Reply with exactly the " - "secret token from that file and no other text." - ), - tools="Read", - requires_tool_result=True, - ) - - -def _interleaved_thinking_tool( - claude_bin: str, - server: RunningServer, - smoke_config: SmokeConfig, - provider_model: ProviderModel, - model_dir: Path, -) -> NimCliMatrixOutcome: - marker = _marker("INTERLEAVED") - workspace = model_dir / "interleaved_thinking_tool" - (workspace / "smoke-interleaved.txt").parent.mkdir(parents=True, exist_ok=True) - (workspace / "smoke-interleaved.txt").write_text(marker, encoding="utf-8") - return _run_probe( - claude_bin=claude_bin, - server=server, - smoke_config=smoke_config, - provider_model=provider_model, - workspace=workspace, - feature="interleaved_thinking_tool", - marker=marker, - prompt=( - "Think privately, use Read on smoke-interleaved.txt, then reply with " - "exactly the secret token from that file and no other text." - ), - tools="Read", - extra_args=("--effort", "high"), - requires_tool_result=True, - ) - - -def _subagent_task( - claude_bin: str, - server: RunningServer, - smoke_config: SmokeConfig, - provider_model: ProviderModel, - model_dir: Path, -) -> NimCliMatrixOutcome: - marker = _marker("TASK") - workspace = model_dir / "subagent_task" - (workspace / "smoke-subagent.txt").parent.mkdir(parents=True, exist_ok=True) - (workspace / "smoke-subagent.txt").write_text(marker, encoding="utf-8") - agents = json.dumps( - { - "smoke_reader": { - "description": "Reads one requested file and returns its token.", - "prompt": ( - "Read the requested file with Read and return only the token " - "inside it." - ), - } - } - ) - return _run_probe( - claude_bin=claude_bin, - server=server, - smoke_config=smoke_config, - provider_model=provider_model, - workspace=workspace, - feature="subagent_task", - marker=marker, - prompt=( - "Use the smoke_reader subagent with Task to read smoke-subagent.txt. " - "Reply with exactly the token the subagent returns and no other text." - ), - tools="Task,Read", - extra_args=("--agents", agents), - requires_tool_result=True, - ) - - -def _compact_command( - claude_bin: str, - server: RunningServer, - smoke_config: SmokeConfig, - provider_model: ProviderModel, - model_dir: Path, -) -> NimCliMatrixOutcome: - marker = _marker("COMPACT") - workspace = model_dir / "compact_command" - session_id = str(uuid.uuid4()) - offset = read_log_offset(server.log_path) - first = run_claude_cli( - claude_bin=claude_bin, - server=server, - config=smoke_config, - cwd=workspace, - prompt=f"Remember this smoke token: {marker}. Reply with exactly {marker}.", - tools="", - session_id=session_id, - no_session_persistence=False, - ) - second = run_claude_cli( - claude_bin=claude_bin, - server=server, - config=smoke_config, - cwd=workspace, - prompt=f"/compact preserve {marker}", - tools="", - resume_session_id=session_id, - no_session_persistence=False, - ) - log_delta = read_log_delta(server.log_path, offset) - run = ClaudeCliRun( - command=(*first.command, "&&", *second.command), - returncode=second.returncode if first.returncode == 0 else first.returncode, - stdout=f"{first.stdout}\n{second.stdout}", - stderr=f"{first.stderr}\n{second.stderr}", - duration_s=first.duration_s + second.duration_s, - timed_out=first.timed_out or second.timed_out, - ) - return make_outcome( - model=provider_model.model_name, - full_model=provider_model.full_model, - source=provider_model.source, - feature="compact_command", - marker="", - run=run, - log_delta=log_delta, - log_path=server.log_path, - requires_compact=True, - ) - - -def _run_probe( - *, - claude_bin: str, - server: RunningServer, - smoke_config: SmokeConfig, - provider_model: ProviderModel, - workspace: Path, - feature: str, - marker: str, - prompt: str, - tools: str | None, - extra_args: tuple[str, ...] = (), - requires_tool_result: bool = False, - requires_task: bool = False, -) -> NimCliMatrixOutcome: - offset = read_log_offset(server.log_path) - run = run_claude_cli( - claude_bin=claude_bin, - server=server, - config=smoke_config, - cwd=workspace, - prompt=prompt, - tools=tools, - extra_args=extra_args, - ) - log_delta = read_log_delta(server.log_path, offset) - return make_outcome( - model=provider_model.model_name, - full_model=provider_model.full_model, - source=provider_model.source, - feature=feature, - marker=marker, - run=run, - log_delta=log_delta, - log_path=server.log_path, - requires_tool_result=requires_tool_result, - requires_task=requires_task, - ) - - -def _marker(prefix: str) -> str: - return f"FCC_NIM_{prefix}_{uuid.uuid4().hex[:8].upper()}" - - def _slug(value: str) -> str: return "".join(char if char.isalnum() else "-" for char in value).strip("-") diff --git a/smoke/product/test_openrouter_free_cli_product_live.py b/smoke/product/test_openrouter_free_cli_product_live.py new file mode 100644 index 00000000..ac36b8d0 --- /dev/null +++ b/smoke/product/test_openrouter_free_cli_product_live.py @@ -0,0 +1,72 @@ +from __future__ import annotations + +import shutil +from pathlib import Path + +import pytest + +from smoke.lib.claude_cli_matrix import ( + CliMatrixOutcome, + regression_failures, + run_cli_feature_probes, + write_matrix_report, +) +from smoke.lib.config import SmokeConfig +from smoke.lib.e2e import SmokeServerDriver + +pytestmark = [pytest.mark.live, pytest.mark.smoke_target("openrouter_free_cli")] + + +def test_openrouter_free_cli_matrix_e2e( + smoke_config: SmokeConfig, tmp_path: Path +) -> None: + if not smoke_config.has_provider_configuration("open_router"): + pytest.skip("missing_env: OPENROUTER_API_KEY is not configured") + + claude_bin = shutil.which(smoke_config.claude_bin) + if not claude_bin: + pytest.skip(f"missing_env: Claude CLI not found: {smoke_config.claude_bin}") + + provider_models = smoke_config.openrouter_free_cli_models() + if not provider_models: + pytest.skip("missing_env: no OpenRouter free CLI smoke models configured") + + outcomes: list[CliMatrixOutcome] = [] + for provider_model in provider_models: + with SmokeServerDriver( + smoke_config, + name=f"product-openrouter-free-cli-{_slug(provider_model.model_name)}", + env_overrides={ + "MODEL": provider_model.full_model, + "MESSAGING_PLATFORM": "none", + "ENABLE_MODEL_THINKING": "true", + "LOG_RAW_API_PAYLOADS": "true", + "LOG_RAW_SSE_EVENTS": "true", + }, + ).run() as server: + outcomes.extend( + run_cli_feature_probes( + claude_bin=claude_bin, + server=server, + smoke_config=smoke_config, + provider_model=provider_model, + model_dir=tmp_path / _slug(provider_model.model_name), + marker_prefix="OPENROUTER_FREE", + ) + ) + + report_path = write_matrix_report( + smoke_config, + outcomes, + target="openrouter_free_cli", + filename_prefix="openrouter-free-cli", + ) + failures = regression_failures(outcomes) + assert not failures, ( + f"OpenRouter free CLI matrix regressions written to {report_path}:\n" + + "\n".join(failures) + ) + + +def _slug(value: str) -> str: + return "".join(char if char.isalnum() else "-" for char in value).strip("-") diff --git a/tests/contracts/test_nvidia_nim_cli_matrix.py b/tests/contracts/test_nvidia_nim_cli_matrix.py index 70a622dc..57959b36 100644 --- a/tests/contracts/test_nvidia_nim_cli_matrix.py +++ b/tests/contracts/test_nvidia_nim_cli_matrix.py @@ -4,13 +4,15 @@ import json from pathlib import Path from config.settings import Settings -from smoke.lib.config import DEFAULT_TARGETS, SmokeConfig -from smoke.lib.nvidia_nim_cli import ( +from smoke.lib.claude_cli_matrix import ( ClaudeCliRun, + _build_claude_cli_command, + _subagent_probe_options, make_outcome, regression_failures, write_matrix_report, ) +from smoke.lib.config import DEFAULT_TARGETS, SmokeConfig def _smoke_config(tmp_path: Path) -> SmokeConfig: @@ -51,7 +53,12 @@ def test_nvidia_nim_cli_matrix_report_shape_and_redaction( log_path=tmp_path / "server.log", ) - path = write_matrix_report(_smoke_config(tmp_path), [outcome]) + path = write_matrix_report( + _smoke_config(tmp_path), + [outcome], + target="nvidia_nim_cli", + filename_prefix="nvidia-nim-cli", + ) payload = json.loads(path.read_text(encoding="utf-8")) assert path.name.startswith("nvidia-nim-cli-matrix-test-worker-") @@ -62,9 +69,56 @@ def test_nvidia_nim_cli_matrix_report_shape_and_redaction( assert saved["classification"] == "passed" assert saved["request_count"] == 1 assert saved["token_evidence"]["marker_present"] is True + assert saved["token_evidence"]["agent_catalog_present"] is False + assert saved["token_evidence"]["agent_tool_count"] == 0 + assert saved["token_evidence"]["agent_result_count"] == 0 assert "secret-nim-key" not in path.read_text(encoding="utf-8") +def test_openrouter_free_cli_matrix_report_shape_and_redaction( + tmp_path: Path, monkeypatch +) -> None: + monkeypatch.setenv("OPENROUTER_API_KEY", "secret-openrouter-key") + run = ClaudeCliRun( + command=("claude", "-p", "redacted"), + returncode=0, + stdout="FCC_OPENROUTER_FREE_BASIC secret-openrouter-key", + stderr="", + duration_s=1.25, + ) + outcome = make_outcome( + model="openai/gpt-oss-120b:free", + full_model="open_router/openai/gpt-oss-120b:free", + source="openrouter_free_cli_default", + feature="basic_text", + marker="FCC_OPENROUTER_FREE_BASIC", + run=run, + log_delta='POST /v1/messages HTTP/1.1" 200 OK secret-openrouter-key', + log_path=tmp_path / "server.log", + ) + + path = write_matrix_report( + _smoke_config(tmp_path), + [outcome], + target="openrouter_free_cli", + filename_prefix="openrouter-free-cli", + ) + payload = json.loads(path.read_text(encoding="utf-8")) + + assert path.name.startswith("openrouter-free-cli-matrix-test-worker-") + assert payload["target"] == "openrouter_free_cli" + assert payload["models"] == ["open_router/openai/gpt-oss-120b:free"] + saved = payload["outcomes"][0] + assert saved["feature"] == "basic_text" + assert saved["classification"] == "passed" + assert saved["request_count"] == 1 + assert saved["token_evidence"]["marker_present"] is True + assert saved["token_evidence"]["agent_catalog_present"] is False + assert saved["token_evidence"]["agent_tool_count"] == 0 + assert saved["token_evidence"]["agent_result_count"] == 0 + assert "secret-openrouter-key" not in path.read_text(encoding="utf-8") + + def test_nvidia_nim_cli_matrix_regression_detection(tmp_path: Path) -> None: run = ClaudeCliRun( command=("claude", "-p", "x"), @@ -143,6 +197,140 @@ def test_nvidia_nim_cli_raw_payload_log_counts_as_proxy_request( assert regression_failures([outcome]) == [] +def test_cli_matrix_missing_agent_catalog_is_harness_bug(tmp_path: Path) -> None: + run = ClaudeCliRun( + command=("claude", "-p", "x"), + returncode=0, + stdout="ordinary answer", + stderr="", + duration_s=0.1, + ) + outcome = make_outcome( + model="openai/gpt-oss-120b:free", + full_model="open_router/openai/gpt-oss-120b:free", + source="openrouter_free_cli_default", + feature="subagent_task", + marker="FCC_OPENROUTER_FREE_TASK", + run=run, + log_delta=( + "API_REQUEST: request_id=req_1 model=openai/gpt-oss-120b:free " + "messages=1\n" + "FULL_PAYLOAD [req_1]: {'messages': [], 'tools': [{'name': 'Read'}], " + "'tool_choice': None}" + ), + log_path=tmp_path / "server.log", + requires_agent=True, + ) + + assert outcome.classification == "harness_bug" + assert outcome.token_evidence["agent_catalog_present"] is False + + +def test_cli_matrix_agent_catalog_without_agent_use_is_model_feature_failure( + tmp_path: Path, +) -> None: + marker = "FCC_OPENROUTER_FREE_TASK" + run = ClaudeCliRun( + command=("claude", "-p", "x"), + returncode=0, + stdout=( + f'{marker}\n{{"type":"tool_use","name":"Read"}}\n{{"type":"tool_result"}}' + ), + stderr="", + duration_s=0.1, + ) + outcome = make_outcome( + model="openai/gpt-oss-120b:free", + full_model="open_router/openai/gpt-oss-120b:free", + source="openrouter_free_cli_default", + feature="subagent_task", + marker=marker, + run=run, + log_delta=( + "API_REQUEST: request_id=req_1 model=openai/gpt-oss-120b:free " + "messages=1\n" + "FULL_PAYLOAD [req_1]: {'messages': [], 'tools': " + "[{'name': 'Agent'}, {'name': 'Read'}], 'tool_choice': None}" + ), + log_path=tmp_path / "server.log", + requires_tool_result=True, + requires_agent=True, + ) + + assert outcome.classification == "model_feature_failure" + assert outcome.token_evidence["agent_catalog_present"] is True + assert outcome.token_evidence["agent_tool_count"] == 0 + + +def test_cli_matrix_agent_use_result_and_marker_pass(tmp_path: Path) -> None: + marker = "FCC_OPENROUTER_FREE_TASK" + run = ClaudeCliRun( + command=("claude", "-p", "x"), + returncode=0, + stdout=( + f'{marker}\n{{"type":"tool_use","name":"Agent"}}\n' + '{"type":"tool_result","content":"agentId: abc123"}' + ), + stderr="", + duration_s=0.1, + ) + outcome = make_outcome( + model="openai/gpt-oss-120b:free", + full_model="open_router/openai/gpt-oss-120b:free", + source="openrouter_free_cli_default", + feature="subagent_task", + marker=marker, + run=run, + log_delta=( + "API_REQUEST: request_id=req_1 model=openai/gpt-oss-120b:free " + "messages=1\n" + "FULL_PAYLOAD [req_1]: {'messages': [], 'tools': " + "[{'name': 'Agent'}, {'name': 'Read'}], 'tool_choice': None}" + ), + log_path=tmp_path / "server.log", + requires_tool_result=True, + requires_agent=True, + ) + + assert outcome.classification == "passed" + assert outcome.token_evidence["agent_catalog_present"] is True + assert outcome.token_evidence["agent_tool_count"] == 1 + assert outcome.token_evidence["agent_result_count"] == 1 + + +def test_cli_matrix_agent_prompt_text_without_tool_evidence_does_not_pass( + tmp_path: Path, +) -> None: + marker = "FCC_OPENROUTER_FREE_TASK" + run = ClaudeCliRun( + command=("claude", "-p", "x"), + returncode=0, + stdout=f"{marker}\nAgent should read the file.", + stderr="", + duration_s=0.1, + ) + outcome = make_outcome( + model="openai/gpt-oss-120b:free", + full_model="open_router/openai/gpt-oss-120b:free", + source="openrouter_free_cli_default", + feature="subagent_task", + marker=marker, + run=run, + log_delta=( + "API_REQUEST: request_id=req_1 model=openai/gpt-oss-120b:free " + "messages=1\n" + "FULL_PAYLOAD [req_1]: {'messages': [], 'tools': " + "[{'name': 'Agent'}, {'name': 'Read'}], 'tool_choice': None}" + ), + log_path=tmp_path / "server.log", + requires_agent=True, + ) + + assert outcome.classification == "model_feature_failure" + assert outcome.token_evidence["agent_catalog_present"] is True + assert outcome.token_evidence["agent_tool_count"] == 0 + + def test_nvidia_nim_cli_timeout_is_not_model_missing( tmp_path: Path, ) -> None: @@ -194,3 +382,89 @@ def test_nvidia_nim_cli_success_beats_verbose_timeout_words(tmp_path: Path) -> N assert outcome.classification == "passed" assert outcome.request_count == 1 + + +def test_cli_matrix_uuid_429_does_not_count_as_upstream_unavailable( + tmp_path: Path, +) -> None: + run = ClaudeCliRun( + command=("claude", "-p", "x"), + returncode=0, + stdout='{"uuid":"d3c76eea-3634-4299-aec0-e7634b3716da"}', + stderr="", + duration_s=0.1, + ) + outcome = make_outcome( + model="openai/gpt-oss-120b:free", + full_model="open_router/openai/gpt-oss-120b:free", + source="openrouter_free_cli_default", + feature="subagent_task", + marker="FCC_OPENROUTER_FREE_TASK", + run=run, + log_delta="API_REQUEST: request_id=req_1 model=openai/gpt-oss-120b:free messages=2", + log_path=tmp_path / "server.log", + requires_task=True, + ) + + assert outcome.classification == "model_feature_failure" + + +def test_cli_matrix_real_http_429_counts_as_upstream_unavailable( + tmp_path: Path, +) -> None: + run = ClaudeCliRun( + command=("claude", "-p", "x"), + returncode=0, + stdout="ordinary answer", + stderr="", + duration_s=0.1, + ) + outcome = make_outcome( + model="openai/gpt-oss-120b:free", + full_model="open_router/openai/gpt-oss-120b:free", + source="openrouter_free_cli_default", + feature="subagent_task", + marker="FCC_OPENROUTER_FREE_TASK", + run=run, + log_delta=( + "API_REQUEST: request_id=req_1 model=openai/gpt-oss-120b:free " + 'messages=2 upstream HTTP/1.1" 429 Too Many Requests' + ), + log_path=tmp_path / "server.log", + requires_task=True, + ) + + assert outcome.classification == "upstream_unavailable" + + +def test_cli_matrix_default_command_uses_bare_mode() -> None: + command = _build_claude_cli_command( + claude_bin="claude", + prompt="hello", + tools="Read", + ) + + assert command[:2] == ("claude", "--bare") + assert "--tools" in command + assert "Read" in command + + +def test_cli_matrix_subagent_command_uses_agent_without_bare_or_task() -> None: + bare, tools, pre_tool_args, extra_args = _subagent_probe_options("{}") + command = _build_claude_cli_command( + claude_bin="claude", + prompt="hello", + tools=tools, + bare=bare, + pre_tool_args=pre_tool_args, + extra_args=extra_args, + ) + + assert "--bare" not in command + assert command[command.index("--setting-sources") + 1] == "local" + assert "--strict-mcp-config" in command + assert command[command.index("--mcp-config") + 1] == '{"mcpServers":{}}' + assert command[command.index("--tools") + 1] == "Agent,Read" + assert command[command.index("--allowedTools") + 1] == "Agent,Read" + assert command[command.index("--agents") + 1] == "{}" + assert "Task,Read" not in command diff --git a/tests/contracts/test_smoke_config.py b/tests/contracts/test_smoke_config.py index a687ded3..3b368aa3 100644 --- a/tests/contracts/test_smoke_config.py +++ b/tests/contracts/test_smoke_config.py @@ -7,11 +7,13 @@ from smoke.lib.config import ( ALL_TARGETS, DEFAULT_TARGETS, NVIDIA_NIM_CLI_DEFAULT_MODELS, + OPENROUTER_FREE_CLI_DEFAULT_MODELS, OPT_IN_TARGETS, PROVIDER_SMOKE_DEFAULT_MODELS, TARGET_REQUIRED_ENV, SmokeConfig, nvidia_nim_cli_model_refs, + openrouter_free_cli_model_refs, ) @@ -61,6 +63,10 @@ def test_nvidia_nim_cli_is_opt_in_smoke_target() -> None: assert "nvidia_nim_cli" in OPT_IN_TARGETS assert "nvidia_nim_cli" in ALL_TARGETS assert "nvidia_nim_cli" in TARGET_REQUIRED_ENV + assert "openrouter_free_cli" not in DEFAULT_TARGETS + assert "openrouter_free_cli" in OPT_IN_TARGETS + assert "openrouter_free_cli" in ALL_TARGETS + assert "openrouter_free_cli" in TARGET_REQUIRED_ENV def test_ollama_provider_configuration_uses_base_url() -> None: @@ -265,3 +271,77 @@ def test_smoke_config_returns_nvidia_nim_cli_provider_models(monkeypatch) -> Non assert models[0].provider == "nvidia_nim" assert models[0].full_model == "nvidia_nim/z-ai/glm-5.1" assert models[0].source == "nvidia_nim_cli_default" + + +def test_openrouter_free_cli_default_models_are_normalized() -> None: + refs = openrouter_free_cli_model_refs({}) + + assert tuple(refs) == tuple( + f"open_router/{model}" for model in OPENROUTER_FREE_CLI_DEFAULT_MODELS + ) + assert "open_router/nvidia/nemotron-3-super-120b-a12b:free" in refs + assert "open_router/poolside/laguna-m.1:free" in refs + assert set(refs.values()) == {"openrouter_free_cli_default"} + + +def test_openrouter_free_cli_models_override_and_append() -> None: + refs = openrouter_free_cli_model_refs( + { + "FCC_SMOKE_OPENROUTER_FREE_MODELS": ( + "openai/gpt-oss-120b:free,open_router/custom/model:free" + ), + "FCC_SMOKE_OPENROUTER_FREE_EXTRA_MODELS": ( + "poolside/laguna-m.1:free,openai/gpt-oss-120b:free" + ), + } + ) + + assert tuple(refs) == ( + "open_router/openai/gpt-oss-120b:free", + "open_router/custom/model:free", + "open_router/poolside/laguna-m.1:free", + ) + assert refs["open_router/openai/gpt-oss-120b:free"] == ( + "FCC_SMOKE_OPENROUTER_FREE_MODELS" + ) + assert refs["open_router/poolside/laguna-m.1:free"] == ( + "FCC_SMOKE_OPENROUTER_FREE_EXTRA_MODELS" + ) + + +def test_openrouter_free_cli_models_reject_empty_override() -> None: + try: + openrouter_free_cli_model_refs({"FCC_SMOKE_OPENROUTER_FREE_MODELS": " , "}) + except ValueError as exc: + assert "FCC_SMOKE_OPENROUTER_FREE_MODELS" in str(exc) + else: + raise AssertionError("expected empty OpenRouter free CLI override to fail") + + +def test_openrouter_free_cli_models_reject_wrong_provider_prefix() -> None: + try: + openrouter_free_cli_model_refs( + {"FCC_SMOKE_OPENROUTER_FREE_MODELS": "nvidia_nim/model"} + ) + except ValueError as exc: + assert "open_router" in str(exc) + else: + raise AssertionError("expected wrong provider prefix to fail") + + +def test_smoke_config_returns_openrouter_free_cli_provider_models(monkeypatch) -> None: + monkeypatch.delenv("FCC_SMOKE_OPENROUTER_FREE_MODELS", raising=False) + monkeypatch.delenv("FCC_SMOKE_OPENROUTER_FREE_EXTRA_MODELS", raising=False) + config = _smoke_config( + settings=_settings( + model="open_router/openai/gpt-oss-120b:free", + open_router_api_key="openrouter-key", + ollama_base_url="", + ) + ) + + models = config.openrouter_free_cli_models() + + assert models[0].provider == "open_router" + assert models[0].full_model == "open_router/nvidia/nemotron-3-super-120b-a12b:free" + assert models[0].source == "openrouter_free_cli_default"