diff --git a/.env.example b/.env.example index fa458d74..07ccf0a3 100644 --- a/.env.example +++ b/.env.example @@ -32,26 +32,4 @@ FAST_PREFIX_DETECTION=true ENABLE_NETWORK_PROBE_MOCK=true ENABLE_TITLE_GENERATION_SKIP=true ENABLE_SUGGESTION_MODE_SKIP=true -ENABLE_FILEPATH_EXTRACTION_MOCK=true - - -# All NVIDIA_NIM_* settings are strictly validated (unknown keys will error). -NVIDIA_NIM_TEMPERATURE=1.0 -NVIDIA_NIM_TOP_P=1.0 -NVIDIA_NIM_TOP_K=-1 -NVIDIA_NIM_MAX_TOKENS=81920 -NVIDIA_NIM_PRESENCE_PENALTY=0.0 -NVIDIA_NIM_FREQUENCY_PENALTY=0.0 -NVIDIA_NIM_MIN_P=0.0 -NVIDIA_NIM_REPETITION_PENALTY=1.0 -NVIDIA_NIM_SEED= -NVIDIA_NIM_STOP= -NVIDIA_NIM_PARALLEL_TOOL_CALLS=true -NVIDIA_NIM_RETURN_TOKENS_AS_TOKEN_IDS=false -NVIDIA_NIM_INCLUDE_STOP_STR_IN_OUTPUT=false -NVIDIA_NIM_IGNORE_EOS=false -NVIDIA_NIM_MIN_TOKENS=0 -NVIDIA_NIM_CHAT_TEMPLATE="" -NVIDIA_NIM_REQUEST_ID="" -NVIDIA_NIM_REASONING_EFFORT=high -NVIDIA_NIM_INCLUDE_REASONING=true \ No newline at end of file +ENABLE_FILEPATH_EXTRACTION_MOCK=true \ No newline at end of file diff --git a/README.md b/README.md index 1783077c..6dfc04b1 100644 --- a/README.md +++ b/README.md @@ -219,32 +219,6 @@ Browse: [openrouter.ai/models](https://openrouter.ai/models) - **NVIDIA NIM** base URL: `https://integrate.api.nvidia.com/v1` - **OpenRouter** base URL: `https://openrouter.ai/api/v1` -**NIM Settings (prefix `NVIDIA_NIM_`)** - -| Variable | Description | Default | -| --------------------------------------- | ----------------------------- | ------- | -| `NVIDIA_NIM_TEMPERATURE` | Sampling temperature | `1.0` | -| `NVIDIA_NIM_TOP_P` | Top-p nucleus sampling | `1.0` | -| `NVIDIA_NIM_TOP_K` | Top-k sampling | `-1` | -| `NVIDIA_NIM_MAX_TOKENS` | Max tokens for generation | `81920` | -| `NVIDIA_NIM_PRESENCE_PENALTY` | Presence penalty | `0.0` | -| `NVIDIA_NIM_FREQUENCY_PENALTY` | Frequency penalty | `0.0` | -| `NVIDIA_NIM_MIN_P` | Min-p sampling | `0.0` | -| `NVIDIA_NIM_REPETITION_PENALTY` | Repetition penalty | `1.0` | -| `NVIDIA_NIM_SEED` | RNG seed (blank = unset) | unset | -| `NVIDIA_NIM_STOP` | Stop string (blank = unset) | unset | -| `NVIDIA_NIM_PARALLEL_TOOL_CALLS` | Parallel tool calls | `true` | -| `NVIDIA_NIM_RETURN_TOKENS_AS_TOKEN_IDS` | Return token ids | `false` | -| `NVIDIA_NIM_INCLUDE_STOP_STR_IN_OUTPUT` | Include stop string in output | `false` | -| `NVIDIA_NIM_IGNORE_EOS` | Ignore EOS token | `false` | -| `NVIDIA_NIM_MIN_TOKENS` | Minimum generated tokens | `0` | -| `NVIDIA_NIM_CHAT_TEMPLATE` | Chat template override | unset | -| `NVIDIA_NIM_REQUEST_ID` | Request id override | unset | -| `NVIDIA_NIM_REASONING_EFFORT` | Reasoning effort | `high` | -| `NVIDIA_NIM_INCLUDE_REASONING` | Include reasoning in response | `true` | - -All `NVIDIA_NIM_*` settings are strictly validated; unknown keys with this prefix will cause startup errors. - See [`.env.example`](.env.example) for all supported parameters. ## Development diff --git a/config/nim.py b/config/nim.py index a95af482..83dd3cd2 100644 --- a/config/nim.py +++ b/config/nim.py @@ -1,13 +1,12 @@ -"""NVIDIA NIM settings (strict validation).""" +"""NVIDIA NIM settings (fixed values, no env config).""" -from typing import Optional, Literal +from typing import Literal, Optional -from pydantic import Field, field_validator -from pydantic_settings import BaseSettings, SettingsConfigDict +from pydantic import BaseModel, ConfigDict, Field, field_validator -class NimSettings(BaseSettings): - """Strictly validated NVIDIA NIM settings.""" +class NimSettings(BaseModel): + """Fixed NVIDIA NIM settings (not configurable via env).""" temperature: float = Field(1.0, ge=0.0) top_p: float = Field(1.0, ge=0.0, le=1.0) @@ -34,12 +33,7 @@ class NimSettings(BaseSettings): reasoning_effort: Literal["low", "medium", "high"] = "high" include_reasoning: bool = True - model_config = SettingsConfigDict( - env_prefix="NVIDIA_NIM_", - # Rely on global load_dotenv in config.settings to avoid - # reading unrelated .env keys into this settings model. - extra="forbid", - ) + model_config = ConfigDict(extra="forbid") @field_validator("top_k") @classmethod