From 7c36c3d784e3b9083a413227b2b0da4a94e960bd Mon Sep 17 00:00:00 2001 From: Jim Schaff Date: Wed, 2 Sep 2026 11:54:38 -0400 Subject: [PATCH 1/4] feat(k8s): give litellm a federated GCP identity for Vertex AI Adds the plumbing for GCP Workload Identity Federation so the proxy can reach Vertex AI with no stored credential: a dedicated ServiceAccount, a projected 1-hour token scoped to the WIF provider's audience, and the external-account config describing the exchange. A downloaded service account key is not an option -- org policy iam.managed.disableServiceAccountKeyCreation blocks key creation on the CAIA project -- and we would rather not hold one anyway. Three things worth knowing about the shape of this: - The ServiceAccount is dedicated rather than the namespace default, because the pool's attribute condition names exactly one subject. If we federated `default`, any pod in the namespace could mint Vertex credentials. - gcp-wif.json is in the plain ConfigMap, not a sealed secret. Google is explicit that a credential configuration "doesn't contain a private key and doesn't need to be kept confidential" -- it only names the pool, the service account to impersonate, and where to read the token. - The projected volume's audience uses the https:// form while gcp-wif.json uses the schemeless // form for the same provider. That asymmetry is Google's, not a typo; both are default allowed audiences. The pool and provider IDs are placeholders until Burwood creates them (REQ0040169), so this is not yet deployable. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_014ZbiJ9tw2PWi8A4coNs5w5 --- kustomize/base/gcp-wif.json | 14 +++++++++++ kustomize/base/kustomization.yaml | 8 ++++++ kustomize/base/litellm.yaml | 42 +++++++++++++++++++++++++++++++ 3 files changed, 64 insertions(+) create mode 100644 kustomize/base/gcp-wif.json diff --git a/kustomize/base/gcp-wif.json b/kustomize/base/gcp-wif.json new file mode 100644 index 0000000..52e657b --- /dev/null +++ b/kustomize/base/gcp-wif.json @@ -0,0 +1,14 @@ +{ + "universe_domain": "googleapis.com", + "type": "external_account", + "audience": "//iam.googleapis.com/projects/637433805721/locations/global/workloadIdentityPools/POOL_ID/providers/PROVIDER_ID", + "subject_token_type": "urn:ietf:params:oauth:token-type:jwt", + "token_url": "https://sts.googleapis.com/v1/token", + "service_account_impersonation_url": "https://iamcredentials.googleapis.com/v1/projects/-/serviceAccounts/SA_EMAIL:generateAccessToken", + "credential_source": { + "file": "/var/run/secrets/gcp/token", + "format": { + "type": "text" + } + } +} diff --git a/kustomize/base/kustomization.yaml b/kustomize/base/kustomization.yaml index 8d4ee1c..9df8d20 100644 --- a/kustomize/base/kustomization.yaml +++ b/kustomize/base/kustomization.yaml @@ -10,7 +10,15 @@ resources: # litellm_config.yaml is env-agnostic, so it is generated here (mounted read-only # into the litellm pod at /app/config.yaml). +# +# gcp-wif.json rides along in the same ConfigMap and is mounted at +# /etc/gcp/wif.json. It is the Workload Identity Federation external-account +# config for Vertex AI. Google's docs are explicit that this file "doesn't +# contain a private key and doesn't need to be kept confidential" -- it only +# names the identity pool, the Google service account to impersonate, and the +# path of the projected token -- so it is deliberately NOT a sealed secret. configMapGenerator: - name: litellm-config-file files: - config.yaml=litellm_config.yaml + - gcp-wif.json diff --git a/kustomize/base/litellm.yaml b/kustomize/base/litellm.yaml index b0d7d69..f466a30 100644 --- a/kustomize/base/litellm.yaml +++ b/kustomize/base/litellm.yaml @@ -1,3 +1,15 @@ +# Dedicated identity for the proxy. This exists so GCP Workload Identity +# Federation can be scoped to *this* workload: the WIF provider's attribute +# condition pins assertion.sub to "system:serviceaccount::litellm", so the +# namespace default SA (which every other pod here uses) cannot assume the +# Google service account. See docs/gcp-wif-findings.md. +apiVersion: v1 +kind: ServiceAccount +metadata: + name: litellm + labels: + app: litellm +--- apiVersion: apps/v1 kind: Deployment metadata: @@ -14,6 +26,7 @@ spec: labels: app: litellm spec: + serviceAccountName: litellm containers: - name: litellm image: ghcr.io/berriai/litellm:main-stable @@ -74,6 +87,16 @@ spec: - name: litellm-config-file mountPath: /app/config.yaml subPath: config.yaml + # The external-account config that vertex_credentials points at. It + # holds no key material -- only the pool/provider path, the target + # Google service account, and where to read the token -- so it rides + # in the same ConfigMap as config.yaml rather than a sealed secret. + - name: litellm-config-file + mountPath: /etc/gcp/wif.json + subPath: gcp-wif.json + - name: gcp-token + mountPath: /var/run/secrets/gcp + readOnly: true resources: requests: memory: "512Mi" @@ -93,6 +116,25 @@ spec: - name: litellm-config-file configMap: name: litellm-config-file + # Short-lived, audience-scoped ServiceAccount token -- the credential we + # trade at Google's STS for a Vertex AI access token. kubelet rotates the + # file at ~80% of its lifetime and google-auth re-reads it on refresh, so + # nothing here expires. This replaces a downloadable service account key, + # which org policy iam.managed.disableServiceAccountKeyCreation forbids. + - name: gcp-token + projected: + sources: + - serviceAccountToken: + # REPLACE_ME: the WIF provider's full resource audience, once + # Burwood creates the pool (REQ0040169). + # + # The leading "https://" is deliberate and is NOT a typo to be + # tidied up: Google's manifest uses the https:// form here, while + # gcp-wif.json uses the schemeless "//iam.googleapis.com/..." form + # for the same provider. Both are default allowed audiences. + audience: https://iam.googleapis.com/projects/637433805721/locations/global/workloadIdentityPools/POOL_ID/providers/PROVIDER_ID + expirationSeconds: 3600 + path: token imagePullSecrets: - name: ghcr-secret From d8997e760e4f6ea100525bfbf4e41f56139d4fa2 Mon Sep 17 00:00:00 2001 From: Jim Schaff Date: Wed, 2 Sep 2026 11:54:47 -0400 Subject: [PATCH 2/4] feat(litellm): add the gemini-model alias, backed by Vertex AI Third model option alongside openai-model and local-model, serving gemini-2.5-flash from us-central1 on the CAIA grant project. The alias carries no api_key. LiteLLM calls google.auth.default(), which follows GOOGLE_APPLICATION_CREDENTIALS to the external-account config and from there to the projected ServiceAccount token. Using the env var rather than litellm_params.vertex_credentials is deliberate. Both work, but leaving the variable unset makes google.auth.default() fall back to the developer's own gcloud ADC, so the same alias works locally with no config change -- which is how it was first prototyped. Pointing vertex_credentials at a path that does not exist locally would instead fail outright, because LiteLLM treats a non-existent path as inline JSON and tries to parse it. Hence vcell-ai-local sets the Vertex project and location but not GOOGLE_APPLICATION_CREDENTIALS: there is no projected token outside the cluster to point it at. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_014ZbiJ9tw2PWi8A4coNs5w5 --- kustomize/base/litellm_config.yaml | 16 ++++++++++++++++ kustomize/config/vcell-ai-local/litellm.env | 11 +++++++++++ kustomize/config/vcell-ai-rke-dev/litellm.env | 10 ++++++++++ kustomize/config/vcell-ai-rke/litellm.env | 10 ++++++++++ litellm.env.example | 12 ++++++++++++ litellm_config.yaml | 16 ++++++++++++++++ 6 files changed, 75 insertions(+) diff --git a/kustomize/base/litellm_config.yaml b/kustomize/base/litellm_config.yaml index e26c814..ebcd996 100644 --- a/kustomize/base/litellm_config.yaml +++ b/kustomize/base/litellm_config.yaml @@ -23,6 +23,22 @@ model_list: model: os.environ/LOCAL_LLM_MODEL api_base: os.environ/LOCAL_LLM_API_BASE + # "gemini-model" -> Gemini on Vertex AI, in the CAIA grant project. There is no + # api_key here on purpose: auth is Workload Identity Federation. LiteLLM calls + # google.auth.default(), which reads GOOGLE_APPLICATION_CREDENTIALS -> the + # external-account config at /etc/gcp/wif.json -> the projected ServiceAccount + # token at /var/run/secrets/gcp/token, and trades that at Google's STS. Nothing + # long-lived is stored anywhere. See docs/gcp-wif-findings.md. + # + # Leaving GOOGLE_APPLICATION_CREDENTIALS unset (as vcell-ai-local does) makes + # google.auth.default() fall back to the developer's own gcloud ADC, so this + # same alias works locally after `gcloud auth application-default login`. + - model_name: gemini-model + litellm_params: + model: os.environ/GEMINI_MODEL + vertex_project: os.environ/VERTEX_PROJECT + vertex_location: os.environ/VERTEX_LOCATION + litellm_settings: success_callback: ["langfuse"] fallbacks: diff --git a/kustomize/config/vcell-ai-local/litellm.env b/kustomize/config/vcell-ai-local/litellm.env index 1d511aa..75cc493 100644 --- a/kustomize/config/vcell-ai-local/litellm.env +++ b/kustomize/config/vcell-ai-local/litellm.env @@ -12,6 +12,17 @@ OPENAI_MODEL=openai/gpt-4o-mini LOCAL_LLM_MODEL=ollama/phi4-mini LOCAL_LLM_API_BASE=http://ollama:11434 +# gemini-model alias -> Gemini on Vertex AI (CAIA grant project). +GEMINI_MODEL=vertex_ai/gemini-2.5-flash +VERTEX_PROJECT=caia-fac-uconn-2-225-2026 +VERTEX_LOCATION=us-central1 +# GOOGLE_APPLICATION_CREDENTIALS is deliberately NOT set here. Workload Identity +# Federation needs a projected ServiceAccount token that only exists in the +# cluster, so locally we let google.auth.default() fall back to your own ADC: +# gcloud auth application-default login +# gcloud config set project caia-fac-uconn-2-225-2026 +# You need roles/aiplatform.user on the project for this to work. + # Langfuse tracing callback (host only; keys are secret) LANGFUSE_HOST=https://cloud.langfuse.com diff --git a/kustomize/config/vcell-ai-rke-dev/litellm.env b/kustomize/config/vcell-ai-rke-dev/litellm.env index 8919a2f..c397742 100644 --- a/kustomize/config/vcell-ai-rke-dev/litellm.env +++ b/kustomize/config/vcell-ai-rke-dev/litellm.env @@ -12,6 +12,16 @@ OPENAI_MODEL=openai/gpt-4o-mini LOCAL_LLM_MODEL=ollama/phi4-mini LOCAL_LLM_API_BASE=http://ollama:11434 +# gemini-model alias -> Gemini on Vertex AI (CAIA grant project). No API key: +# auth is Workload Identity Federation, so the only "credential" in the pod is a +# 1-hour projected ServiceAccount token that kubelet keeps rotating. +GEMINI_MODEL=vertex_ai/gemini-2.5-flash +VERTEX_PROJECT=caia-fac-uconn-2-225-2026 +VERTEX_LOCATION=us-central1 +# Points at the external-account config in the litellm-config-file ConfigMap. +# google.auth.default() reads this; it holds no key material. +GOOGLE_APPLICATION_CREDENTIALS=/etc/gcp/wif.json + # Langfuse tracing callback (host only; keys are secret) LANGFUSE_HOST=https://cloud.langfuse.com diff --git a/kustomize/config/vcell-ai-rke/litellm.env b/kustomize/config/vcell-ai-rke/litellm.env index ed878ef..2bc6443 100644 --- a/kustomize/config/vcell-ai-rke/litellm.env +++ b/kustomize/config/vcell-ai-rke/litellm.env @@ -12,6 +12,16 @@ OPENAI_MODEL=openai/gpt-4o-mini LOCAL_LLM_MODEL=ollama/phi4-mini LOCAL_LLM_API_BASE=http://ollama:11434 +# gemini-model alias -> Gemini on Vertex AI (CAIA grant project). No API key: +# auth is Workload Identity Federation, so the only "credential" in the pod is a +# 1-hour projected ServiceAccount token that kubelet keeps rotating. +GEMINI_MODEL=vertex_ai/gemini-2.5-flash +VERTEX_PROJECT=caia-fac-uconn-2-225-2026 +VERTEX_LOCATION=us-central1 +# Points at the external-account config in the litellm-config-file ConfigMap. +# google.auth.default() reads this; it holds no key material. +GOOGLE_APPLICATION_CREDENTIALS=/etc/gcp/wif.json + # Langfuse tracing callback (host only; keys are secret) LANGFUSE_HOST=https://cloud.langfuse.com diff --git a/litellm.env.example b/litellm.env.example index db0c9c2..0847bde 100644 --- a/litellm.env.example +++ b/litellm.env.example @@ -14,6 +14,18 @@ AZURE_API_VERSION=2024-02-01 OPENAI_API_KEY=sk-your-openai-api-key OPENAI_MODEL=openai/gpt-4o-mini +# gemini-model alias -> Gemini on Vertex AI (project caia-fac-uconn-2-225-2026). +# There is no API key: LiteLLM calls google.auth.default(), so for local compose +# use your own Application Default Credentials -- +# gcloud auth application-default login +# gcloud config set project caia-fac-uconn-2-225-2026 +# and make sure docker-compose passes ~/.config/gcloud through to the container. +# In the cluster this same alias authenticates by Workload Identity Federation +# instead; see docs/gcp-wif-findings.md. +GEMINI_MODEL=vertex_ai/gemini-2.5-flash +VERTEX_PROJECT=caia-fac-uconn-2-225-2026 +VERTEX_LOCATION=us-central1 + # Local provider for the "local-model" LiteLLM alias. # For Ollama, keep the model prefixed with "ollama/". LOCAL_LLM_MODEL=ollama/phi4-mini diff --git a/litellm_config.yaml b/litellm_config.yaml index e26c814..ebcd996 100644 --- a/litellm_config.yaml +++ b/litellm_config.yaml @@ -23,6 +23,22 @@ model_list: model: os.environ/LOCAL_LLM_MODEL api_base: os.environ/LOCAL_LLM_API_BASE + # "gemini-model" -> Gemini on Vertex AI, in the CAIA grant project. There is no + # api_key here on purpose: auth is Workload Identity Federation. LiteLLM calls + # google.auth.default(), which reads GOOGLE_APPLICATION_CREDENTIALS -> the + # external-account config at /etc/gcp/wif.json -> the projected ServiceAccount + # token at /var/run/secrets/gcp/token, and trades that at Google's STS. Nothing + # long-lived is stored anywhere. See docs/gcp-wif-findings.md. + # + # Leaving GOOGLE_APPLICATION_CREDENTIALS unset (as vcell-ai-local does) makes + # google.auth.default() fall back to the developer's own gcloud ADC, so this + # same alias works locally after `gcloud auth application-default login`. + - model_name: gemini-model + litellm_params: + model: os.environ/GEMINI_MODEL + vertex_project: os.environ/VERTEX_PROJECT + vertex_location: os.environ/VERTEX_LOCATION + litellm_settings: success_callback: ["langfuse"] fallbacks: From 815335283d35664a327b4a2c1e76b83f731c8c43 Mon Sep 17 00:00:00 2001 From: Jim Schaff Date: Wed, 2 Sep 2026 11:54:59 -0400 Subject: [PATCH 3/4] feat(app): offer Gemini in the model picker The alias name is a three-way contract -- the LiteLLM config, the backend LLMModel Literal, and the frontend ModelId union all have to agree, or the dropdown offers a model the API rejects. The budget-exceeded fallback in llms_service.py needs no change: it only special-cases local-model, and treats everything else uniformly. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_014ZbiJ9tw2PWi8A4coNs5w5 --- backend/app/schemas/llms_schema.py | 2 +- frontend/components/ChatBox.tsx | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/backend/app/schemas/llms_schema.py b/backend/app/schemas/llms_schema.py index 49f8492..772e1a7 100644 --- a/backend/app/schemas/llms_schema.py +++ b/backend/app/schemas/llms_schema.py @@ -2,7 +2,7 @@ from pydantic import BaseModel, Field -LLMModel = Literal["openai-model", "local-model"] +LLMModel = Literal["openai-model", "gemini-model", "local-model"] class ChatRequest(BaseModel): diff --git a/frontend/components/ChatBox.tsx b/frontend/components/ChatBox.tsx index 9a4c832..41b8a72 100644 --- a/frontend/components/ChatBox.tsx +++ b/frontend/components/ChatBox.tsx @@ -17,7 +17,7 @@ import { LoginRequiredDialog } from "@/components/login-required-dialog"; import { useChatHistory } from "@/hooks/use-chat-history"; import type { ConversationSurface, StoredMessage } from "@/lib/chat-history"; -type ModelId = "openai-model" | "local-model"; +type ModelId = "openai-model" | "gemini-model" | "local-model"; export interface Message { id: string; @@ -443,6 +443,7 @@ export const ChatBox: React.FC = ({ OpenAI + Gemini Local LLM From c4a689c31ebde9a6d6558ca9fe4ddb606c386b31 Mon Sep 17 00:00:00 2001 From: Jim Schaff Date: Wed, 2 Sep 2026 11:54:59 -0400 Subject: [PATCH 4/4] docs: record how Gemini authenticates, and what was measured Gemini is the first model here without an API key, and the reason is not obvious from the manifests alone. This writes down the mechanism, the measurements taken before committing to it, and the one ongoing maintenance item. The non-obvious constraint: our API server is VPN-only and the issuer (https://kubernetes.default.svc.cluster.local) does not resolve publicly, so Google cannot fetch our JWKS. The provider has to be created with the keys uploaded inline via --jwk-json-path. Google documents this for self-hosted clusters and states the cluster "doesn't need to be accessible over the internet." The consequence is that the uploaded JWKS is a static snapshot: if the cluster's service-account signing key is ever rotated, every Gemini call starts failing until someone re-uploads it. That is a silent failure mode with no alerting, so it gets its own section and an error-to-cause table. Also notes in the kustomize README why Gemini is deliberately absent from the sealed-secret inventory. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_014ZbiJ9tw2PWi8A4coNs5w5 --- docs/gcp-wif-findings.md | 242 +++++++++++++++++++++++++++++++++++++++ kustomize/README.md | 10 ++ 2 files changed, 252 insertions(+) create mode 100644 docs/gcp-wif-findings.md diff --git a/docs/gcp-wif-findings.md b/docs/gcp-wif-findings.md new file mode 100644 index 0000000..5e56908 --- /dev/null +++ b/docs/gcp-wif-findings.md @@ -0,0 +1,242 @@ +# Gemini on Vertex AI: how the LiteLLM proxy authenticates to GCP + +VCell-AI offers Gemini through the same LiteLLM proxy that serves OpenAI and the local +Ollama model. Unlike those two, Gemini has **no API key anywhere** — not in a sealed secret, +not in an env file, not in the pod. It authenticates with GCP **Workload Identity +Federation** (WIF), trading a short-lived Kubernetes ServiceAccount token for a Vertex AI +access token on every refresh. + +This document records why, and what was measured before committing to the design. It is +written to be readable by someone who has not seen the ticket (Burwood REQ0040169). + +## Why not a service account key + +The obvious approach — download a GCP service account JSON key, seal it, mount it — is +blocked by org policy `iam.managed.disableServiceAccountKeyCreation` on the CAIA project +`caia-fac-uconn-2-225-2026`. Burwood offered to exclude our project from that policy. We +declined, because the policy is right: a downloaded key is a non-expiring, portable +credential that would live in git (sealed, but still), that we would own the rotation of, +and that would make this project the org's standing exception. WIF costs one extra `gcloud` +invocation on their side and about fifteen lines of YAML on ours. + +Google's own guidance agrees — service account keys are "persistent, portable credentials" +whose alternatives "provide the same access without the risks." + +## How it works + +``` + litellm pod + │ ServiceAccount: litellm + │ + ├─ projected volume /var/run/secrets/gcp/token + │ a JWT signed by the cluster, aud = the WIF provider, sub = + │ system:serviceaccount:vcell-ai-rke-dev:litellm, 1-hour lifetime, + │ rotated in place by kubelet at ~80% of that + │ + ├─ ConfigMap file /etc/gcp/wif.json (GOOGLE_APPLICATION_CREDENTIALS) + │ external-account config: which pool/provider to present the token to, + │ which Google service account to impersonate, where the token file is. + │ Holds no key material. + │ + └─ google.auth.default() + → POST sts.googleapis.com (k8s token → federated token) + → POST iamcredentials.googleapis.com (impersonate the Google SA) + → Bearer token for aiplatform.googleapis.com +``` + +Nothing long-lived is stored. The only durable trust is on the Google side: the identity +pool holds a copy of our cluster's **public** signing key and a condition naming exactly one +Kubernetes ServiceAccount. + +## What was measured + +All of the following was verified against the live `vcell-ai-rke-dev` namespace and the +running `litellm` pod before the design was settled, rather than assumed from documentation. + +| Fact | Result | +|---|---| +| Cluster | RKE2 `v1.31.13+rke2r1`, on-prem VxRail, 7 nodes | +| SA token issuer | `https://kubernetes.default.svc.cluster.local` (`--service-account-issuer` on `kube-apiserver-k8s-cp-01`, and `/.well-known/openid-configuration`) | +| JWKS | a single RSA/RS256 key, kid `_sEcsu_IaVE_5SVJKSmCtYi3oj6lhCvQMorJjUmimXI` | +| Custom-audience token minting | **works.** A token minted with a GCP audience carries `iss` as above and `sub = system:serviceaccount:vcell-ai-rke-dev:` — exactly the claims the attribute mapping needs | +| `--api-audiences` | set to `https://kubernetes.default.svc.cluster.local,rke2`, and it does **not** restrict projected-token audiences (it governs tokens presented *to* the API server) | +| Pod egress to Google | `sts.googleapis.com`, `iamcredentials.googleapis.com`, `us-central1-aiplatform.googleapis.com` all reachable from the litellm pod — TLS completes and Google answers. No egress firewall or proxy in the way | +| LiteLLM | 1.92.0, with `google-auth` 2.52.0 and `google-cloud-aiplatform` 1.133.0 | +| LiteLLM WIF support | first-class. `vertex_llm_base.py` branches on `type == "external_account"` and calls `google.auth.identity_pool.Credentials.from_info()`; a `credential_source.file` config (our case) hits that branch | +| GCP project number | `637433805721` | +| **End-to-end rehearsal** | the real rendered `gcp-wif.json` plus a real cluster token was fed to `google.auth` inside the litellm pod: it parsed as an identity-pool credential, wired up impersonation, read the projected token, and **reached Google's STS**, which rejected only the `POOL_ID` placeholder. Every link except the pool itself is therefore proven | + +## The one non-obvious part: the issuer is not on the internet + +Our API server is `https://155.37.250.221:6443` and is VPN-only. The issuer string +`https://kubernetes.default.svc.cluster.local` does not resolve publicly, and anonymous JWKS +discovery is not enabled on this cluster (the `system:service-account-issuer-discovery` +ClusterRoleBinding is bound to `system:serviceaccounts`, not `system:unauthenticated`). + +So **Google cannot fetch our discovery document or JWKS**, and a provider created the normal +way — pointing at an issuer URL and letting Google go and read the keys — would fail. + +The supported answer is to upload the JWKS inline with `--jwk-json-path` when creating the +provider. Google documents this precisely for self-hosted clusters and states plainly: +*"The cluster doesn't need to be accessible over the internet."* See +[Configure Workload Identity Federation with Kubernetes](https://docs.cloud.google.com/iam/docs/workload-identity-federation-with-kubernetes). + +This is the single correction we sent back to Burwood — their first reply asked for the +issuer URL and JWKS, which was right, but the provider has to be told to *trust* the uploaded +keys rather than to go and get them. + +## Operational consequence: JWKS rotation + +The uploaded JWKS is a **static snapshot**, with a maximum of 8 keys. If the cluster's +service-account signing key ever changes — RKE2 key rotation, a control-plane rebuild, a +cluster re-creation — federation breaks silently and every Gemini call starts returning 403 +until someone re-uploads: + +```bash +kubectl get --raw /openid/v1/jwks > cluster-jwks.json +gcloud iam workload-identity-pools providers update-oidc PROVIDER_ID \ + --location=global --workload-identity-pool=POOL_ID \ + --jwk-json-path=cluster-jwks.json +``` + +Note `update-oidc` **replaces** the uploaded keys; the previous set cannot be restored. If +Gemini starts failing with auth errors and nothing in this repo changed, check this first. + +## Design choices worth knowing + +**Impersonation, not direct resource access.** The credential config includes +`service_account_impersonation_url`, so the federated identity impersonates a Google service +account that holds `roles/aiplatform.user`. Direct resource access (binding the role straight +to the federated principal) also appears supported for Vertex AI, but it is the less +battle-tested path, it leaves no SA-based quota/billing project, and it is the configuration +where LiteLLM's since-fixed missing-scopes bug ([#17377](https://github.com/BerriAI/litellm/issues/17377), +fixed in v1.80.8) used to bite. The IAM difference is one binding. + +**A dedicated ServiceAccount, not `default`.** Every other pod in this namespace runs as the +namespace `default` ServiceAccount. Federating that would mean *any* pod here could obtain +Vertex credentials. The `litellm` ServiceAccount exists solely so the pool's attribute +condition can name one workload. + +**`GOOGLE_APPLICATION_CREDENTIALS`, not `vertex_credentials`.** Both work — LiteLLM's +`vertex_credentials` accepts a path — but the env var route means `google.auth.default()` +handles it, which gives a useful property: with the variable unset, it falls back to a +developer's own `gcloud auth application-default login`. The same `gemini-model` alias +therefore works locally with no config changes. Pointing `vertex_credentials` at a path that +does not exist locally would instead fail, because LiteLLM treats a non-existent path as +inline JSON and tries to parse it. + +**The credential config is not a secret.** `gcp-wif.json` lives in the plain +`litellm-config-file` ConfigMap next to `config.yaml`, not in the sealed `litellm-secrets`. +Google is explicit that it "doesn't contain a private key and doesn't need to be kept +confidential." This is the practical payoff of WIF: adding Gemini required no `kubeseal` +round-trip and no change to `secrets.dat` or `sealed_secret_litellm.sh`. + +**The two audience strings differ by scheme, on purpose.** The projected volume's `audience:` +uses `https://iam.googleapis.com/...` (Google's manifest form); `gcp-wif.json`'s `audience` +field uses the schemeless `//iam.googleapis.com/...` (what `create-cred-config` generates). +Both are default allowed audiences for the same provider. Do not "fix" one to match the other. + +## GCP-side configuration + +Run by Burwood in project `caia-fac-uconn-2-225-2026` (number `637433805721`). + +```bash +POOL=vcell-ai-pool +PROVIDER=rke2-vcell-ai +PROJECT=caia-fac-uconn-2-225-2026 +PROJECT_NUMBER=637433805721 +KSA=system:serviceaccount:vcell-ai-rke-dev:litellm + +gcloud iam workload-identity-pools create $POOL \ + --project=$PROJECT --location=global \ + --display-name="VCell-AI on-prem RKE2" + +# --jwk-json-path is what makes this work without a publicly reachable issuer. +gcloud iam workload-identity-pools providers create-oidc $PROVIDER \ + --project=$PROJECT --location=global --workload-identity-pool=$POOL \ + --issuer-uri="https://kubernetes.default.svc.cluster.local" \ + --jwk-json-path=cluster-jwks.json \ + --attribute-mapping="google.subject=assertion.sub" \ + --attribute-condition="assertion.sub == '${KSA}'" + +gcloud iam service-accounts create vcell-ai-litellm --project=$PROJECT + +gcloud projects add-iam-policy-binding $PROJECT \ + --member="serviceAccount:vcell-ai-litellm@${PROJECT}.iam.gserviceaccount.com" \ + --role="roles/aiplatform.user" + +# The project NUMBER is required in this member string; the project ID is not supported. +gcloud iam service-accounts add-iam-policy-binding \ + vcell-ai-litellm@${PROJECT}.iam.gserviceaccount.com --project=$PROJECT \ + --member="principal://iam.googleapis.com/projects/${PROJECT_NUMBER}/locations/global/workloadIdentityPools/${POOL}/subject/${KSA}" \ + --role="roles/iam.workloadIdentityUser" + +gcloud services enable aiplatform.googleapis.com sts.googleapis.com \ + iamcredentials.googleapis.com --project=$PROJECT +``` + +The `attribute-condition` is the load-bearing security control: without it the pool trusts +every ServiceAccount in the cluster. + +## Verifying a change + +In order — each step isolates one failure domain, so the first one that fails tells you where +the problem is. + +**1. Token shape.** Confirms the cluster mints what the pool expects. + +```bash +AUD="https://iam.googleapis.com/projects/637433805721/locations/global/workloadIdentityPools/POOL_ID/providers/PROVIDER_ID" +kubectl -n vcell-ai-rke-dev create token litellm --audience="$AUD" \ + | cut -d. -f2 | base64 -d 2>/dev/null | jq '{iss, aud, sub}' +``` + +Expect `sub` to be `system:serviceaccount:vcell-ai-rke-dev:litellm`. + +**2. The STS exchange, before involving LiteLLM.** An error here means the GCP-side attribute +condition or an IAM binding is wrong; success means federation is done and anything that +fails later is our config. + +```bash +kubectl -n vcell-ai-rke-dev exec deploy/litellm -- python -c " +import google.auth, google.auth.transport.requests +c,_ = google.auth.default(scopes=['https://www.googleapis.com/auth/cloud-platform']) +c.refresh(google.auth.transport.requests.Request()) +print('token acquired, expires', c.expiry)" +``` + +Note that `google.auth.default()` and `load_credentials_from_file()` resolve the project ID +eagerly, which performs a token exchange *at load time*. A misconfigured pool therefore +raises inside the `default()` call, before you ever reach `refresh()` — so wrap the whole +snippet, not just the refresh, when you adapt it. + +Reading the error matters: + +| STS says | Means | +|---|---| +| `Invalid value for "audience"` | the audience string is malformed or names a pool/provider that does not exist | +| `Unable to parse the provided JWT` / issuer errors | the uploaded JWKS no longer matches the cluster's signing key — re-upload it (see above) | +| `The given credential is rejected by the attribute condition` | the condition does not match `system:serviceaccount:vcell-ai-rke-dev:litellm` | +| a 403 on `generateAccessToken` | the `roles/iam.workloadIdentityUser` binding on the Google service account is missing or names the project ID instead of the project number | + +**3. The LiteLLM alias**, from inside the cluster: + +```bash +kubectl -n vcell-ai-rke-dev exec deploy/litellm -- \ + curl -sS localhost:4000/v1/chat/completions \ + -H "Authorization: Bearer $LITELLM_MASTER_KEY" -H 'Content-Type: application/json' \ + -d '{"model":"gemini-model","messages":[{"role":"user","content":"say ok"}]}' +``` + +**4. End to end** — pick Gemini in the dropdown at https://vcell-ai-dev.cam.uchc.edu and +confirm the call shows up in Langfuse. + +**5. Token refresh.** The one failure mode a smoke test will not catch: confirm a Gemini call +still succeeds more than an hour after the pod started, which exercises kubelet's in-place +rotation of the projected token and google-auth re-reading the file. + +## Related + +- Ticket: Burwood REQ0040169 +- Grant project `caia-fac-uconn-2-225-2026` has `award_end=2026-12-31`; this setup should be + cheap to tear down or move. diff --git a/kustomize/README.md b/kustomize/README.md index 95f0450..fe8cc85 100644 --- a/kustomize/README.md +++ b/kustomize/README.md @@ -70,6 +70,16 @@ sealed secret — the backend needs the master key to provision virtual keys, an LiteLLM needs it plus Langfuse for tracing. Provide each value once in `secrets.dat`; `secrets.sh` seals it into the right secrets.) +Gemini is deliberately **absent** from that list. It authenticates to Vertex AI +with GCP Workload Identity Federation, so there is no key to seal: the pod +presents a 1-hour projected ServiceAccount token, and the external-account config +that describes the exchange (`base/gcp-wif.json`, mounted at `/etc/gcp/wif.json`) +carries no key material and rides in the plain `litellm-config-file` ConfigMap. +Adding a Gemini model therefore needs no `kubeseal` round-trip and no change to +`secrets.dat`. See [`docs/gcp-wif-findings.md`](../docs/gcp-wif-findings.md) — in +particular the note that the uploaded JWKS must be refreshed if the cluster's +service-account signing key is ever rotated. + `secrets.dat` (plaintext) and the generated `secret-*.yaml` are **gitignored**. Run `secrets.sh` to (re)generate the sealed manifests before applying an overlay.