mirror of
https://github.com/anthropics/claude-code.git
synced 2026-08-18 13:33:46 +00:00
Compare commits
32 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
354757e5b2 | ||
|
|
ae58f7a0a2 | ||
|
|
0fa8c19d50 | ||
|
|
1f6015b5d5 | ||
|
|
be90077c6a | ||
|
|
9923819368 | ||
|
|
681a8be245 | ||
|
|
54cc51a08a | ||
|
|
2bb6069614 | ||
|
|
53f9910f6e | ||
|
|
66edf53583 | ||
|
|
5cf69b18c8 | ||
|
|
3b272769d0 | ||
|
|
dd79613923 | ||
|
|
7ef6eec9d9 | ||
|
|
0c188278cd | ||
|
|
2982f95155 | ||
|
|
ac062f33ab | ||
|
|
c4dbd740a7 | ||
|
|
843297f6b1 | ||
|
|
b799fcaf9f | ||
|
|
4d07874235 | ||
|
|
015170d3fd | ||
|
|
07dcb0e135 | ||
|
|
67f390c9a0 | ||
|
|
c39cb0f14b | ||
|
|
b7784f2c63 | ||
|
|
c9181ca6eb | ||
|
|
988b3e5643 | ||
|
|
1fb278b85d | ||
|
|
d4d8fbbb33 | ||
|
|
d945a61bc6 |
@@ -68,7 +68,6 @@ for domain in \
|
||||
"registry.npmjs.org" \
|
||||
"api.anthropic.com" \
|
||||
"sentry.io" \
|
||||
"statsig.anthropic.com" \
|
||||
"statsig.com" \
|
||||
"marketplace.visualstudio.com" \
|
||||
"vscode.blob.core.windows.net" \
|
||||
|
||||
740
CHANGELOG.md
740
CHANGELOG.md
@@ -1,5 +1,742 @@
|
||||
# Changelog
|
||||
|
||||
## 2.1.234
|
||||
|
||||
- Added the optional `CLAUDE_CODE_PROJECT_DIR_NAME` environment variable: hosts that give each session its own config directory can choose a short name for the per-project transcript directory
|
||||
- Added the `selection:clear` keybinding action, so a key can be bound to clear an in-app text selection; also works in the agents view
|
||||
- Added a GitLab merge request badge to the footer and statusline: repos with a GitLab remote and an authenticated glab CLI show MR !N with draft/pending/green states
|
||||
- Claude Code now continues your session automatically when a claude.ai usage limit resets; turn it off in `/config` ("Continue automatically at usage limit")
|
||||
- Claude is now told to use your account email only to identify you, and not to send it to unrelated services unless you ask
|
||||
- Security: remote file reads, session restore, CLAUDE.md includes, workflow scripts and file uploads now reject Windows NT-namespace (`\??\`) paths, hardening the remaining pre-approval file accesses against the NTLM credential-leak vector
|
||||
- Fixed auto mode in very long sessions repeatedly re-checking and denying sandboxed commands' network access after the conversation had been compacted
|
||||
- Fixed session-scoped permission answers (including denies) being dropped when answering background subagent tool permission prompts
|
||||
- Fixed a crash when an API response on the non-streaming fallback path (typically via third-party gateways) contained a thinking block missing its thinking field or a text block missing its text field
|
||||
- Fixed markdown rendering becoming extremely slow for some messages containing unusual Unicode sequences
|
||||
- Fixed `SendMessage` rejecting a recipient copied from `ListAgents` when the session name is at the 200-character cap or emoji-heavy
|
||||
- Fixed repository detection mis-reading the host of git remotes with unusual userinfo, producing links and repo-specific behavior for the wrong host
|
||||
- Fixed MCP diagnostics printing resolved secrets: scope-conflict warnings now show the configured `${VAR}` form, and connection-failure details show only the server origin
|
||||
- Fixed `strictKnownMarketplaces` allowlists accepting SCP-style git marketplace sources whose host differs from the one git would actually connect to
|
||||
- Fixed modal text such as the `/login` OAuth URL losing characters when copied in fullscreen
|
||||
- Fixed a `---` horizontal rule in rendered markdown running into the line after it
|
||||
- Fixed consecutive shell commands splitting into multiple "Ran 1 shell command" rows when todo/task updates were interleaved between them
|
||||
- Fixed dialogs like `/permissions` opened while a `!` shell command was running being dismissed when the command finished
|
||||
- Fixed a queued `!` shell command being sent to the model as plain text after pressing up-arrow to edit the queued input
|
||||
- Fixed queued messages reappearing in the prompt history while still queued, Esc while selecting a queued message no longer interrupts the turn, and `!` mode no longer sticks after a mid-turn submit
|
||||
- Fixed accepting the "Try the new fullscreen renderer?" prompt restarting the session without its permission mode (e.g. `--dangerously-skip-permissions`), tool allow/deny rules, model or effort flags
|
||||
- Fixed `/tui` dropping launch `--allowed-tools`/`--disallowed-tools` rules when it restarts; it now declines to switch, with the reason, when the session has restrictions a restart can't carry over
|
||||
- Fixed trust prompts omitting the repository-wide scope warning when the directory was first seen before the repository existed there
|
||||
- Fixed a case where an IDE diff tab closing during a permission re-prompt could answer the new prompt with the previous input
|
||||
- Fixed: files sent to the user during Remote Control sessions hosted by Claude Code Desktop or VS Code now upload, so they open on phone and web instead of showing an empty card
|
||||
- Fixed: after `/login` while `CLAUDE_CODE_OAUTH_TOKEN` is set, the stale-token reminder no longer leaks into Claude's automatically resumed turn — it now appears only to you
|
||||
- Fixed: permission previews now relay only to channel servers admitted by the inbound trust gate, and a server's explicit permission-capability opt-out is honored
|
||||
- Fixed: credential masking on relayed permission previews can no longer hide commands, paths, or destinations from the approver; oversized private-key blocks now redact under full-strength redaction
|
||||
- Fixed: provider API tokens that mask on permission previews now mask even when directly followed by shell delimiters
|
||||
- Fixed Claude Desktop inter-session messages being silently dropped by the recipient session when cross-session messaging read as disabled, which left the sender's query "thinking" for many minutes
|
||||
- Remote Control: signing this computer in to a different claude.ai account or organization now stops the running session within seconds and says why, instead of a misleading HTTP 404 hours later
|
||||
- Remote Control sessions started from Claude Code Desktop or VS Code now keep phones and claude.ai/code updated on the session's permission mode (and claude.ai/code on the model) as they change
|
||||
- Remote Control: effort picks made on a phone or on claude.ai/code now apply to terminal- and Desktop/VS Code-hosted sessions, and the session publishes its effort level to connected clients
|
||||
- `SendMessage` and `ListAgents` now say when your account's session list was too long to check completely, instead of treating unseen sessions as absent
|
||||
- Expired Anthropic profile credential now points you at `/login` when a claude.ai login would take precedence
|
||||
- Improved the transcript: your own prompts now render markdown (highlighted code blocks, inline code, lists) the same way replies do
|
||||
- Improved the "API returned an empty or malformed response" error to say what came back (content type, body kind, size, request ID) and why the original streaming request failed
|
||||
- Improved auto-generated session titles to read as short, specific names (e.g. "Login button bug") rather than sentences restating your request (e.g. "Fix the login button on mobile")
|
||||
- Reduced the context cost of loading the built-in `claude-api` skill from ~200k+ tokens to ~25k by loading reference docs on demand
|
||||
- `/permissions` can now be opened while Claude is working — rule changes apply to the rest of the current turn
|
||||
- `/add-dir <path>` can now be used while Claude is working; `/add-dir`, `/autocompact`, `/theme`, `/help`, `/config` and `/advisor` dialogs open mid-turn in the fullscreen TUI
|
||||
- `/goal` now clears itself with a notice when a turn dies on an unrecoverable error (e.g. revoked auth, an exhausted credit balance, or a context overflow) instead of staying armed
|
||||
- `/goal`: when background tasks keep a goal waiting for 30+ minutes, Claude now checks in on them instead of waiting indefinitely (set `CLAUDE_CODE_GOAL_CHECKIN_MINUTES=0` to opt out)
|
||||
- `claude setup-token` now rejects unexpected extra arguments instead of silently ignoring them
|
||||
- Changed Esc in fullscreen mode to no longer clear a mouse text selection: it interrupts or dismisses as usual and the selection stays highlighted
|
||||
- Removed the redundant "Allowed by auto mode classifier" line that auto mode showed under every Agent tool call
|
||||
- Removed the "Default teammate model" setting from `/config`; agent-team teammates now use the leader's model unless the spawn names one
|
||||
- Dimmed the elapsed-time counter on the running tool header so it no longer competes with the bold counts
|
||||
- Background task notifications delivered between turns are now sent to the model inside `<system-reminder>` tags, matching mid-turn delivery
|
||||
- Mantle: skip the admin-pin availability probe at startup when a main-loop model is already picked
|
||||
- Windows: startup no longer stalls on repeated rename retries when `~/.claude.json` is read-only
|
||||
|
||||
## 2.1.233
|
||||
|
||||
- Added GitLab merge request URL support to the `--worktree` flag and the `claude agents` view (where MRs display as `!N`)
|
||||
- Added an opt-in `forward_user_identity` apps gateway setting on Anthropic upstreams that sends the signed-in user's identity as headers, so a proxy behind the gateway can attribute spend per user
|
||||
- Added opt-in memory cgroup support for Bash tool commands on Linux (`CLAUDE_CODE_TOOL_MEMORY_LIMIT`) so a runaway build can't stall the session
|
||||
- Added `CLAUDE_CODE_WEBFETCH_CACHE_TTL_MS` environment variable to configure the WebFetch session URL cache TTL (default unchanged: 15 minutes)
|
||||
- Fixed cloud sessions occasionally being marked as lost when the environment shut down while Claude was waiting on a permission prompt
|
||||
- Fixed MCP v2 connections endlessly reopening the subscriptions/listen stream against servers that terminate long-held streams on a fixed timeout (e.g. serverless hosts)
|
||||
- Fixed Notification hooks not firing for permission prompts when running under Claude Desktop or VS Code
|
||||
- Fixed idle sessions on Linux sometimes keeping one CPU core at 100% when sandboxing is enabled
|
||||
- Fixed bundled skill aliases like `/checkup` and `/review` reporting "Unknown command" in `-p` mode or with plugins/MCP loaded when a user or project skill shadows the bundled skill
|
||||
- Fixed skill/command argument substitution to prevent argument values from being re-expanded as template markers
|
||||
- Fixed Windows paths spelled with the NT `\??\` device prefix bypassing UNC path validation, closing an NTLM credential-leak vector
|
||||
- Improved `claude self-hosted-runner` session start time: the session branch is now created without rewriting the working tree, and two server round trips no longer block the agent's launch
|
||||
- Improved apps gateway error forwarding: 400/413 errors from Vertex, Foundry, and Claude Platform on AWS upstreams now carry the upstream's own message; fixes a bug with auto-compact on apps gateway
|
||||
- Improved `claude plugin validate` to check a bare `.claude/skills` directory, reporting SKILL.md files whose frontmatter fails to parse
|
||||
- Improved screen reader mode: the `/effort` selector renders as a numbered list with a typed-number prompt, and hint and dialog text is no longer clipped
|
||||
- Improved print mode diagnostics: a `[claude-code:unrecognized_model]` line is written to stderr when a request goes out for a model ID Claude Code doesn't recognize; map it with `modelOverrides` to silence
|
||||
- Changed the GitHub app setup tip to no longer appear in repositories whose origin remote is on gitlab.com or bitbucket.org; the enterprise marketplace tip now covers non-GitHub internal git hosts
|
||||
- Todo/task-tracking tools (TaskCreate/Get/Update/List, TodoWrite) are no longer available on Opus 4.8, Sonnet 5, Fable 5, Mythos 5, and newer models; set `CLAUDE_CODE_ENABLE_TODO_TOOLS=1` to bring them back
|
||||
- Windows: fixed auto mode repeatedly stopping for manual approval on ordinary `cd <dir> && <command> > file` Bash commands (a 2.1.232 regression)
|
||||
- Reverted the 2.1.232 Bash permission changes for Cygwin-style symlinks on Windows and for input redirections (`< file`); a narrower version will return in a later release
|
||||
|
||||
## 2.1.232
|
||||
|
||||
- Subagent forking is now on by default: a `subagent_type: "fork"` subagent inherits the full conversation and prompt cache, and non-teammate agent spawns in interactive sessions now run in the background by default
|
||||
- Type `@` in the prompt to mention another Claude session by name; Claude then uses `SendMessage` to reach that session directly
|
||||
- `SendMessage` now delivers to a bare name that exactly matches one live session, instead of asking to confirm with a ref first
|
||||
- Interactive sessions on one machine now keep unique names: starting or renaming a session to a name another live session already uses gives it a `name-word-word` variant and tells you
|
||||
- Added `/config` rows for "Dialog expiry" and "Messages from your other sessions" (cross-session inbound accept/hold/refuse)
|
||||
- Added secret redaction for GitLab token families (`glrt-`, `gloas-`, `glptt-`, `glagent-`, `glimt-`, `glsoat-`, `glcbt-`, `glft-`, `glffct-`) and full redaction of routable `glpat-`/`gldt-` tokens; the `glab` CLI config store gets the same sandbox and credential-path protection as `gh`
|
||||
- Added GitLab support to plugin marketplaces: bare `gitlab.com` repo URLs (including nested subgroups) now clone like `github.com` URLs, and clone auth-failure hints name your actual git host
|
||||
- Settings: `additionalMarketplaces` and `allowedMarketplaces` are now accepted as friendlier aliases for `extraKnownMarketplaces` and `strictKnownMarketplaces`
|
||||
- Enterprise policy: a url-typed `blockedMarketplaces` entry for a bare repo URL keeps blocking that URL when the CLI classifies it as a git clone
|
||||
- Gateway: the `desktop:` overlay now accepts every released Desktop setting (was 11 hand-listed keys), validated at boot against Desktop's own schema; unknown or invalid keys fail boot
|
||||
- Gateway: empty `managed.policies[].match.groups`/`admin.admin_groups` entries and malformed `email_domain` values (empty, or containing `@`, whitespace, or commas) now fail at boot instead of silently matching no one or granting admin access
|
||||
- Fable 5 is offered as an advisor in `/advisor` again for organizations with Fable access, with usage-credits consent set up through `/model fable`
|
||||
- Fixed a PowerShell permission bypass where variable-writing parameters could silently overwrite `$PSDefaultParameterValues` and redirect later commands' file access
|
||||
- Fixed a Windows permission bypass where Git Bash followed Cygwin-style symlinks that path validation saw as regular files; writes through them now require permission approval
|
||||
- Fixed nested git repositories inheriting trust from a parent directory; each repository now requires its own trust confirmation
|
||||
- Fixed MCP connections hanging for the full 30-second connect timeout when a server fails to answer or sends a malformed reply to the protocol-version probe
|
||||
- Fixed Remote Control sessions hosted by a bridge inside a cloud session inheriting that session's transcript or credentials
|
||||
- Fixed Remote Control sessions started from Claude Desktop or an IDE appearing as a new claude.ai session each time the local session was resumed; they now reattach to the existing one
|
||||
- Fixed Remote Control sessions appearing unreachable to newly attached clients while idle
|
||||
- Fixed Remote Control bridge sessions not restoring conversation history when the session worker restarts
|
||||
- Remote Control: resuming a conversation whose session was deleted from claude.ai or the app now starts a replacement instead of failing with a message about your login (regressed in v2.1.227)
|
||||
- Fixed Cloud gateway `/login` exiting silently or leaving an unresponsive terminal after "Press Enter to continue" when managed settings failed to load; the reason is now shown
|
||||
- Fixed voice mode on native builds getting stuck on "listening…" when the voice service rejected the connection; the rejection is now shown immediately
|
||||
- Fixed mTLS client certificate rotation requiring a restart; Claude Code now reloads the rotated cert and key automatically on connection errors
|
||||
- Fixed malformed AWS or Vertex region values being used to build request URLs; they now fall back to the default region
|
||||
- Fixed stream idle timeout errors failing the request instead of recovering on Bedrock, Vertex, and gateway deployments
|
||||
- Fixed content-sized overlays containing truncated text rendering one column too wide, and start-truncated text collapsing to an ellipsis
|
||||
- Fixed a stray garbled character where a long shell-command or agent-description preview was cut off mid-emoji
|
||||
- Fixed a startup race that could silently unregister a plugin marketplace due to concurrent writes to `known_marketplaces.json`
|
||||
- Fixed `/update` and `/tui` refusing to restart while work that survives the relaunch was running
|
||||
- Fixed usage-limit guidance suggesting unavailable slash commands in SDK and remote sessions
|
||||
- Fixed the consent message for interactive `--advisor fable` launches, which told you to run `/model fable` in an interactive session that had just exited
|
||||
- Improved fullscreen streaming: long sessions stay responsive because the whole conversation is no longer re-normalized on every update
|
||||
- Improved the managed settings approval dialog: shows endpoint URLs, uses clearer wording for telemetry-only changes, skips routine OpenTelemetry options, and requires approval for server-managed sandbox binary overrides (`sandbox.bwrapPath`, `sandbox.socatPath`, `sandbox.ripgrep`)
|
||||
- `/feedback` and `/bug` now open immediately when invoked while Claude is responding, instead of waiting for the turn to finish
|
||||
- `/plugin install plugin@marketplace` now refreshes the marketplace first, so newly published plugins install without a manual marketplace update
|
||||
- `/code-review` at high, xhigh, and max effort now runs in a background agent like the other levels
|
||||
- Pasted and clipboard images are read without blocking the event loop
|
||||
- Remote Control now keeps reconnecting for about 30 minutes after a network blip and no longer drops after a few blips spread across an hour
|
||||
- Remote Control: resuming a conversation no longer silently takes Remote Control away from another Claude Code on the same machine that still has it; run `/remote-control` there to move it
|
||||
- Updated agent panel: completed subagents hide immediately with a `/tasks` footer hint, and the "↓ N more" overflow indicator moved left for visibility
|
||||
- Remote Control: the terminal now says whether a session was taken over by another device, ended from another app, or deleted, and stops suggesting a reconnect that would undo it
|
||||
- Bash input redirections (`< file`) are now permission-checked like their argument spellings on all platforms
|
||||
- Shortened the message shown when resuming a completed background agent
|
||||
- Cowork sessions no longer inline external @-imports from user-scope memory files
|
||||
- Hardened the auto-generated cross-session messaging socket directory on shared `/tmp`: a pre-planted symlink or another user's directory is now refused instead of used
|
||||
- Hardened the Linux filesystem sandbox against a protected-path bypass
|
||||
- Changed `sandbox.ripgrep` to be honored only from user, managed, and `--settings` settings; project settings can no longer override the sandbox's ripgrep binary
|
||||
- Removed the startup tip suggesting you create custom subagents, and the matching nudge in the `/powerup` tour
|
||||
|
||||
## 2.1.231
|
||||
|
||||
- Fixed MCP OAuth sign-in failing with a redirect URI mismatch for servers that use a pre-registered OAuth client, such as Slack
|
||||
|
||||
## 2.1.229
|
||||
|
||||
- Documented `claude remote-control --continue` for resuming the most recent Remote Control session
|
||||
- Added server-supplied Claude Code hook support for self-hosted runner sessions, matching managed-environment behavior
|
||||
- Added SSE keepalive pings to gateway streaming responses during long thinking pauses, preventing idle-timeout disconnects on Vertex and Bedrock upstreams
|
||||
- Added plugin marketplace `command` sources: a local command (e.g. an IDE) prints the plugin directory, which is re-resolved each session and applied without a restart; `mode: "link"` uses it in place
|
||||
- `ListAgents` now marks disconnected Remote Control sessions as `offline` and labels your cloud sessions as `cloud`
|
||||
- Fixed long responses partly disappearing while streaming and being printed twice in the terminal
|
||||
- Fixed a crash to the error screen (including on `--resume` of the affected session) when a tool call had a non-string `glob`, `file_path`, or `command` value
|
||||
- Fixed a RangeError crash when a progress bar or markdown table rendered in a very narrow terminal window (could also crash `claude --continue`/`--resume` at startup)
|
||||
- Fixed a crash on Windows when a tool call or message referenced a file by an extended-length (`\\?\`) or UNC path
|
||||
- Fixed auto mode failing on every tool call for users who disable the attribution header via `CLAUDE_CODE_ATTRIBUTION_HEADER` (direct Anthropic API connections)
|
||||
- Fixed `/model` rejecting Sonnet/Opus 1M for claude.ai subscribers using a custom `ANTHROPIC_BASE_URL` gateway
|
||||
- Fixed MCP OAuth with strict authorization servers by using `127.0.0.1` instead of `localhost` in the redirect URI
|
||||
- Fixed Remote Control clients showing a stuck working spinner after a slash command typed in the laptop terminal
|
||||
- Fixed the Claude Code Review workflow generated by `/install-github-app` completing without posting its review on the pull request
|
||||
- Fixed multi-second UI stalls after editing a file with thousands of IDE diagnostics while the IDE extension is connected
|
||||
- Fixed one-shot `claude plugin` commands leaving a stray liveness file that could prevent cleanup of outdated plugin versions
|
||||
- Fixed dynamic workflows inside CPU-limited containers using the host machine's core count instead of the container's CPU limit
|
||||
- Fixed a file-watcher handle leak after atomic file replacements, and an uncaught error on Windows when the scheduled-tasks watcher failed on a network or virtual filesystem
|
||||
- Fixed SDK and `--input-format stream-json` sessions getting a 400 API error when a whitespace-only message was submitted
|
||||
- Fixed conversations whose messages alone exceed the API's 32 MB request limit retrying compaction when no images or documents can be stripped; they now fail once with a clear message
|
||||
- Fixed OpenTelemetry export from Claude Desktop sessions being rejected by the Desktop-managed gateway when that gateway is also the telemetry endpoint
|
||||
- Fixed self-hosted runner and other remote sessions exiting at startup when `managed-mcp.json` is deployed and the server delivers MCP servers; those servers are now skipped with a warning
|
||||
- Fixed self-hosted runner repository preparation hanging on a Git Credential Manager prompt; git now fails fast when credentials are missing
|
||||
- Improved workflow fan-outs to stagger same-prefix sibling agents so subsequent agents read the cached prompt prefix instead of re-paying it (`CLAUDE_CODE_WORKFLOW_PREFIX_STAGGER_MS=0` disables)
|
||||
- Improved "prompt is too long" errors to explain why automatic compaction could not recover instead of only suggesting `/compact`
|
||||
- Improved sandbox: IPv6 literals in network domain lists are now bracketed (`[::1]:443`), and ambiguous spellings are enforced fail-closed and flagged by `/doctor`
|
||||
- Updated `/login` to repeat the `CLAUDE_CODE_OAUTH_TOKEN` override warning after a successful login
|
||||
- Changed `/commit-push-pr` so git/gh commands with dangerous flags (`--force`, `--amend`, `--no-verify`, etc.) are no longer auto-approved
|
||||
- Changed self-hosted runner Windows startup to require an explicit `--base-dir`; there is no default checkout directory on Windows
|
||||
- [VSCode] "Report a problem" and `/bug` now open the built-in feedback dialog instead of a retired survey link
|
||||
- [VSCode] Made the `/btw` side-question panel resizable by dragging its boundary, in both side-docked and stacked layouts
|
||||
- [VSCode] Added session groups in the sidebar — right-click to create, rename, or delete; Cmd/Ctrl- or Shift-click to move several sessions at once
|
||||
|
||||
## 2.1.228
|
||||
|
||||
- Fixed interactive sessions that could stop redrawing entirely, while the process kept running, after a rare internal layout error
|
||||
- Fixed `git` / Git Bash not being found on Windows when Claude Code is launched from a parent folder of the git installation
|
||||
- Fixed `/tui` reverting the session to an earlier model when `/model` had been changed since the last response
|
||||
- Fixed cross-session messaging sometimes starting without an inbox in the first session after install or upgrade
|
||||
- Fixed Remote Control `/resume` while connected leaking the resumed conversation's title or history into the connected session
|
||||
- Fixed `claude self-hosted-runner` sessions failing on every fresh runner when the `checkout` hook fails for a repository the session doesn't push to; that repository is now skipped with a warning
|
||||
- Fixed self-hosted runners ending sessions in the gap between a background task finishing and the follow-up turn starting
|
||||
- Fixed session cleanup deleting contents inside a project's memory folder
|
||||
- Fixed background plugin-cache cleanup deleting a plugin's cache when its only version is a symlinked development checkout
|
||||
- Fixed a settings-merge issue where a marketplace entry redefined in a higher-precedence settings tier could inherit another tier's custom headers; marketplace entries now merge as whole entries
|
||||
- Fixed the deferred-tools reminder occasionally being sent to the model twice after a skill invocation
|
||||
- Hardened skills synced from claude.ai: they no longer shadow local commands or MCP prompts, their descriptions are sanitized and labeled, and on your machine their bodies don't run `!` commands or expand `@` files
|
||||
- Improved cross-session messages: the sender and body now display inline instead of a collapsed line, and messages to Remote Control sessions on other machines show your Remote Control session name as the sender
|
||||
- Improved Vertex AI credential handling: expired or missing Google Cloud credentials now fail within seconds instead of retrying for minutes
|
||||
- Improved compaction progress: the retry countdown and stall hint now appear during compaction instead of only a progress bar
|
||||
- Updated terminal title busy-spinner glyphs to reduce tab-bar jitter on some terminals
|
||||
- Changed the Write tool so newer models can overwrite an existing file they haven't read this session, matching the Edit tool's rules; older models still require the read first
|
||||
- Removed the outdated note about auto mode sessions costing slightly more from the first-use notice for Pro, Max, and Team plans
|
||||
|
||||
## 2.1.227
|
||||
|
||||
- Fixed feature flags being evaluated without the user's subscription tier when a session started with an expired login token, which could wrongly prompt Max plan users to enable usage credits for Fable
|
||||
- Fixed every Bash command failing under `claude-code-action` with `allowed_non_write_users` on GitHub-hosted runners
|
||||
- Fixed `/tui` bringing back a conversation that had been rewound to before its first message
|
||||
- Improved slash-command menu: blue now marks only the selected row, matched characters are bolded instead of recolored, and emoji or accented names keep their glyphs
|
||||
- Improved performance: fewer event-loop stalls on file-not-found suggestions and at-mention size checks
|
||||
|
||||
## 2.1.226
|
||||
|
||||
- Bug fixes and reliability improvements
|
||||
|
||||
## 2.1.225
|
||||
|
||||
- Added gateway spend-limit support to Claude Code's usage warning; the limit-reached message now names the cap, its reset time, and the operator's message (requires the gateway on 2.1.225)
|
||||
- Added a workspace trust prompt to `claude agents` for untrusted directories, matching the behavior of `claude`
|
||||
- Fixed a transient 401 replacing a long-lived `CLAUDE_CODE_OAUTH_TOKEN` with a stored login's short-lived token, breaking headless sessions until restart
|
||||
- Fixed MCP OAuth servers on macOS intermittently failing with a burst of 401 errors, as if never authenticated, after a keychain read timed out
|
||||
- Fixed auto mode counting a safety-filter refusal of its own permission check toward the consecutive-block limit; the action is still denied, but the model is now told to move on rather than retry
|
||||
- Fixed cross-session messages staying parked without a notice or expiry in headless sessions and during startup
|
||||
- Fixed conversation history breaking on Remote Control session resume after very large conversations were compacted
|
||||
- Fixed hovering over a session in another project in the agents list changing the directory the next agent starts in
|
||||
- Fixed `claude self-hosted-runner` registering and then failing every session when `--base-dir` cannot be created or written; it now exits at startup with a clear error
|
||||
- Fixed Claude Code on the web sessions being misreported as stuck, re-sending a growing event backlog on every reconnect
|
||||
- Improved Remote Control: photos attached from the Claude app are now shown to Claude directly instead of being read from disk with a separate tool call
|
||||
- [VSCode] Fixed Focus view folding away the latest to-do list, a pending question's context, and settled answers; thinking-only folds show "Thought for Ns" and re-collapse when their turn completes
|
||||
- SendMessage can now start a conversation with your Remote Control sessions on other machines by name (`ListAgents` shows them as `name [ref]`), instead of only replying after they message you first
|
||||
- SendMessage: a Remote Control recipient you already confirmed is never swapped for a same-named session on this machine when its own list couldn't be checked
|
||||
|
||||
## 2.1.224
|
||||
|
||||
- Added self-hosted environments: `claude self-hosted-runner` turns your own machines or containers into a place Claude Code web, mobile, and desktop sessions can run, on Team and Enterprise plans
|
||||
- Added `archive` plugin source: install plugins from a zip over HTTPS without git or npm, with optional SHA-256 pinning
|
||||
- Added a cancel-and-confirm step when removing an unavailable paste changes a command's text
|
||||
- Added `ANTHROPIC_BEDROCK_REGION_PREFIX` env var for Bedrock to prefer a specific cross-region inference profile over the `AWS_REGION`-derived one
|
||||
- Added `crossSessionInbound` and `dialogExpiry` settings: cross-session messages sent to a session running with bypassed permissions are held for your approval, and messages to other sessions auto-deliver
|
||||
- Added sandbox credential-masking options: `extract` and `onExtractNoMatch` for structured env values, `decode: "jwt"` with `maskClaims` for JWT-aware masking, and `awsPairs`/`sigv4` for AWS SigV4 re-signing; these need `network.tlsTerminate` and are honored only from user, managed, or `--settings` settings
|
||||
- Added cross-session `SendMessage`: Claude Code sessions can now message each other, on any of your machines, with `ListAgents` to discover them (macOS and Linux)
|
||||
- Fixed long (>200 char) project paths resolving to another project's session directory under a shared sanitized prefix; session list, rename, fork, delete and `/resume` no longer cross projects
|
||||
- Fixed `SendMessage` reporting "Message sent" when the write to a teammate's inbox had actually failed; failed deliveries are now reported as errors
|
||||
- Fixed sandbox filesystem deny entries written with a trailing slash (e.g. `denyRead: "~/.aws/"`) being silently bypassable on Linux and macOS
|
||||
- Fixed sandbox violation details never appearing in Bash tool results; Claude now sees which file or network access was denied and why
|
||||
- Fixed MCP tools that connect mid-turn being deferred for tool search without their names announced to the model
|
||||
- Fixed plugin install records being silently corrupted when the same plugin is installed in multiple projects
|
||||
- Fixed recalled or restored paste content occasionally attaching wrong data or silently losing text when the paste had aged out or placeholder numbers collided
|
||||
- Fixed copy-on-select on Wayland sometimes not reaching the clipboard; the two selection writes no longer race
|
||||
- Fixed the feedback survey's transcript share silently failing on long sessions; a failed share now shows an error instead of a success message
|
||||
- Fixed Remote Control auto-start intermittently failing with "Remote credentials fetch failed" on a cold start with a stale login token
|
||||
- Fixed Remote Control and SDK clients showing a blank "(no content)" message after `/clear` and other output-less commands
|
||||
- Fixed a Remote Control session recreated after its server session expired uploading prior local conversation history into the new session
|
||||
- Improved fullscreen mode to keep the full pre-compaction history in scrollback across repeated compactions, instead of only the most recent interval
|
||||
- Improved Remote Control: attached web and mobile clients now see compaction progress and the post-compaction boundary instead of a silent pause; `/clear` resets now propagate to attached clients
|
||||
- Improved Remote Control: connection failures now show a persistent failure indicator with details and a reconnect shortcut, instead of only an 8-second toast
|
||||
- Removed the 200-subagent-per-session spawn cap; long-running sessions no longer refuse new agents (concurrency and depth limits still apply)
|
||||
- Changed managed settings: the approval prompt no longer re-appears after re-login or org switching when the organization's settings are unchanged
|
||||
- Changed the feedback-survey transcript share: with your consent it now also uploads the last request's model settings — the system prompt (which includes your `CLAUDE.md` instructions), tool definitions, and model parameters. Secrets are redacted as before, and these fields are dropped first if the share is too large
|
||||
- Changed the Bash tool description to always note that command output is displayed to the model, not reliably to the user
|
||||
- Changed recalled paste placeholder numbers to renumber when accepted into the input
|
||||
- Changed Remote Control to archive the stale server session instead of leaving a dead one listed when a fresh session is minted after compaction or `/resume`
|
||||
- [VSCode] Fixed the extension showing Remote Control as connected after the connection failed
|
||||
- Fixed a session resume silently reconnecting Remote Control after the user turned it off (`--resume`, SDK hosts, and the VS Code extension)
|
||||
- [VSCode] Fixed sessions not honoring `remoteControlAtStartup` when explicitly enabled
|
||||
|
||||
## 2.1.223
|
||||
|
||||
- Added owner wildcard entries (`"owner/*"`) to the `strictKnownMarketplaces` and `blockedMarketplaces` managed settings for allowing or blocking all marketplace repos under a GitHub org
|
||||
- Added a warning when workflow agents, forked skills, slash commands, or resumed background agents' requested subagent model is restricted and the parent model runs instead
|
||||
- Added a `/teleport` hint in cloud sessions showing how to continue locally with `claude --teleport <session id>`
|
||||
- Fixed a Bash permission bypass where a crafted command could hide parts of itself from permission checks
|
||||
- Fixed permission prompts so commands padded with tabs or invisible Unicode can no longer hide part of the command from the approval dialog
|
||||
- Fixed workflow scripts being able to use dynamic `import()` to run code outside the workflow sandbox
|
||||
- Fixed a permission gap where an agent definition's `bypassPermissions` mode ignored the org bypass-permissions disable policy
|
||||
- Fixed resuming a session after a mid-session `/cd` coming back empty
|
||||
- Fixed gateway model discovery hiding Claude models registered under provider-prefixed IDs such as `vertex_ai/claude-*` or `bedrock/anthropic.claude-*`
|
||||
- Fixed `modelOverrides` keys that aren't Anthropic model IDs being treated as the session's canonical model ID; unknown keys are now ignored as documented
|
||||
- Fixed managed settings: server-delivered settings no longer disable the env block of a machine-local `managed-settings.json` or MDM profile; admin env now merges per key
|
||||
- Fixed sandboxed commands failing to start on Linux when `sandbox.filesystem.denyWrite` covers the working directory
|
||||
- Fixed forked background agents getting stuck "already resuming" for the rest of the session when rebuilding the fork's parent prompt failed during resume
|
||||
- Fixed a resumed session failing every turn, or leaving the interactive app on an unresponsive error screen, when its history held a malformed diagnostics attachment
|
||||
- Fixed a rare hang when parsing unusual `git push` output
|
||||
- Changed `CLAUDE_CODE_DISABLE_1M_CONTEXT` to hold every Claude model with a native 1M window to 200K via auto-compaction, not just a fixed list; a startup warning now appears when auto-compaction isn't holding the session to 200K
|
||||
- Changed auto-compact to keep sessions on unrecognized model IDs within the assumed context window instead of letting them grow past it; set `CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT=1` to restore the previous behavior
|
||||
- Changed `/review` to be an alias of `/code-review`, which reviews the current diff or a PR (`/code-review <level> <pr#>`); use `/code-review ultra` for a deep cloud review
|
||||
- Changed `/code-review` with no effort level to reuse the level you typed last; type a level like `/code-review high` to change it
|
||||
|
||||
## 2.1.222
|
||||
|
||||
- Fixed worktree-isolated sessions and their subagents being able to run destructive git commands against the main checkout; isolation now applies to file edits and Bash in every session type
|
||||
- Fixed PreToolUse auto-allow hooks bypassing tool restrictions in background agent tasks (summaries, compaction, renames)
|
||||
- Fixed `/usage-credits` on Team and Enterprise showing "you've already sent a usage credit request" for members whose earlier request was dismissed, blocking them from sending a new one
|
||||
- Fixed the startup connectivity check hanging and then failing behind an HTTPS proxy; it now uses the same proxy-aware transport as API requests and times out with a clear message
|
||||
- Fixed "Connection closed mid-response" errors being reported on responses that had actually completed
|
||||
- Fixed `/usage` overattributing usage to MCP servers: a server's share now reflects only the requests that actually consumed its tool results, instead of every turn after any call to it
|
||||
- Fixed sessions not linking to pull requests created after the branch was pushed, including through the GitHub REST API
|
||||
- Fixed org-restricted `model: opus`-style subagent and teammate family aliases dropping to the parent model instead of stepping down to the newest org-allowed model in the family
|
||||
- Fixed stream idle timeout firing on custom `ANTHROPIC_BASE_URL` gateways despite server keep-alive pings arriving on the wire
|
||||
- Fixed claude.ai connectors being falsely marked as needing authorization when the session token is invalid — they now show a `/login` hint instead
|
||||
- Fixed tool errors not being displayed for tools no longer available locally, for example after an MCP server is removed
|
||||
- Fixed `SendMessage` rejecting a long summary — it now truncates instead, so sends no longer fail on a character limit
|
||||
- Fixed the spinner's effort label in a subagent's transcript view showing the session's effort level instead of the subagent's own `effort:` setting
|
||||
- Fixed rare crashes when a file watcher hit a filesystem error or during file-watcher teardown
|
||||
- Fixed screen readers re-reading the whole input line on every backspace in `--ax-screen-reader` mode — end-of-line deletions now echo just the deleted characters
|
||||
- Fixed host model-selection keys not taking precedence over a stale on-disk `managed-settings.json` when `CLAUDE_CODE_PROVIDER_MANAGED_BY_HOST` is set
|
||||
- Improved auto mode safety: messages sent to other agent sessions via `SendMessage` are now evaluated by the permission classifier before dispatch
|
||||
- Improved the refusal when Claude tries to invoke a skill with `disable-model-invocation`: Claude is now told to ask you to run the skill instead of replicating its workflow
|
||||
- Improved the `/diff` view, the Remote Control workspace diff, and file-edit diffs in Claude Code on the web sessions to use raw git blob content, ignoring workspace-configured diff drivers and textconv
|
||||
- Changed Remote Control auto-start so repo-local settings (`.claude/settings.json` or `.claude/settings.local.json`) can no longer turn it on (they can still turn it off); enable it at user scope via `/config`
|
||||
- Removed ultraplan feature
|
||||
|
||||
## 2.1.221
|
||||
|
||||
- [VSCode] Added Focus view: a chat-menu toggle that hides tool activity behind an expandable per-turn summary with a live running-tool indicator, toggled with `Ctrl+Alt+F` or the "Claude Code: Toggle Focus view" command
|
||||
- Added `mode: "mask"` for sandbox credential files on Linux and WSL — sandboxed commands read a sentinel copy (the whole file, or just the spans captured by an `extract` regex) while the sandbox proxy substitutes the real value on egress; on macOS file masking falls back to `deny`
|
||||
- Added warnings to `claude plugin validate` when a marketplace or plugin name would be rejected by Claude Desktop's managed marketplace sync
|
||||
- Added a `prompt-audit` subcommand to the `claude-api` skill for auditing prompts and tool descriptions for patterns written for older models
|
||||
- Fixed a Bash tool permission-check bypass where zsh could execute hidden commands in `[[ ]]` regex conditionals; affected commands now prompt for permission
|
||||
- Fixed PowerShell permission checks mishandling paths containing quote characters on Windows; such paths now prompt for approval
|
||||
- Fixed the thinking toggle having no effect for the rest of a session that started with thinking off; disabling an MCP server mid-connect no longer silently reverts
|
||||
- Fixed MCP servers from `--mcp-config` not being connected before the first turn in print mode (`-p`), which made the model emit tool calls as literal text
|
||||
- Fixed @-mentioned files being silently dropped when pressing Esc to retract a prompt and resubmitting it
|
||||
- Fixed a crash when preparing API requests for SDK MCP tools named after built-in object properties such as `constructor`
|
||||
- Fixed WebSearch failing with a 400 error at effort `xhigh`/`max` when thinking is disabled
|
||||
- Fixed sandboxed large uploads failing with TLS errors through the sandbox proxy
|
||||
- Fixed Team and Enterprise spend-limit message incorrectly blaming the org's monthly limit instead of your individual spend limit
|
||||
- Fixed Bedrock authentication with AWS SSO named profiles failing in desktop-managed sessions on Windows machines that set a stray `HOME` environment variable
|
||||
- Fixed `CLAUDE_CODE_RESUME_INTERRUPTED_TURN=0` not disabling interrupted-turn auto-resume; falsy values are now honored
|
||||
- Fixed a rare wake-from-sleep race where two Claude Code processes could both refresh the same MCP connector or WIF OAuth token at once, forcing re-authentication
|
||||
- Fixed renaming a session from Claude Code Desktop or claude.ai not updating the CLI's session name; session names from every rename surface are now sanitized
|
||||
- Fixed plugin- and org-delivered skills named after terminal-only built-ins (e.g. `/help`, `/feedback`) being un-invocable in non-interactive sessions
|
||||
- Fixed the "Plugins changed" notification lingering after plugins were reloaded instead of clearing
|
||||
- Fixed Vim mode: the yank register now survives dialogs, history search, and the transcript view instead of being silently emptied
|
||||
- Fixed Vim mode: undoing back to an empty prompt now arms the "press ← again" confirm before returning to the agent view
|
||||
- Improved tool search on Google Vertex AI: re-enabled for Claude 4.5-generation and newer models
|
||||
- Improved auto mode: permission checks for parallel tool calls are now cache-efficient, and switching modes while a check is pending reliably prompts instead of applying the stale result
|
||||
- Reduced prompt-cache costs for auto-mode permission checks by reusing the cached conversation prefix across decisions
|
||||
- Improved Stats panel to count cache tokens in its token totals, with a breakdown by input, output, cache read, and cache write
|
||||
- Improved `/ultrareview` error messages when a repo shares no history with its base: a checkout with no branches is now refused up front with advice to create one, and refusal hints no longer suggest `git fetch --unshallow` on clones that are already complete
|
||||
- Improved Windows startup: process creation times are now read via a native kernel32 call instead of spawning PowerShell, so endpoint security tools that gate `powershell.exe` no longer prompt
|
||||
- Changed background sessions to commit and push to preserve work, open a draft PR only when the task calls for one, follow your CLAUDE.md git instructions, and always end by reporting where the work lives
|
||||
- Changed `/plugin install` to refresh a stale marketplace catalog and retry before reporting a plugin not found
|
||||
- Changed plugins installed from `/plugin` to activate immediately when safe, instead of always requiring `/reload-plugins`
|
||||
- Changed plugins to accept `"."` as a `skills` path, and the root-level `SKILL.md` validation error now suggests using the plugin root
|
||||
- Changed `/status` to show the session kind: `interactive`, or a background job that is `attached` or `unattended`
|
||||
- Changed emoji autocomplete to accept common alternate shortcodes like `:thumbsup:`, `:thumbsdown:`, and `:love:`
|
||||
- Changed sessions forked with `/fork` to create a new worktree of their own instead of working in the original session's checkout
|
||||
- Changed Claude in Chrome to close the browser tabs it opens once it no longer needs them
|
||||
- Changed fast mode to report on the stream when usage credits run out mid-session, instead of failing silently
|
||||
- Changed Monitor: a watch that exits without producing any output now says so instead of reporting "stream ended"
|
||||
- Changed the Gateway `model` field validation: non-string values are rejected with a 400 instead of being forwarded
|
||||
- Removed the repeated "Permission mode changed while the auto-mode classifier call was queued" notice from approval prompts
|
||||
|
||||
## 2.1.220
|
||||
|
||||
- Bug fixes and reliability improvements
|
||||
|
||||
## 2.1.219
|
||||
|
||||
- Added Claude Opus 5 (`claude-opus-5`), now the default Opus model — 1M context, fast mode at $10/$50 per Mtok
|
||||
- Added `sandbox.network.strictAllowlist` setting to deny non-allowlisted hosts for sandboxed commands without prompting
|
||||
- Added `DirectoryAdded` hook that fires after `/add-dir` or the SDK `register_repo_root` control request registers a new working directory mid-session
|
||||
- Added `mcp_server_errors` to the headless stream-json init event, listing `--mcp-config` entries skipped by config validation; terminal runs print a startup warning
|
||||
- Added the `workflowSizeGuideline` settings key so the advisory Dynamic workflow size guideline can be set from any settings file; the `/config` row is hidden while one does
|
||||
- Added nested subagent forwarding in stream-json: subagents spawned at depth-2+ now appear when `--forward-subagent-text` is set, keyed by their spawning Agent `tool_use` id
|
||||
- Fixed `claude -p` text output dropping the answer already produced when a turn dies on a mid-stream API error
|
||||
- Added HTTP status and error text to `claude mcp list` and `/mcp` when a server fails to connect, and a warning for MCP config values with hidden leading or trailing whitespace
|
||||
- Fixed the Fable model row showing "Requires usage credits" for plans that include it, when a stale cache had baked the label in
|
||||
- Fixed the `/model` picker showing the merged Opus row as plain "Opus" instead of "Opus (1M context)"
|
||||
- Fixed copy-on-select inside GNU screen printing base64 into the terminal instead of copying the selection
|
||||
- Fixed Remote Control clients keeping a stale fast-mode status after a model switch, reconnect, or failed org check
|
||||
- Fixed `CLAUDE_CODE_GIT_BASH_PATH` on Windows exiting or being used as bash when the path isn't a bash/sh binary; it's now ignored with a warning
|
||||
- Fixed Vim mode: pressing ← on an empty prompt now returns to the agent view from NORMAL mode, not just INSERT
|
||||
- Fixed screen-reader mode rewriting the entire input line on every keystroke instead of echoing only the typed character
|
||||
- Improved the "Remote Control is only available via api.anthropic.com" error to name the specific setting that caused it
|
||||
- Improved `claude --teleport` to show which repo your current checkout points at when it doesn't match the session's repo
|
||||
- Changed dynamic workflows to default to a medium size guideline (aim for fewer than 15 agents); pick another size or unrestricted with Dynamic workflow size in `/config`
|
||||
- Changed managed MCP allowlist/denylist `${VAR}` entries to resolve from the startup environment and managed-settings env instead of settings-file env
|
||||
- Changed the `/model` picker to highlight only the newest model's name, so the highlight marks the new release rather than an arbitrary subset of the list
|
||||
- Added the current default workflow size to the running-workflow status line, with a pointer to `/config` for changing it
|
||||
- Removed Opus 4.7 from fast mode; `/fast` now applies to Opus 5 and Opus 4.8
|
||||
- Updated the claude-api skill to default to Claude Opus 5, with a migration path from Opus 4.8
|
||||
- Subagents can now spawn nested subagents up to depth 3 by default (was 1); set CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH=1 to disable nesting
|
||||
|
||||
## 2.1.218
|
||||
|
||||
- Changed `/code-review` to run as a background subagent, so review work no longer fills your conversation and keeps stacked slash commands as its review target
|
||||
- Added screen-reader announcements of deleted text for word and line deletions (`Option+Delete`, `Ctrl+W`, `Cmd+Backspace`, `Ctrl+U`, `Ctrl+K`) in `--ax-screen-reader` mode
|
||||
- Fixed Windows paths with `\u`-prefixed segments (like `C:\Users\unicorn`) being corrupted into CJK characters in tool inputs, which made those files inaccessible
|
||||
- Fixed the left arrow key discarding the conversation with no undo: presses right after editing now ask to confirm, and Esc in the agent view returns to the conversation it backgrounded
|
||||
- Fixed multi-line paste collapsing into one line with `j` in place of newlines in terminals that encode pasted newlines as Ctrl+J
|
||||
- Fixed `/context` reporting stale pre-compact token usage after compacting from the message picker
|
||||
- Fixed `/ultrareview` failing on descriptive arguments like "review my auth changes" — they now run a review of your current branch with the text applied as a note to the findings
|
||||
- Fixed `/code-review ultra` silently running a local review in non-interactive sessions — it now launches the cloud review
|
||||
- Fixed gateway spend metering to price Bedrock application-inference-profile ARNs and other config-mapped upstream model IDs at the configured model's rates
|
||||
- Fixed mojibake when a long IDE selection was truncated mid-emoji, and a case where a tool executor error could be silently dropped
|
||||
- Fixed an engine teardown race that could start and abandon a phantom turn, and made input pushed after close consistently rejected
|
||||
- Fixed spurious "[Request interrupted by user]" messages after interrupted tool calls, and an unpaired `tool_use` block left in the transcript when a tool aborted mid-response
|
||||
- Fixed VoiceOver reading "new line" instead of echoing the typed space at the end of the input in `--ax-screen-reader` mode
|
||||
- Fixed plugin and settings panels not moving the terminal cursor to the focused row, so screen readers and magnifiers can follow arrow-key navigation
|
||||
- Fixed crashes (maximum call stack exceeded) when a deeply nested watched directory tree was deleted or moved, and when rendering deeply nested UI trees
|
||||
- Fixed pull request events occasionally being lost when a session exited immediately after creating or linking a PR
|
||||
- Fixed the Bedrock setup wizard failing profile verification for assume-role profiles in partitioned AWS regions and on proxy-only networks
|
||||
- Fixed rare negative or incorrect turn duration measurements after a system clock adjustment by timing turns with a monotonic clock
|
||||
- Fixed the "N MCP servers need authentication" startup notice over-counting claude.ai connectors that aren't connected in claude.ai
|
||||
- Fixed prompt history entries being dropped or duplicated when history writes raced or failed
|
||||
- Fixed a retry loop that re-sent identical doomed requests after a context-overflow error with a large thinking budget; `Ctrl+B` backgrounding now applies the same background-shell caps as other paths
|
||||
- Fixed agent frontmatter hooks running from untrusted folders: hooks now require the agent file's own folder to have accepted workspace trust
|
||||
- Fixed fork-session lineage being lost after compaction in headless and SDK sessions
|
||||
- Fixed a resumed session failing every turn, or crashing on resume, when its history held a malformed delta attachment
|
||||
- Improved `/ultrareview` error feedback so Claude can correct an invalid argument instead of retrying it unchanged
|
||||
- Improved auto mode: the dangerous-rm, background-`&`, and suspicious-Windows-path checks no longer open permission dialogs; the auto-mode classifier adjudicates them instead
|
||||
- Improved sandbox command restrictions for IDE interactions
|
||||
- Improved trust dialogs to name the repository root the grant covers
|
||||
- Changed `/deep-research` to start only when invoked manually; Claude no longer launches it on its own
|
||||
- Changed plan mode with auto to no longer prompt for Bash commands the static analyzer can't prove read-only; the auto-mode classifier judges them instead
|
||||
- Added an announcement when fast mode changes as a result of switching models via `/config model=<x>` or Remote Control
|
||||
- Changed server-managed settings so benign feature and cost toggles no longer trigger the settings-approval prompt
|
||||
- Changed agent markdown files to reject agent names containing `:`, which is reserved for plugin namespacing
|
||||
- Changed skills with `context: fork` to run in the background by default; opt out per skill with `background: false`
|
||||
- Added `yes`/`no`/`on`/`off`/`1`/`0` (case-insensitive) as accepted values for skill and plugin frontmatter booleans, alongside `true`/`false`
|
||||
- Fixed remote sessions continuing to send heartbeats after their worker was replaced, which left long-lived desktop and IDE processes retrying a rejected request every few seconds forever
|
||||
|
||||
## 2.1.217
|
||||
|
||||
- Added emoji shortcode autocomplete in the prompt input: type `:heart:` to insert ❤️, or `:hea` for suggestions — disable with the `emojiCompletionEnabled` setting
|
||||
- Added warnings when transcript writes are failing (e.g. disk full) or when session saving is off due to an inherited environment variable, instead of losing transcripts silently
|
||||
- Fixed a memory leak where truncated MCP tool outputs kept the full untruncated result in memory for the rest of the session
|
||||
- Fixed Windows auto-update failures that could leave `claude.exe` missing; failed updates now restore the preserved executable automatically
|
||||
- Fixed background session isolation not canonicalizing symlinked working directories, which could let sessions escape their workspace folder
|
||||
- Fixed auto-compact never triggering for Claude Opus 4.8 on Bedrock and `/compact` failing once over the limit
|
||||
- Fixed corporate mTLS, TLS-verify, OAuth scope, and proxy settings being ignored in Claude Desktop sessions
|
||||
- Fixed screen reader mode's startup announcement being cut off by the first prompt render, and the thinking status row re-rendering every few seconds to update elapsed time and token counts
|
||||
- Fixed managed settings that set `OTEL_EXPORTER_OTLP_ENDPOINT` not governing all signals — lower-scope signal-specific overrides no longer redirect telemetry away from the managed endpoint
|
||||
- Fixed `--resume`/`--continue` and `/resume` failing with a TypeError when a transcript has a malformed attachment entry
|
||||
- Fixed Remote Control sessions not showing a pending permission prompt or dialog to viewers that connected after it appeared
|
||||
- Fixed background shells sometimes becoming impossible to stop after a session is sent to the background (`/background` or `←`) or when the session exits on a heavily loaded machine, most visible on Windows
|
||||
- Fixed a `CLAUDE.md` or `SKILL.md` paths frontmatter value with many brace groups OOM-killing or stalling the CLI at startup — brace expansion is now budget-bounded
|
||||
- Fixed the transcript preview sitting flush against the input area when attaching to a starting background session; it now leaves the same one-line gap as the live layout, so the transcript no longer shifts when the session takes over
|
||||
- Improved footer PR badge links to be clickable hyperlinks even when terminal support can't be detected (e.g. over ssh/tmux); set `FORCE_HYPERLINK=0` to opt out
|
||||
- Changed the login-expiry warning to appear 3 days before expiry instead of 5
|
||||
- Capped the frontend-design plugin suggestion tip at 3 lifetime impressions instead of repeating indefinitely
|
||||
- Added a cap on concurrently-running subagents (default 20, override with `CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS`) so one message can't fan out unbounded background agents
|
||||
- Changed subagents to no longer spawn nested subagents by default; set `CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH` to allow deeper nesting
|
||||
- Fixed `--max-budget-usd` not stopping background subagents: once the cap is reached, new spawns are denied and running background agents are halted
|
||||
|
||||
## 2.1.216
|
||||
|
||||
- Added `sandbox.filesystem.disabled` setting to skip filesystem isolation while keeping network egress control
|
||||
- Fixed a slowdown in long sessions where message normalization cost grew quadratically with the number of turns, causing multi-second stalls and slow resumes
|
||||
- Fixed auto mode denying commands with "HTTP 401" classifier errors after the OAuth token expired or rotated mid-session
|
||||
- Fixed AskUserQuestion telling Claude to continue even when your answer asked it to wait or explain first — free-text answers now get neutral wording
|
||||
- Fixed Claude Code on the web re-asking the same question and dropping your answer after the session sat idle for a few minutes
|
||||
- Fixed @-mentions silently attaching nothing after file-modifying hooks, vim dot-repeat of `c`-operators and paste, statusline running twice on resume, and resume-picker hangs on failure
|
||||
- Fixed resumed background agent sessions reverting to the default agent: the agent's prompt and tool restrictions are now restored
|
||||
- Fixed worktree-isolated subagents redirecting git into the shared checkout via `git -C`, `--git-dir`, or `GIT_DIR`/`GIT_WORK_TREE`
|
||||
- Fixed worktree sessions landing in another project's leftover worktree when the working directory did not match the selected project
|
||||
- Fixed background sessions whose worktree has no git repository being undeletable
|
||||
- Fixed `claude daemon stop --any` potentially terminating an unrelated process via a stale legacy daemon lockfile
|
||||
- Fixed Esc-Esc at an idle prompt not opening the rewind picker in long-running sessions with background tasks
|
||||
- Fixed Bash command permission checking for compound statements with redirects inside `&&` lists or negations
|
||||
- Fixed pressing Ctrl+X twice in the agent list failing to delete a session, and deleted sessions reappearing when their background worker had died
|
||||
- Fixed background subagents getting cancelled when a high-priority message arrives during their startup window
|
||||
- Fixed mouse and focus garbage in the terminal while a GUI editor from `/memory`, `/plan`, `/keybindings`, or Ctrl+G is open; `/memory` no longer waits for the editor to close
|
||||
- Fixed Claude-in-Chrome 403-looping on reconnect when the session's OAuth token lacks a required scope
|
||||
- Fixed workflow saves and scheduled-task writes following a symlink at `.claude`, which could redirect writes outside the project
|
||||
- Fixed MCP re-authenticate revoking working credentials before the new sign-in succeeds, and the reconnect needs-auth message in background sessions pointing at an unusable command
|
||||
- Fixed read-only commands on Windows accessing network paths without a permission prompt
|
||||
- Fixed Bash command parsing of non-ASCII characters to match real shell word boundaries
|
||||
- Fixed PowerShell tool permission validation of commands containing invisible Unicode characters
|
||||
- Fixed dialogs in fullscreen mode stretching past the right-hand edge of their panel
|
||||
- Fixed the `/config` settings list in fullscreen mode clipping its keyboard-hint footer
|
||||
- Fixed the transcript-mode (Ctrl+O) footer hint wrapping on terminals narrower than 104 columns
|
||||
- Fixed the Prometheus metrics endpoint (`OTEL_METRICS_EXPORTER=prometheus`) emitting invalid `# UNIT` lines
|
||||
- Fixed skills and commands changed during a session not appearing in the slash menu until restart
|
||||
- Fixed plugin skills with a `name` frontmatter field losing their plugin prefix in slash-command autocomplete
|
||||
- Fixed telemetry misreporting permission denials: failed permission-prompt requests no longer count as user rejections, and user interrupts are now reported as user aborts instead of rejections
|
||||
- Improved the `/fork` confirmation to one line with the new session's name, `claude attach` id, and a note when the copy shares your checkout
|
||||
- Improved validation of `git` and `gh` command arguments in the PowerShell tool
|
||||
- Improved the `/ultrareview` diff-too-large error to show configured limits, measured diff size, and largest contributing files
|
||||
- Improved `/code-review ultra` empty-diff message to name the exact base ref and suggest passing an explicit base
|
||||
- Improved the spend limit adjustment prompt to show the server's reason when a spend limit change is rejected
|
||||
- `/context` now shows an explicit warning when the conversation exceeds the context window, and a failed `/compact` displays as an error
|
||||
- `/rewind` no longer restores or deletes files through symlinks or hard links at tracked paths and reports how many paths it skipped
|
||||
- Background sessions: `/mcp` and `/install-github-app` now park a "needs input" request in the agent view when no client is attached
|
||||
- Updated the bundled dataviz skill: reordered the default chart palette and fixed guidance that suggested direct labels for four-series charts
|
||||
- [VSCode] Fixed right-to-left text (Arabic, Hebrew, Persian) rendering in the wrong order when mixed with English or code
|
||||
- Fixed cloud sessions dropping the in-flight message when the session's container restarts mid-turn — the interrupted turn now re-runs on resume instead of leaving the session unresponsive
|
||||
|
||||
## 2.1.215
|
||||
|
||||
- Claude no longer runs the `/verify` and `/code-review` skills on its own; invoke them with `/verify` or `/code-review` when you want them
|
||||
|
||||
## 2.1.214
|
||||
|
||||
- Fixed single-segment `dir/**` allow rules like `Edit(src/**)` auto-approving writes to nested `dir/` directories anywhere in the tree instead of only `<cwd>/dir`
|
||||
- Fixed a permission-check bypass affecting commands run in Windows PowerShell 5.1 sessions
|
||||
- Fixed Bash permission checks to fail closed on file-descriptor redirect forms that bash parses differently than the permission analyzer
|
||||
- Fixed Bash permission checks misjudging very long commands — commands over 10,000 characters now always prompt instead of running automatically
|
||||
- Fixed Bash permission checks treating zsh variable subscripts and modifiers in `[[ ]]` comparisons as inert text — these commands now prompt for approval
|
||||
- Fixed Bash permission checks to no longer auto-approve certain `help` and `man` commands that could run unsafe options, command substitutions, or backslash paths
|
||||
- Fixed permission prompts on remote sessions that could proceed before the local confirmation dialog
|
||||
- Added the EndConversation tool: Claude can end sessions with highly abusive users or jailbreak attempts, as on claude.ai since 2025 — see https://www.anthropic.com/research/end-subset-conversations
|
||||
- Added a periodic progress heartbeat for long-running tool calls that previously went silent
|
||||
- Added an ISO `modified` timestamp to memory file frontmatter
|
||||
- Added `message.uuid`, `client_request_id`, and `tool_source` attributes to OpenTelemetry log events for message-level correlation and tool provenance
|
||||
- Added `CLAUDE_CODE_OTEL_CONTENT_MAX_LENGTH` to configure the 60 KB truncation limit on OpenTelemetry content attributes
|
||||
- Added reasoning effort to the `subagentStatusLine` payload, so custom agent rows can render model and effort
|
||||
- Added permission prompts for `docker` commands (including the Podman `docker` shim) carrying daemon-redirect flags (`--url`, `--connection`, `--identity`, and Podman's remote mode) that previously ran without one
|
||||
- Fixed a crash when a GrowthBook feature evaluates to null, and a bug where a malformed flag payload could wipe the cached feature flags
|
||||
- Fixed Bash tool killing the Claude session when a `pkill -f` pattern accidentally matched the CLI's own process (Linux)
|
||||
- Fixed unbounded memory growth when `--settings` points at a device file or multi-GB file; oversized (>2 MiB) settings files now fail at startup with a clear error
|
||||
- Fixed streaming turns failing with "Socket is closed" behind corporate proxies on Windows
|
||||
- Fixed stream-json output truncation at exit for slow-reading SDK/pipeline consumers; the exit drain now scales with queued bytes instead of a flat 2s cap
|
||||
- Fixed scheduled tasks refusing their own configured prompt as untrusted input — the fired prompt is now delivered as the session's assigned task
|
||||
- Fixed PowerShell tool commands hanging until timeout when a child process waited on standard input (Windows)
|
||||
- Fixed Python scripts under the PowerShell tool crashing with UnicodeDecodeError when reading non-UTF-8 data from standard input (Windows)
|
||||
- Fixed Python scripts run via the PowerShell tool crashing with UnicodeEncodeError on non-ASCII output, and PowerShell 7 error messages containing raw ANSI escape sequences (Windows)
|
||||
- Fixed the PowerShell tool reporting `where.exe`, `fc.exe`, and `diff.exe` as errors when they return a valid negative answer (Windows)
|
||||
- Fixed `>` and `>>` under the PowerShell tool on Windows PowerShell 5.1 writing UTF-16LE files that other tools couldn't read as UTF-8
|
||||
- Fixed a displaced background daemon deleting its successor's control socket on shutdown, which made the next client kill the healthy replacement daemon
|
||||
- Fixed background sessions parked with `←` or `/background` and left idle keeping the background daemon and a worker process alive indefinitely
|
||||
- Fixed completed background sessions being impossible to remove via `claude rm` or the agent view once the background service had gone idle
|
||||
- Fixed background sessions dispatched from a non-git folder being impossible to delete from the agents view
|
||||
- Fixed reopening a stopped background session failing to restore its saved conversation when an unreadable folder exists in the session store
|
||||
- Fixed the Remote Control "session ready" push notification firing for sessions where Remote Control was not explicitly enabled
|
||||
- Fixed `/install-github-app` and the `/mcp` settings menu being blocked in agent-view sessions — they're now refused only in background sessions with no terminal attached
|
||||
- Fixed plugins enabled via the `--settings` CLI flag not loading (regression since v2.1.181)
|
||||
- Fixed feature flags going stale in long-running sessions after the OAuth token rotates
|
||||
- Fixed `/ultrareview` refusing to run in repos with no merge base — it now offers to review all tracked files
|
||||
- Fixed `claude update` and `claude doctor` hanging silently, and the `/status` System diagnostics section going blank, when a shell-config path is a directory
|
||||
- Fixed memory frontmatter values being silently truncated at an inline `#` when memory files are saved
|
||||
- Fixed session cost and token telemetry double-counting on streams that emit multiple cumulative `message_delta` frames
|
||||
- Fixed a spurious "check your network" warning that appeared while the advisor was thinking
|
||||
- Fixed hooks with exit code 2 not blocking as documented when the hook's stdout JSON fails schema validation
|
||||
- Fixed OTel log events emitted outside the turn's async context missing the interaction span's trace context
|
||||
- Fixed MCP transient errors during prompts/resources refresh clearing the server's slash commands and resources
|
||||
- Improved the `claude rc` workspace-trust error in the home directory to say trust there is never saved and to suggest running from a project directory
|
||||
- Changed single-segment `dir/**` hook `if:` conditions to match only `<cwd>/dir`; write `**/dir/**` for any-depth matching. `deny`/`ask` permission rules keep their any-depth match.
|
||||
- Changed `file` commands using `-m`/`--magic-file` or `-f`/`--files-from` to require permission instead of being auto-allowed as read-only
|
||||
- Changed keep-alive connection pooling to disable after a stale-connection error, so retries open a fresh socket
|
||||
- Changed SessionStart hooks to report source `"fork"` when a session begins as a fork instead of `"resume"`
|
||||
|
||||
## 2.1.212
|
||||
|
||||
- `/fork` now copies your conversation into a new background session (its own row in `claude agents`) while you keep working; the in-session subagent it used to launch is now `/subtask`
|
||||
- Added `claude auto-mode reset` to restore the default auto-mode configuration, with a confirmation prompt (pass `--yes` to skip)
|
||||
- Added a session-wide limit on WebSearch tool calls (default 200, tunable via `CLAUDE_CODE_MAX_WEB_SEARCHES_PER_SESSION`) to stop runaway search loops
|
||||
- Added a per-session cap on subagent spawns (default 200, override with `CLAUDE_CODE_MAX_SUBAGENTS_PER_SESSION`) to stop runaway delegation loops; `/clear` resets the budget
|
||||
- MCP tool calls running longer than 2 minutes now move to the background automatically so the session stays usable; configure the threshold or disable with `CLAUDE_CODE_MCP_AUTO_BACKGROUND_MS`
|
||||
- Typing `/resume` in the agent view now opens a picker of past sessions — including sessions deleted from the list — and resumes your pick as a background session
|
||||
- Fixed plan mode auto-running file-modifying Bash commands (e.g. `touch`, `rm`) without a permission prompt or SDK `canUseTool` callback
|
||||
- Fixed worktree creation following a repository-committed symlink at `.claude/worktrees`, which could create files outside the repository
|
||||
- Fixed a `continue:false` hook's halt being dropped when the tool fails or completes mid-stream, and hook infrastructure errors being misreported as user rejections
|
||||
- Fixed SIGTERM during a running Bash tool orphaning the command's process tree in print/SDK mode; the CLI now aborts the turn, kills the tree, and exits 143
|
||||
- Fixed `/background` and `claude --bg` failing with "EUNKNOWN: unknown error, uv_spawn" on Windows when Group Policy blocks PowerShell 5.1; the daemon now prefers PowerShell 7
|
||||
- Fixed shell mode (`!`) not executing commands containing file paths while the path autocomplete popup was open
|
||||
- Fixed auto-mode denial notifications rendering broken characters when a long denial reason was truncated mid-emoji
|
||||
- Fixed Ctrl+J not inserting a newline in the agent view dispatch input on terminals with extended key reporting, and surfaced the newline shortcut in the `?` help overlay
|
||||
- Fixed `/ultrareview` rejecting PR references like `#123`, `PR 123`, and pasted PR URLs; error hints now name the command you actually typed
|
||||
- Fixed `/ultrareview <branch>` not fetching the branch from origin when it exists remotely; it now suggests the closest branch name on typos
|
||||
- Fixed `/ultrareview` skipping the billing confirmation in a new conversation after `/clear`
|
||||
- Fixed `/ultrareview`'s "not a git repository" error on Claude Desktop now suggesting the project's repository folder instead of terminal commands
|
||||
- Fixed hosted (host-managed) sessions failing at startup when repository settings configured mTLS certs, extra CA bundles, or OAuth scopes; these transport settings are now ignored with a warning
|
||||
- Fixed a spurious "File has not been read yet" error when editing a file that had been read with offset/limit before resuming a session
|
||||
- Fixed `ExitWorktree` failing with "no active EnterWorktree session" after resuming a session with `--continue`/`--resume` in print/SDK mode
|
||||
- Fixed the workflow agent grid staying empty for Remote Control clients that join a session mid-run
|
||||
- Fixed streaming-mode control requests being marked complete before their handler finished, which could lose the request on session restart
|
||||
- Fixed background sessions created with `/fork` losing their live-parent protection after a state write failure
|
||||
- Fixed reopening a stopped background session from the agent view failing silently — it now resumes the session, or shows why it can't and lets you force a restart
|
||||
- Fixed agent teams: a stopping teammate could send the leader duplicate idle notifications when team initialization re-ran within a session
|
||||
- Fixed the plan-approval dialog footer splitting "ctrl+g to edit in <editor>" apart when the file path is long
|
||||
- Fixed the welcome banner keeping its old panel widths after a combined width+height terminal resize in fullscreen mode
|
||||
- Fixed diff previews losing their line numbers and +/- markers in narrow layouts
|
||||
- Fixed @-mentions attaching nothing after a partial file read, plugin uninstall targeting the wrong marketplace, and false "Command timed out" on exit code 143
|
||||
- Fixed OpenTelemetry HTTP exports being rejected with 411/400 by Azure Monitor and other endpoints that don't accept chunked transfer encoding
|
||||
- Fixed OTLP event log records missing `trace_id`/`span_id` when `TRACEPARENT` is set in SDK/headless mode
|
||||
- Fixed conversations with many images incorrectly failing with "Request too large" errors, and improved the error message to explain the actual cause
|
||||
- Fixed web search and web fetch returning "API Error" text as search results or page content when the API was overloaded
|
||||
- Improved web search and web fetch reliability by retrying 529 errors and rate-limited requests with bounded backoff
|
||||
- Improved prompt caching: the mid-conversation system block now works behind LLM gateways and custom base URLs (Bedrock, Vertex, 1P)
|
||||
- Improved background agent attach: cold-attaching now instantly shows the formatted transcript while the session boots, instead of a blank wait
|
||||
- Reduced token usage in inter-agent messaging: `SendMessage` bodies are no longer duplicated into replayed history and tool results
|
||||
- Changed `/fork` to name the copy after your prompt when the session has no title, so the row is recognizable in the agent view
|
||||
- Changed bare `/btw` to reopen the side-question panel on your most recent exchange so you can browse earlier answers
|
||||
- Changed the `←` footer hint to pulse `N done` for a moment when a background agent finishes while nothing needs your input
|
||||
- Deprecated the Task tool's `mode` parameter (now ignored); subagents inherit the parent session's permission mode by default
|
||||
- Changed Enterprise `forceLoginMethod` to be enforced for VS Code extension, SDK, `setup-token`, and `install-github-app` logins, not just the terminal
|
||||
- Changed session transcripts to record the reasoning effort level on each assistant message
|
||||
- Changed headless/SDK sessions to apply a `set_model` control request mid-turn; the next model round-trip uses the new model instead of waiting for the next turn
|
||||
- Changed agent view / `claude agents --json`: sessions waiting on a sandbox, MCP-input, or managed-settings prompt now show as "Needs input" instead of "Working"
|
||||
- Updated the auth status panel title from "Cloud authentication" to "Authentication"
|
||||
- Corrected an earlier release note (2.1.200): tmux through the 3.6 series lacks synchronized output; newer tmux with support is detected automatically
|
||||
|
||||
## 2.1.211
|
||||
|
||||
- Added `--forward-subagent-text` flag and `CLAUDE_CODE_FORWARD_SUBAGENT_TEXT` environment variable to include subagent text and thinking in stream-json output
|
||||
- Fixed permission previews relayed to chat channels not neutralizing bidirectional-override, zero-width, and look-alike quote characters, so tool inputs cannot visually alter the approval message
|
||||
- Fixed auto mode overriding a PreToolUse hook's `ask` decision for unsandboxed Bash — a hook `ask` now floors the decision at a prompt
|
||||
- Fixed parallel Claude Code sessions all logging out simultaneously after wake-from-sleep when many sessions share one credential store
|
||||
- Fixed plugin MCP servers not reconnecting after an idle web session woke, leaving MCP calls failing until the next message
|
||||
- Fixed Claude Code on Vertex and Bedrock attempting the default Opus model at startup and printing a spurious fallback notice when a model is explicitly configured
|
||||
- Fixed subagents spawned with an explicit model override reverting to the parent's model when resumed or sent a follow-up message
|
||||
- Fixed nested `.claude/rules/*.md` files loading even when setting sources exclude project settings
|
||||
- Fixed file upload validation: filenames ending in a DOS device suffix (`.prn`) or trailing dot are now accepted, and files with multiple hard links are refused
|
||||
- Fixed file uploads to Claude in Chrome from remote and CLI sessions
|
||||
- Fixed edits that leave the input as "?" being silently swallowed and toggling the shortcuts panel
|
||||
- Fixed a startup hang when the Claude in Chrome extension is enabled but Chrome is not running
|
||||
- Fixed a 300ms delay revealing async content (Settings tabs, Stats, diff views, and other loading states)
|
||||
- Fixed reopening a just-stopped background session from the agents view starting a blank conversation under the same session id
|
||||
- Fixed `/loop` hiding the session from `/resume` after a single use
|
||||
- Fixed screen reader users losing the audible terminal bell after `/terminal-setup` or onboarding terminal setup
|
||||
- Fixed background jobs on LLM gateway auth (`ANTHROPIC_AUTH_TOKEN` + `ANTHROPIC_BASE_URL`) coming back "Not logged in" after the daemon respawns them
|
||||
- Fixed `claude agents` jobs becoming permanently undeletable when git no longer recognizes their worktree — the row now shows why the delete was refused instead of silently reappearing
|
||||
- Fixed `/clear` not resetting the session cost counter — the statusline's cost now starts at $0 after `/clear`
|
||||
- Fixed Claude in Chrome setup pages failing to open in the browser on Windows
|
||||
- Fixed headless print-mode sessions on Windows crashing or silently exiting when stdin is unreadable
|
||||
- Fixed background session titles in the agents view showing the naming model's refusal text when the prompt contains a link
|
||||
- Fixed background agents killed by the user auto-respawning, and revived agents re-running stale prompts from old sessions
|
||||
- Fixed routines with no schedule reporting a next run time in the year 1
|
||||
- Hardened synced skill/plugin directory naming on Windows and kept CCR web fetch/search proxies working after `/clear`
|
||||
- Improved terminal layout and rendering performance
|
||||
- Improved background agent result reporting — Claude now reports the status of still-running agents and waits for the real completion instead of fabricating results
|
||||
- Improved the memory index over-limit warning to measure only loaded content, excluding frontmatter and HTML comments
|
||||
- Updated integer environment variables (timeouts, token budgets, retry counts) to accept scientific notation and digit-separator spellings like `1e6` and `64_000`
|
||||
- Updated documentation links to the current docs sites
|
||||
- Changed "always allow" permission rules to save at the repository root, so approvals granted in a git worktree persist across sessions and worktrees
|
||||
- Changed `/usage-credits` to ask for confirmation before sending a request to organization admins
|
||||
- Changed Vim mode `s` and `S` (substitute char/line) to work in NORMAL mode, matching vim behavior
|
||||
- [VSCode] Updated the Remote Control banner to describe what it does
|
||||
- Claude in Chrome: hardened file-upload path validation
|
||||
- Claude in Chrome: `save_to_disk` on screenshot actions now writes the image to disk and returns the path; previously it did nothing
|
||||
- Fixed a prompt-caching regression on Bedrock, Vertex, Mantle, and Foundry that billed the trailing system context block as fresh input tokens on every request.
|
||||
|
||||
## 2.1.210
|
||||
|
||||
- Added a live elapsed-time counter to the collapsed tool summary line so long-running tool calls visibly tick instead of looking stuck
|
||||
- Added a startup warning for `Write(path)`, `NotebookEdit(path)`, and `Glob(path)` permission rules — use `Edit(path)` or `Read(path)` instead
|
||||
- Fixed `isolation: 'worktree'` subagents being able to run git-mutating commands against the main repo checkout instead of their own isolated worktree
|
||||
- Fixed the `ultracode` keyword opt-in firing on non-human-originated input such as webhook payloads and relayed PR comments
|
||||
- Fixed a rendered text fragment leaking into crash telemetry when a UI component returned content outside a styled text element
|
||||
- Fixed paste markers leaking into external editors opened from Claude Code, which could appear as stray È/É characters around pasted text
|
||||
- Fixed `claude attach` sometimes failing with "job not found" or "agent is still starting" errors during session transitions — attach now waits for the daemon to settle, and terminal resizes during a slow attach are applied once it completes
|
||||
- Fixed a session crash when a tool's result renderer returned a numeric bigint value or plain text instead of a UI element
|
||||
- Fixed a hook callback timeout being misreported to the model as a user rejection, which made unattended sessions stop and wait
|
||||
- Fixed Claude assuming a `cd` took effect after its command was moved to the background; the tool result now states the working directory is unchanged
|
||||
- Fixed plugin-provided MCP servers being torn down when MCP servers are re-synced mid-session
|
||||
- Fixed plan approvals without edits being labeled "(edited by user)" and overwriting the plan file with a stale snapshot
|
||||
- Fixed `/doctor` skipping its auto-mode-default proposal on Bedrock, Vertex, and Foundry, where auto mode no longer needs an opt-in
|
||||
- Fixed Grep content mode claiming "No matches found" when paginating past the end of results
|
||||
- Fixed unmatched `$1`/`$2` positional placeholders in skills and commands being silently stripped; they are now preserved verbatim
|
||||
- Fixed plugin cache writes leaving temp files behind on failure and failing on locked-file renames on Windows and network filesystems
|
||||
- Fixed background workers crash-looping when a client resets its connection to the background service
|
||||
- Fixed `claude agents --effort ultracode` not reaching dispatched sessions; the value was silently dropped
|
||||
- Fixed pressing ← to open the agents view dropping the task tracker when returning to the session
|
||||
- Fixed the agents dashboard retaining pasted images from abandoned reply drafts after their session was deleted
|
||||
- Fixed killed background sessions leaving a permanent `git worktree lock` behind; the periodic sweep now releases locks whose owning process is gone
|
||||
- Fixed SDK MCP servers registered via an `initialize` control request waiting until the next turn to start connecting
|
||||
- Fixed returning to the agents view from a session leaving overlapping ghost frames with `CLAUDE_CODE_DISABLE_ALTERNATE_SCREEN=1`
|
||||
- Fixed late-appearing `.claude/*` symlinks not being reconciled into the sandbox deny-write list
|
||||
- Hardened the Agent tool against indirect prompt injection via content a subagent read
|
||||
- Improved the Bash/PowerShell tool message when a command hits its timeout and is auto-backgrounded, so the model can distinguish a hang from an explicit background request
|
||||
- Improved auto mode: the permission classifier now defaults to Sonnet 5 for external sessions, validated on the session's first request and pinned for the session
|
||||
- Improved the bundled dataviz skill's chart color validation with perceptual OKLab color difference and recalibrated color-blindness thresholds
|
||||
- Memory writes that leave a MEMORY.md index over its read limit now produce an explicit error instead of silent truncation
|
||||
- Screen reader mode now announces permission mode changes aloud when cycling modes with Shift+Tab
|
||||
- The agents footer hint now shows how many background agents are waiting on your input, with a brief color emphasis when the count changes
|
||||
- Agent view: the session you pressed ← from stays visibly marked even after mouse hover or arrow keys move the selection
|
||||
- Fable temporarily shows as unavailable in the advisor picker while a server-side issue causing Fable advisor failures is fixed
|
||||
|
||||
## 2.1.209
|
||||
|
||||
- Fixed /model and other dialogs being blocked in `claude agents` background sessions (reverts an overly broad guard)
|
||||
|
||||
## 2.1.208
|
||||
|
||||
- Added screen reader mode: opt-in plain-text rendering for screen reader users. Run `claude --ax-screen-reader`, set CLAUDE_AX_SCREEN_READER=1, or add "axScreenReader": true to settings.
|
||||
- Added `vimInsertModeRemaps` setting: map two-key insert-mode sequences like `jj` to Escape in vim mode
|
||||
- Added `CLAUDE_CODE_PROCESS_WRAPPER`: agent view and the background service now honor a corporate launcher by running every Claude Code self-spawn through a required wrapper executable
|
||||
- Added mouse-click support for multi-select menus and "Other" input rows in fullscreen mode
|
||||
- Changed the Fable 5 usage-credits consent prompt to start with the decline option focused
|
||||
- Fixed fast mode staying off after switching back to a model that supports it — it now restores automatically when enabled in settings
|
||||
- Fixed replies typed to a background agent being lost when delivery fails — the text is now saved and delivered when the session restarts
|
||||
- Fixed background-session attach failing permanently ("Couldn't start the background daemon") after an update replaced the binary a running `claude agents` process was launched from
|
||||
- Fixed the context window (and auto-compact indicator) briefly resetting to 200k after the CLI auto-updates, causing a false "100% context used" when resuming long-context sessions
|
||||
- Fixed supervised and background sessions crashing when a server closed an HTTP/2 connection with a GOAWAY while requests were in flight
|
||||
- Fixed truncated stream-json/JSON output and missing result message when piping large responses from `claude -p`
|
||||
- Fixed `CLAUDE_CODE_MAX_OUTPUT_TOKENS` and similar env vars silently using the mantissa of scientific-notation values (`1e6` became `1`)
|
||||
- Fixed very large markdown tables stalling rendering or using excessive memory; tables over 200 rows show the first 200 with a "… N more rows" notice
|
||||
- Fixed the Edit tool failing on files modified after reading when the target text still matches uniquely
|
||||
- Fixed Read reporting empty files as "shorter than offset", Grep silently returning "No files found" for invalid regex patterns, Grep count mode under-reporting totals when paginated, and Glob crashing with an unclear error when the pattern, path, or working directory contained a null byte
|
||||
- Fixed `apiKeyHelper` script failures being hidden behind a generic 401 after ~10 silent retries; the script's own error is now shown within 3 attempts
|
||||
- Fixed Bedrock streaming requests failing with a misleading "Truncated event message received" when a gateway transforms the response — the error now names the content-type and points at the proxy
|
||||
- Fixed `/upgrade` showing a login flow instead of the upgrade URL when the browser fails to open
|
||||
- Fixed stream-json input killing the session on blank CRLF or whitespace-only lines from Windows-style SDK hosts
|
||||
- Fixed headless stream-json sessions hanging permanently when a `control_request` carried a non-string `set_model` payload; the CLI now answers with an error response
|
||||
- Fixed repeated "No completion record was found" notices on session resume — orphaned background tasks now collapse into a single summary
|
||||
- Fixed Remote Control clients attaching to a terminal-hosted session not seeing background agents and workflow progress until a task started or stopped
|
||||
- Fixed the Agent tool launching with no tools when a subagent's `tools` list resolves to nothing — it now returns a clear error naming the unrecognized entries
|
||||
- Fixed `/usage` showing stale cached bars over fresher data, and `/mcp` not reclassifying placeholder servers after config edits
|
||||
- Fixed "Change directory" in SDK hosts (e.g. Claude Desktop) failing with "A turn is in progress" on idle sessions that have a running background task
|
||||
- Fixed the workflow save dialog showing `~/.claude/workflows/` instead of the `CLAUDE_CONFIG_DIR` location for user-scope saves
|
||||
- Fixed `/release-notes` adding the viewed notes to the model's context — "Show all" previously injected the entire changelog into every subsequent request
|
||||
- Fixed a memory leak in the agent view where pasted images were retained for the screen's lifetime after sending peek replies
|
||||
- Fixed SDK sessions losing agents defined via the initialize request when a plugin refresh ran before the client attached
|
||||
- Fixed several memory leaks in long sessions: MCP stdio server stderr accumulating up to 64 MB per server, LSP documents staying open indefinitely (now LRU with 50-doc cap), async hook output retained after backgrounding, and unbounded growth in headless/SDK sessions from large tool-result payloads
|
||||
- Fixed a memory blowup when reading files with extremely long single lines using offset/limit — the read now returns a clean error instead of loading the whole line
|
||||
- Fixed multi-second per-turn slowdowns in sessions with many permission deny/ask rules — rule matchers are now compiled once and cached
|
||||
- Improved input responsiveness while agent task lists update — task updates no longer re-render the entire UI
|
||||
- Reduced per-tool-call CPU overhead in print/SDK sessions with many MCP tools by caching tool-pool assembly (up to 7x faster tool rounds at high tool counts)
|
||||
- Reduced memory usage by bounding the file edit read cache to 16 MB instead of pinning up to 1,000 full files
|
||||
- Reduced session transcript size (up to 79x in edit-heavy sessions) and bounded checkpoint disk usage by pruning superseded file-history backups
|
||||
- Reduced memory usage when resuming sessions with background agents or forks spawned from large conversations
|
||||
- Completed background agents now stay listed in `/tasks` until cleanup instead of vanishing the moment they finish
|
||||
- Attaching to a stopped background agent now shows its transcript immediately while the session warms up, instead of a blank "Session is starting" screen
|
||||
- Background sessions: an older daemon no longer silently restarts workers spawned by a newer version onto the older binary
|
||||
- Agent view: Ctrl+X now deletes renamed-branch worktrees, never destroys unpushed commits, keeps the session row when a worktree is kept, and reused worktree names reset to the current base
|
||||
- Catastrophic removals (e.g. `rm -rf ~`) in commands containing `$(…)`/backticks/`<(…)` now prompt in `--dangerously-skip-permissions` and auto mode, matching the plain form
|
||||
- `/install-github-app` and the `/mcp` settings menu no longer open in background sessions
|
||||
- MCP servers configured with an empty URL now show as "not configured" in `/mcp` instead of a config error
|
||||
- `/usage` now shows your last-known usage bars with an "as of" note when the usage endpoint is rate-limited, instead of an error screen
|
||||
- Fixed Bedrock auth failing with "Session token not found or invalid" for AWS SSO profiles whose sso_region differs from the Bedrock region (2.1.207 regression)
|
||||
|
||||
## 2.1.207
|
||||
|
||||
- Auto mode is now available without `CLAUDE_CODE_ENABLE_AUTO_MODE` opt-in on Bedrock, Vertex AI, and Foundry; disable via `disableAutoMode` in settings
|
||||
- Fixed the terminal freezing and keystrokes lagging while streaming responses containing very long lists, tables, paragraphs, or code blocks
|
||||
- Fixed remote managed settings from a non-interactive run (`claude -p`, the SDK) being permanently recorded as consented without ever showing the security consent dialog
|
||||
- Fixed spurious prompt-injection warnings triggered by benign system-generated conversation updates
|
||||
- Fixed the auto-updater overwriting a custom launcher script or symlink at `~/.local/bin/claude` on every release; `/doctor` now reports an externally managed launcher
|
||||
- Fixed compound commands with `cd` prompting for permission when the only output redirect was to `/dev/null`
|
||||
- Fixed the transcript jumping above the start of the answer when a response finishes streaming
|
||||
- Fixed `extensions.worktreeConfig` being left in the repo's `.git/config` (breaking go-git tools like `tea`) after the last `worktree.sparsePaths` worktree was removed
|
||||
- Fixed malformed bracket patterns in rules globs, skill paths, `.ignore`, and `.worktreeinclude` breaking file reads, file suggestions, and worktree creation
|
||||
- Fixed a crash loop in agent teams where a malformed teammate mailbox message caused repeated errors every second until the mailbox file was manually deleted
|
||||
- Fixed background sessions auto-named by accepting a plan not showing that name on their agent-view row
|
||||
- Fixed background sessions that entered a git worktree resuming blank after a cold reopen from the agent list
|
||||
- Fixed Remote Control task status updates being lost when the connection recovered from a network interruption or credential refresh
|
||||
- Fixed Remote Control sessions hosted by the desktop app not showing background agent and workflow progress on mobile and web
|
||||
- Fixed Deep research runs labeling every Fetch-phase agent "unknown" — chips now show the source hostname
|
||||
- Fixed Bedrock repeatedly requesting fresh AWS SSO credentials from IAM Identity Center on every API request
|
||||
- Improved agent view: pasting the same text again now expands the collapsed `[Pasted text #N]` placeholder instead of adding a second one
|
||||
- Improved agent view: blocked session peeks now lead with the question and show a worded staleness clock (`waiting 3m`) instead of the same timestamp twice
|
||||
- Changed Bedrock, Vertex, and Claude Platform on AWS to default to Claude Opus 4.8
|
||||
- Changed auto mode to no longer read `autoMode` from `.claude/settings.local.json` (repo-resident); use `~/.claude/settings.json` instead
|
||||
- Fixed an indefinite hang on Windows when AWS credential resolution stalls (e.g. a stuck `credential_process`): the 60-second stall guard now fires instead of waiting forever.
|
||||
- Plugin hooks/monitors/MCP headersHelper: `${user_config.*}` in shell-form commands is now rejected (shell-injection fix). Hooks: use exec form (`args` array) or `$CLAUDE_PLUGIN_OPTION_<KEY>`; monitors and headersHelper: read the value inside the script (config file or the server's `env` block).
|
||||
- Plugin option values (`pluginConfigs`) are no longer read from project-level `.claude/settings.json`; only user, `--settings`, and managed settings are honored
|
||||
- Fixed `/usage-credits` amount inputs silently stripping malformed values (e.g. a pasted timestamp) to digits; malformed amounts are now rejected with an error, and amounts over $1,000 require a typed confirmation
|
||||
|
||||
## 2.1.206
|
||||
|
||||
- Added directory path suggestions to `/cd`, matching `/add-dir` behavior
|
||||
@@ -557,7 +1294,6 @@
|
||||
|
||||
## 2.1.169
|
||||
|
||||
- Self-hosted runner: added a `post-session` lifecycle hook that runs after the session ends and before the workspace is deleted, so you can snapshot uncommitted work or export logs; also made the child-process SIGTERM→SIGKILL window configurable (default unchanged at 5s)
|
||||
- Added `--safe-mode` flag (and `CLAUDE_CODE_SAFE_MODE`) to start Claude Code with all customizations (CLAUDE.md, plugins, skills, hooks, MCP servers) disabled for troubleshooting
|
||||
- Added `/cd` command to move a session to a new working directory without breaking the prompt cache mid-session
|
||||
- Added a `disableBundledSkills` setting and `CLAUDE_CODE_DISABLE_BUNDLED_SKILLS` environment variable to hide bundled skills, workflows, and built-in slash commands from the model
|
||||
@@ -3404,7 +4140,7 @@
|
||||
|
||||
## 2.1.15
|
||||
|
||||
- Added deprecation notification for npm installations - run `claude install` or see https://docs.anthropic.com/en/docs/claude-code/getting-started for more options
|
||||
- Added deprecation notification for npm installations - run `claude install` or see https://code.claude.com/docs/en/setup for more options
|
||||
- Improved UI rendering performance with React Compiler
|
||||
- Fixed the "Context left until auto-compact" warning not disappearing after running `/compact`
|
||||
- Fixed MCP stdio server timeout not killing child process, which could cause UI freezes
|
||||
|
||||
19
examples/gateway/aws/.dockerignore
Normal file
19
examples/gateway/aws/.dockerignore
Normal file
@@ -0,0 +1,19 @@
|
||||
# Keep secrets and generated artifacts out of the build context. The Dockerfile
|
||||
# COPYs the binary, gateway.yaml (unlike the GCP example, the config is baked
|
||||
# into the image — ECS injects only the secrets it references, as env vars),
|
||||
# and the RDS CA bundle. BuildKit (the default builder) only syncs the
|
||||
# referenced COPY sources anyway, so this is a denylist for the classic
|
||||
# builder (DOCKER_BUILDKIT=0) and a conventional signal that the .gitignore'd
|
||||
# secrets in this directory aren't part of the image build.
|
||||
terraform/
|
||||
**/.terraform/
|
||||
*.tfstate*
|
||||
terraform.tfvars
|
||||
secrets/
|
||||
*.pem
|
||||
# The RDS CA bundle is public trust-anchor material (no secret), and the
|
||||
# Dockerfile COPYs it — carve it out of the *.pem exclusion above.
|
||||
!rds-global-bundle.pem
|
||||
*.iam.json
|
||||
claude.download
|
||||
claude.bad
|
||||
18
examples/gateway/aws/.gitignore
vendored
Normal file
18
examples/gateway/aws/.gitignore
vendored
Normal file
@@ -0,0 +1,18 @@
|
||||
# Local, environment-specific config — copy gateway.yaml.example -> gateway.yaml
|
||||
# (gateway.yaml.example IS committed; your filled-in gateway.yaml is not)
|
||||
gateway.yaml
|
||||
|
||||
# Secrets / credentials — never commit. Also covers rds-global-bundle.pem:
|
||||
# not a secret, but downloaded by setup.sh when absent (delete it to refresh
|
||||
# after an RDS CA rotation), so it stays out of git.
|
||||
secrets/
|
||||
*.pem
|
||||
|
||||
# Scratch IAM policy documents written by setup.sh (no secrets, but generated)
|
||||
*.iam.json
|
||||
|
||||
# Vendored release binary — download per release (see setup.sh DIST_URL).
|
||||
# claude.bad is a checksum-mismatched binary that setup.sh set aside.
|
||||
claude
|
||||
claude.download
|
||||
claude.bad
|
||||
67
examples/gateway/aws/Dockerfile
Normal file
67
examples/gateway/aws/Dockerfile
Normal file
@@ -0,0 +1,67 @@
|
||||
# Runtime image for `claude gateway`.
|
||||
#
|
||||
# This image does NOT build the binary. It expects a prebuilt native
|
||||
# linux-x64 `claude` executable in the build context — the Claude Code release
|
||||
# binary, which includes the `gateway` subcommand. setup.sh places it at
|
||||
# ./claude (downloading and checksum-verifying it via DIST_URL/DIST_SHA256 if
|
||||
# missing). Override CLAUDE_BINARY to point at a different path.
|
||||
#
|
||||
# Unlike the GCP example (which mounts the config from Secret Manager at
|
||||
# runtime), this image BAKES gateway.yaml in at /etc/claude/gateway.yaml — on
|
||||
# ECS the task definition injects only the secrets the YAML references, as env
|
||||
# vars. gateway.yaml therefore must be fully filled in (no REPLACE_ME) before
|
||||
# building; setup.sh enforces this. A config edit means a rebuild under a new
|
||||
# tag. The file contains no secret values — every credential resolves at boot
|
||||
# via ${ENV_VAR} expansion.
|
||||
#
|
||||
# The image also bakes in the AWS RDS CA bundle (rds-global-bundle.pem —
|
||||
# setup.sh downloads it from https://truststore.pki.rds.amazonaws.com before
|
||||
# the build) and trusts it via NODE_EXTRA_CA_CERTS, so the store connection
|
||||
# string's `?sslmode=verify-full` verifies the RDS server certificate chain
|
||||
# and hostname. NOTE the gateway's driver reads `sslmode` from the URL but NOT
|
||||
# a libpq-style `sslrootcert=` param — the CA must come from this env var.
|
||||
#
|
||||
# Build:
|
||||
# docker build --platform=linux/amd64 --provenance=false \
|
||||
# --build-arg CLAUDE_BINARY=./claude -t claude-gateway .
|
||||
#
|
||||
# (For Fargate on ARM64/Graviton: build --platform=linux/arm64 with the
|
||||
# linux-arm64 binary and set the task definition's cpuArchitecture to ARM64.)
|
||||
#
|
||||
# Run:
|
||||
# docker run --rm -p 8080:8080 \
|
||||
# -e OIDC_CLIENT_SECRET -e GATEWAY_JWT_SECRET -e GATEWAY_POSTGRES_URL \
|
||||
# claude-gateway
|
||||
|
||||
ARG CLAUDE_BINARY=./claude
|
||||
ARG GATEWAY_CONFIG=./gateway.yaml
|
||||
ARG RDS_CA_BUNDLE=./rds-global-bundle.pem
|
||||
|
||||
# distroless/cc provides glibc + libstdc++ (required by the Bun-compiled
|
||||
# native binary). The :nonroot tag runs as uid/gid 65532. Pinned by digest so
|
||||
# the build never silently takes new upstream bytes (the digest is the
|
||||
# multi-arch OCI index, so --platform still selects amd64/arm64). To refresh
|
||||
# the pin after reviewing upstream changes:
|
||||
# docker manifest inspect -v gcr.io/distroless/cc-debian12:nonroot # prints the index digest
|
||||
FROM gcr.io/distroless/cc-debian12:nonroot@sha256:ce0d66bc0f64aae46e6a03add867b07f42cc7b8799c949c2e898057b7f75a151
|
||||
|
||||
ARG CLAUDE_BINARY
|
||||
ARG GATEWAY_CONFIG
|
||||
ARG RDS_CA_BUNDLE
|
||||
COPY --chmod=0755 ${CLAUDE_BINARY} /usr/local/bin/claude
|
||||
# WORKDIR pre-creates /etc/claude with 0755 — without it, COPY --chmod would
|
||||
# also stamp the auto-created parent directory 0644 (no execute bit), making
|
||||
# the config unreadable for the nonroot user.
|
||||
WORKDIR /etc/claude
|
||||
COPY --chmod=0644 ${GATEWAY_CONFIG} /etc/claude/gateway.yaml
|
||||
COPY --chmod=0644 ${RDS_CA_BUNDLE} /etc/claude/rds-global-bundle.pem
|
||||
WORKDIR /
|
||||
|
||||
ENV CLAUDE_CONFIG_DIR=/tmp/.claude
|
||||
# Trust anchor for the store's sslmode=verify-full (see header comment).
|
||||
ENV NODE_EXTRA_CA_CERTS=/etc/claude/rds-global-bundle.pem
|
||||
|
||||
EXPOSE 8080
|
||||
USER nonroot
|
||||
|
||||
ENTRYPOINT ["/usr/local/bin/claude", "gateway", "--config", "/etc/claude/gateway.yaml"]
|
||||
19
examples/gateway/aws/README.md
Normal file
19
examples/gateway/aws/README.md
Normal file
@@ -0,0 +1,19 @@
|
||||
# Claude apps gateway on AWS
|
||||
|
||||
Reference deployment artifacts for running Claude apps gateway on AWS with
|
||||
Amazon Bedrock as the upstream: ECS on Fargate or EKS, Amazon RDS for
|
||||
PostgreSQL, AWS Secrets Manager, and IAM-role auth to Bedrock.
|
||||
|
||||
These files are provided as a working example rather than a supported production
|
||||
deployment. Adapt them to your own environment.
|
||||
|
||||
- **Walkthrough**: https://code.claude.com/docs/en/claude-apps-gateway-on-aws
|
||||
- **Related**: AWS-maintained samples for various customer environments at
|
||||
https://github.com/aws-samples/anthropic-on-aws/tree/main/claude-apps-gateway
|
||||
|
||||
| File | Purpose |
|
||||
|---|---|
|
||||
| `setup.sh` | Scripts the walkthrough end to end via the `aws` CLI |
|
||||
| `Dockerfile` | Runtime image for the `claude gateway` binary (bakes in `gateway.yaml`) |
|
||||
| `gateway.yaml.example` | Gateway config template, AWS-shaped (Bedrock upstream, Okta IdP) |
|
||||
| `terraform/` | Provisions the full architecture (two-pass apply — see `terraform/README.md`) |
|
||||
174
examples/gateway/aws/gateway.yaml.example
Normal file
174
examples/gateway/aws/gateway.yaml.example
Normal file
@@ -0,0 +1,174 @@
|
||||
# gateway.yaml.example — Claude apps gateway config template, AWS-shaped (walkthrough §4).
|
||||
#
|
||||
# Okta IdP + Bedrock upstream, following the walkthrough at
|
||||
# https://code.claude.com/docs/en/claude-apps-gateway-on-aws. The active sections
|
||||
# below are a strict subset of the full configuration reference at
|
||||
# https://code.claude.com/docs/en/claude-apps-gateway-config; optional keys are
|
||||
# included commented-out.
|
||||
#
|
||||
# USAGE — this is the shippable TEMPLATE. Copy it to gateway.yaml and fill it in:
|
||||
# cp gateway.yaml.example gateway.yaml
|
||||
# setup.sh and terraform/ read gateway.yaml (your filled-in copy, which is
|
||||
# gitignored). Unlike the GCP example it is NOT published to a secret store:
|
||||
# the Dockerfile bakes it into the image at /etc/claude/gateway.yaml — the
|
||||
# container ENTRYPOINT runs `claude gateway --config /etc/claude/gateway.yaml`.
|
||||
# It holds no secret values; a config edit means an image rebuild (setup.sh
|
||||
# tags images with a hash of this file, so a re-run rebuilds automatically).
|
||||
#
|
||||
# Secret expansion: ${ENV_VAR} reads an env var; ${file:/path} reads a mounted file.
|
||||
# On ECS, the task definition injects the JWT / OIDC / Postgres secrets as ENV
|
||||
# VARS via its `secrets` field (valueFrom -> Secrets Manager ARN). On EKS you
|
||||
# may mount them as files instead and use ${file:/secrets/...}.
|
||||
#
|
||||
# BEFORE BUILD — replace every REPLACE_ME placeholder below (setup.sh refuses to
|
||||
# build the image while any remain — the config is baked in, so a half-filled
|
||||
# config would ship), and create the referenced secrets:
|
||||
# gateway-jwt-secret (setup.sh generates this)
|
||||
# gateway-oidc-client-secret (from the Okta admin console OIDC web app)
|
||||
# gateway-postgres-url (setup.sh generates this)
|
||||
|
||||
# ── Listener ─────────────────────────────────────────────────────────────────
|
||||
listen:
|
||||
host: 0.0.0.0
|
||||
port: 8080 # the target group forwards ALB :443 -> :8080
|
||||
# Required. Fixes the IdP redirect_uri, the OIDC discovery doc, and the
|
||||
# gateway-token issuer so none are derived from the client-controlled Host
|
||||
# header (X-Forwarded-Host/-Proto are likewise never trusted). Set it to the
|
||||
# internal hostname you picked in the prerequisites — the Route 53 private
|
||||
# zone name your ACM certificate covers (e.g.
|
||||
# https://claude-gateway.internal.example.com). Unlike Cloud Run there is no
|
||||
# first-deploy placeholder dance: you choose the hostname up front, alias it
|
||||
# to the internal ALB after the deploy, and register the same host's
|
||||
# /oauth/callback on the Okta app.
|
||||
public_url: REPLACE_ME
|
||||
# Register this exact redirect URI on the Okta OIDC web application:
|
||||
# https://<public_url host>/oauth/callback
|
||||
#
|
||||
# Behind the internal ALB every request arrives via the load balancer, so the
|
||||
# gateway sees ALB-node peer IPs for all developers — set trusted_proxies so
|
||||
# X-Forwarded-For from those proxies is trusted and per-IP rate limiting /
|
||||
# audit IPs record the real client. ALB nodes take addresses from the subnets
|
||||
# the ALB is attached to, so list those subnets' CIDRs (the private subnets
|
||||
# from the prerequisites).
|
||||
#
|
||||
# NOTE: listing the ALB subnets' CIDRs trusts every host in those subnets as a
|
||||
# proxy — any co-located workload that can reach the ALB can then spoof the
|
||||
# client IP via X-Forwarded-For (audit logs, per-IP rate limits, IP
|
||||
# allowlists). Keep the ALB :443 ingress source (CORP_CIDR / corporate_cidr)
|
||||
# from overlapping these subnets, and don't share the subnets with untrusted
|
||||
# workloads.
|
||||
trusted_proxies: [REPLACE_ME] # e.g. [10.0.1.0/24, 10.0.2.0/24]
|
||||
#
|
||||
# Alternative — terminate TLS in the gateway itself instead of at the ALB:
|
||||
# tls:
|
||||
# cert: /certs/gateway.crt
|
||||
# key: /certs/gateway.key
|
||||
|
||||
# ── Identity provider — Okta ─────────────────────────────────────────────────
|
||||
oidc:
|
||||
issuer: REPLACE_ME # e.g. https://example.okta.com (or your custom auth server URL)
|
||||
client_id: REPLACE_ME # Okta OIDC web app client ID (not secret)
|
||||
client_secret: ${OIDC_CLIENT_SECRET} # EKS file mounts: ${file:/secrets/oidc-client-secret}
|
||||
allowed_email_domains: [REPLACE_ME] # e.g. [example.com] — reject id_tokens outside your org
|
||||
# The Okta org authorization server returns a thin id_token that omits email
|
||||
# and groups; the gateway fills them from /userinfo.
|
||||
userinfo_fallback: true
|
||||
# offline_access yields refresh tokens (silent renewal + the deprovision
|
||||
# leash); Okta emits groups only when the `groups` scope is requested AND the
|
||||
# app's groups claim filter allows them (Okta admin console -> the app's
|
||||
# Sign On tab -> OpenID Connect ID Token -> Groups claim filter).
|
||||
scopes: [openid, profile, email, offline_access, groups]
|
||||
# groups_claim: groups # Okta default. Entra app roles=roles; see the config reference
|
||||
# ca_cert_pem: ${file:/secrets/idp-ca.pem} # only for an IdP behind a private CA
|
||||
|
||||
# ── Sessions ─────────────────────────────────────────────────────────────────
|
||||
session:
|
||||
jwt_secret: ${GATEWAY_JWT_SECRET} # >= 32 bytes; openssl rand -base64 32
|
||||
# Okta issues refresh tokens (offline_access above), so sessions renew
|
||||
# silently and this mainly bounds deprovision latency. 8 is a sane default;
|
||||
# lower toward 1 for tighter revocation. Array form rotates keys:
|
||||
# [new, old] (index 0 signs, all verify).
|
||||
ttl_hours: 8
|
||||
|
||||
# ── Store (REQUIRED — the gateway refuses to boot without it) ─────────────────
|
||||
store:
|
||||
postgres_url: ${GATEWAY_POSTGRES_URL} # private-subnet RDS; built with ?sslmode=verify-full by setup.sh
|
||||
# (the image trusts the RDS CA bundle via NODE_EXTRA_CA_CERTS — see Dockerfile)
|
||||
|
||||
# ── Upstreams — Amazon Bedrock ───────────────────────────────────────────────
|
||||
upstreams:
|
||||
- provider: bedrock
|
||||
# Must equal the region you provision in (setup.sh's AWS_REGION /
|
||||
# terraform's region): the IAM policy's inference-profile ARNs are scoped
|
||||
# to that region, and Bedrock model access is enabled there (cross-region
|
||||
# us.anthropic.* profiles need access in every spanned region). NOTE: the
|
||||
# walkthrough is scoped to US regions — the built-in model catalog maps to
|
||||
# us.anthropic.* (US-geo) profiles; a non-US region also needs a models:
|
||||
# list below (see the model catalog section).
|
||||
region: REPLACE_ME # e.g. us-east-1
|
||||
auth: {} # AWS default credential chain: ECS task role / IRSA on EKS (preferred — no static keys)
|
||||
# base_url: https://bedrock-runtime.us-east-1.amazonaws.com # bedrock-runtime interface VPC endpoint, to keep model traffic off the public path
|
||||
# Add more upstreams for failover (tried top→bottom on 5xx/timeout/501): a
|
||||
# second region, or an anthropic/vertex fallback. See
|
||||
# https://code.claude.com/docs/en/claude-apps-gateway.
|
||||
|
||||
# ── Telemetry fan-out (OPTIONAL) ─────────────────────────────────────────────
|
||||
# The CLI sends OTLP/HTTP to the gateway; the gateway fans out, stamping
|
||||
# user.id/user.email/user.groups server-side. On AWS, point at an OpenTelemetry
|
||||
# Collector (e.g. the AWS Distro for OpenTelemetry -> CloudWatch / Managed
|
||||
# Prometheus). When forward_to and public_url are both configured the gateway
|
||||
# pushes CLAUDE_CODE_ENABLE_TELEMETRY and the OTEL exporter selectors to every
|
||||
# client automatically — no per-developer config needed.
|
||||
# telemetry:
|
||||
# forward_to:
|
||||
# - url: https://otel-collector.internal.example.com:4318
|
||||
# headers:
|
||||
# Authorization: ${file:/secrets/otlp-token}
|
||||
# metrics: true # safe aggregate counters (default)
|
||||
# logs: false # carries bash commands / tool inputs — opt in deliberately
|
||||
# traces: false
|
||||
|
||||
# ── RBAC + managed settings (OPTIONAL; first-match-wins, top -> bottom) ───────
|
||||
# With Okta as IdP, match on the group names the `groups` scope emits (subject
|
||||
# to the app's groups claim filter), or on email_domain.
|
||||
# managed:
|
||||
# policies:
|
||||
# - match: { groups: [engineering] }
|
||||
# cli:
|
||||
# availableModels: [claude-opus-4-8, claude-sonnet-4-6, claude-haiku-4-5]
|
||||
# permissions: { deny: ["Read(./.env)", "Read(./secrets/**)"] }
|
||||
# - match: {} # catch-all floor — keep LAST
|
||||
# cli:
|
||||
# availableModels: [claude-sonnet-4-6, claude-haiku-4-5]
|
||||
|
||||
# ── Admin API (OPTIONAL — enables db-mode runtime config + spend caps) ───────
|
||||
# admin_groups needs a groups claim — Okta provides one via the `groups` scope
|
||||
# above — or use the bootstrap keys below instead. Named keys for attribution
|
||||
# in the audit log; 32-char minimum on key values. On ECS add these to the task
|
||||
# definition's `secrets` field (valueFrom -> a Secrets Manager ARN), same as the
|
||||
# JWT/OIDC/Postgres secrets above; on EKS you may use ${file:...}.
|
||||
# admin:
|
||||
# write_keys:
|
||||
# - id: terraform
|
||||
# key: ${GATEWAY_ADMIN_WRITE_KEY}
|
||||
# read_keys:
|
||||
# - id: reporting
|
||||
# key: ${GATEWAY_ADMIN_READ_KEY}
|
||||
# # admin_groups: [platform-finops] # Okta group names via the groups scope
|
||||
|
||||
# ── Model catalog (OPTIONAL for US regions) ──────────────────────────────────
|
||||
# Default true: every built-in Claude model is exposed and auto-translated per
|
||||
# upstream (the built-in table already maps to us.anthropic.* cross-region
|
||||
# inference profiles). Set false + a models: list to pin IDs (e.g. an
|
||||
# application or provisioned-throughput inference-profile ARN).
|
||||
# NON-US REGIONS: the built-in us.anthropic.* mappings do not exist outside
|
||||
# the US geo — set auto_include_builtin_models: false and list your region's
|
||||
# inference profiles (eu.anthropic.*, apac.anthropic.*, ...) here, and widen
|
||||
# the geo prefix in the deploy's bedrock-invoke IAM policy to match. See the
|
||||
# models: guidance in the config reference:
|
||||
# https://code.claude.com/docs/en/claude-apps-gateway-config
|
||||
# auto_include_builtin_models: true
|
||||
# models:
|
||||
# - id: claude-opus-4-8
|
||||
# label: Claude Opus 4.8
|
||||
# upstream_model: { bedrock: us.anthropic.claude-opus-4-8 }
|
||||
964
examples/gateway/aws/setup.sh
Executable file
964
examples/gateway/aws/setup.sh
Executable file
@@ -0,0 +1,964 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# setup.sh — AWS setup for Claude apps gateway (walkthrough §1–7, ECS track).
|
||||
#
|
||||
# Provisions, in this order: the three security groups (§1), the task +
|
||||
# execution IAM roles (§2), the gateway container image in Amazon ECR (§6),
|
||||
# an RDS for PostgreSQL instance in the private subnets with no public
|
||||
# address (§3), the JWT + postgres-url secrets (§5), and an ECS Fargate
|
||||
# service behind an internal Application Load Balancer (§7).
|
||||
#
|
||||
# gateway.yaml (§4 of the walkthrough) is BAKED INTO THE IMAGE on this track —
|
||||
# the task definition injects only the secrets it references, as env vars — so
|
||||
# the config step here lives inside the image build (§6): the build is gated on
|
||||
# a fully filled-in gateway.yaml and the image tag carries a hash of it, so a
|
||||
# config edit triggers a rebuild on the next run.
|
||||
#
|
||||
# Section markers (§N) below map to the walkthrough:
|
||||
# https://code.claude.com/docs/en/claude-apps-gateway-on-aws
|
||||
#
|
||||
# Covers here: security groups (§1) -> IAM roles + Bedrock model-access note (§2)
|
||||
# -> build & push image, config baked in (§6 + §4) -> DB subnet group
|
||||
# + RDS instance (§3) -> jwt + postgres-url secrets (§5) -> ECS
|
||||
# cluster/task definition/service + internal ALB (§7, ECS Fargate tab).
|
||||
# Not covered: EKS track (§7's EKS tab) — ECS Fargate is the lower-friction path here.
|
||||
# Bedrock model access (§2) — console-only; the script reminds you.
|
||||
# Route 53 alias — see the next steps it prints. Client MDM
|
||||
# push (§8) is covered by the walkthrough, not this script.
|
||||
#
|
||||
# Idempotent: existing resources are detected and skipped, so it is safe to re-run.
|
||||
# Reuse is by NAME, so a pre-existing resource may not match what this script
|
||||
# would have created: reuse that would change the exposure model is fatal (an
|
||||
# ALB that is not internal/in ${VPC_ID}); upsert-able settings are converged on
|
||||
# every run; other posture drift (extra security group ingress, a public or
|
||||
# unencrypted RDS instance, wrong-VPC target group) is checked and warned
|
||||
# about, never silently adopted.
|
||||
# Override any default below via environment variable, e.g. `AWS_REGION=us-west-2 ./setup.sh`.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# ---- configuration (env-overridable) ----------------------------------------
|
||||
AWS_REGION="${AWS_REGION:-$(aws configure get region 2>/dev/null || true)}" # guide uses us-east-1 (a region where Bedrock serves the Claude models you need)
|
||||
ACCOUNT_ID="${ACCOUNT_ID:-$(aws sts get-caller-identity --query Account --output text 2>/dev/null || true)}"
|
||||
|
||||
VPC_ID="${VPC_ID:-}" # REQUIRED — the VPC from the prerequisites
|
||||
PRIVATE_SUBNETS="${PRIVATE_SUBNETS:-}" # REQUIRED — two+ private subnet IDs in different AZs, space-separated
|
||||
CORP_CIDR="${CORP_CIDR:-}" # REQUIRED — your corporate network CIDR (ALB :443 ingress source)
|
||||
# Must not overlap PRIVATE_SUBNETS: hosts there are trusted_proxies (gateway.yaml) and could spoof client IPs via X-Forwarded-For.
|
||||
|
||||
# §1 security groups
|
||||
ALB_SG_NAME="${ALB_SG_NAME:-claude-gateway-alb}"
|
||||
GW_SG_NAME="${GW_SG_NAME:-claude-gateway-svc}"
|
||||
DB_SG_NAME="${DB_SG_NAME:-claude-gateway-db}"
|
||||
|
||||
# §2 IAM roles (task role = the gateway's runtime AWS identity; execution role
|
||||
# = the ECS agent's identity for pulling the image and injecting secrets)
|
||||
TASK_ROLE="${TASK_ROLE:-claude-gateway-task}"
|
||||
EXEC_ROLE="${EXEC_ROLE:-claude-gateway-execution}"
|
||||
|
||||
# §6 image
|
||||
ECR_REPO="${ECR_REPO:-claude-gateway}" # ECR repository name
|
||||
VERSION="${VERSION:-}" # REQUIRED — the gateway release tag you build and push (e.g. the linux-x64 binary's version)
|
||||
DOCKERFILE="${DOCKERFILE:-./Dockerfile}"
|
||||
CLAUDE_BINARY="${CLAUDE_BINARY:-./claude}" # prebuilt linux-x64 Claude Code release binary (includes the gateway subcommand)
|
||||
DIST_URL="${DIST_URL:-}" # optional: download URL, used only if $CLAUDE_BINARY is missing
|
||||
DIST_SHA256="${DIST_SHA256:-}" # REQUIRED with DIST_URL: expected sha256 of the binary (verified fail-closed)
|
||||
DIST_SHA256="${DIST_SHA256,,}" # normalize to lowercase — openssl emits lowercase hex; some tools (PowerShell Get-FileHash) publish uppercase
|
||||
# Obtain DIST_SHA256 out-of-band — never from the server that serves DIST_URL.
|
||||
# For binaries from the standard Claude Code release channel, verify the
|
||||
# release's GPG-signed manifest.json and copy the platform checksum from it:
|
||||
# https://code.claude.com/docs/en/setup#binary-integrity-and-code-signing
|
||||
# For any other distribution channel, use the checksum published alongside the
|
||||
# download link on that channel.
|
||||
GATEWAY_YAML="${GATEWAY_YAML:-./gateway.yaml}" # §4 config file — BAKED into the image
|
||||
RDS_CA_BUNDLE="${RDS_CA_BUNDLE:-./rds-global-bundle.pem}" # RDS CA trust anchor — BAKED into the image (downloaded below if missing)
|
||||
# Official AWS RDS truststore. AWS rotates this bundle (new regional CAs get
|
||||
# appended), so no checksum is pinned — a pinned hash would break on every
|
||||
# rotation. The script downloads it only when absent (an existing file is never
|
||||
# re-downloaded); to pick up a rotation, delete the file — and since the image
|
||||
# tag hashes only gateway.yaml, also bump VERSION or set IMAGE_TAG so the
|
||||
# next run rebuilds rather than reusing the existing tag. Operators who want
|
||||
# to pin may pre-place a reviewed copy at ${RDS_CA_BUNDLE}.
|
||||
RDS_CA_BUNDLE_URL="${RDS_CA_BUNDLE_URL:-https://truststore.pki.rds.amazonaws.com/global/global-bundle.pem}"
|
||||
REGISTRY="${ACCOUNT_ID}.dkr.ecr.${AWS_REGION}.amazonaws.com"
|
||||
|
||||
# §3 RDS
|
||||
DB_SUBNET_GROUP="${DB_SUBNET_GROUP:-claude-gateway-db}"
|
||||
DB_PARAM_GROUP="${DB_PARAM_GROUP:-claude-gateway-db}" # carries rds.force_ssl=1 (server-side TLS enforcement)
|
||||
DB_INSTANCE="${DB_INSTANCE:-claude-gateway-db}"
|
||||
DB_CLASS="${DB_CLASS:-db.t4g.micro}"
|
||||
DB_STORAGE_GB="${DB_STORAGE_GB:-20}"
|
||||
DB_NAME="${DB_NAME:-claude_gateway}"
|
||||
DB_USER="${DB_USER:-gateway}"
|
||||
# PG14+ supported; 16 is the recommended default (matches terraform/'s).
|
||||
# Always pinned: the instance's engine version and the parameter group's
|
||||
# family must name the same major, so both derive from this one value.
|
||||
DB_ENGINE_VERSION="${DB_ENGINE_VERSION:-16}"
|
||||
|
||||
SECRET_NAME="${SECRET_NAME:-gateway-postgres-url}" # §5 store.postgres_url
|
||||
JWT_SECRET_NAME="${JWT_SECRET_NAME:-gateway-jwt-secret}" # §5 session.jwt_secret
|
||||
OIDC_SECRET_NAME="${OIDC_SECRET_NAME:-gateway-oidc-client-secret}" # operator-created (Okta OIDC web app)
|
||||
# NOTE: the execution role's secrets-read policy (§2) is built from these
|
||||
# three names, one per-secret ARN prefix each — a rename is picked up on the
|
||||
# next run (put-role-policy is an upsert).
|
||||
|
||||
# §7 ECS + internal ALB deploy
|
||||
CLUSTER="${CLUSTER:-claude-gateway}"
|
||||
SERVICE="${SERVICE:-claude-gateway}"
|
||||
TASK_FAMILY="${TASK_FAMILY:-claude-gateway}"
|
||||
LOG_GROUP="${LOG_GROUP:-/ecs/claude-gateway}"
|
||||
LOG_RETENTION_DAYS="${LOG_RETENTION_DAYS:-90}" # CloudWatch retention — the group carries the gateway's audit events, so align with your audit retention policy
|
||||
ALB_NAME="${ALB_NAME:-claude-gateway}"
|
||||
TG_NAME="${TG_NAME:-claude-gateway}"
|
||||
# Explicit modern TLS policy — omitting it falls back to the legacy
|
||||
# ELBSecurityPolicy-2016-08 default, which still accepts TLS 1.0/1.1.
|
||||
ALB_SSL_POLICY="${ALB_SSL_POLICY:-ELBSecurityPolicy-TLS13-1-2-2021-06}"
|
||||
ACM_CERT_ARN="${ACM_CERT_ARN:-}" # REQUIRED for deploy — ACM cert for your internal gateway hostname
|
||||
TASK_CPU="${TASK_CPU:-1024}"
|
||||
TASK_MEMORY="${TASK_MEMORY:-2048}"
|
||||
DESIRED_COUNT="${DESIRED_COUNT:-1}" # each task opens a Postgres pool of up to 5 connections (store.max_connections default); keep DESIRED_COUNT × 5 below the DB class's max_connections (~80 on db.t4g.micro)
|
||||
DEPLOY="${DEPLOY:-1}" # set DEPLOY=0 to provision only, no ECS/ALB deploy
|
||||
|
||||
# ---- helpers ----------------------------------------------------------------
|
||||
log() { printf '\n==> %s\n' "$*"; }
|
||||
skip() { printf ' (exists) %s\n' "$*"; }
|
||||
curl_https() { curl --proto '=https' --proto-redir '=https' --tlsv1.2 "$@"; } # refuse plaintext/protocol-downgrade
|
||||
sha_of() { openssl dgst -sha256 "$1" | awk '{print $NF}'; } # openssl avoids shasum/sha256sum portability gaps
|
||||
|
||||
# authorize-security-group-ingress is NOT idempotent (re-adding a rule errors),
|
||||
# so tolerate exactly the duplicate-rule error and fail on anything else.
|
||||
authorize_ingress() {
|
||||
local out
|
||||
if out="$(aws ec2 authorize-security-group-ingress "$@" 2>&1)"; then
|
||||
return 0
|
||||
elif grep -q 'InvalidPermission.Duplicate' <<<"${out}"; then
|
||||
skip "ingress rule already present"
|
||||
else
|
||||
printf '%s\n' "${out}" >&2
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# Security-group lookup by name within the VPC; prints the GroupId or "None".
|
||||
sg_id() {
|
||||
aws ec2 describe-security-groups \
|
||||
--filters "Name=group-name,Values=$1" "Name=vpc-id,Values=${VPC_ID}" \
|
||||
--query 'SecurityGroups[0].GroupId' --output text 2>/dev/null || echo None
|
||||
}
|
||||
|
||||
# Name-based reuse can adopt a pre-existing group carrying ingress this script
|
||||
# never added. Audit after the intended rule is ensured: each group's traffic
|
||||
# path is exactly one rule (tcp <port> from <cidr-or-source-group>), so anything
|
||||
# else is flagged on stderr. Non-fatal — an extra rule may be a deliberate
|
||||
# operator addition — but every one widens the path, so it must be visible.
|
||||
warn_unexpected_ingress() { # <group-id> <group-name> <port> <expected cidr or source group-id>
|
||||
local perms
|
||||
if ! perms="$(aws ec2 describe-security-groups --group-ids "$1" \
|
||||
--query 'SecurityGroups[0].IpPermissions' --output json 2>/dev/null)"; then
|
||||
echo " WARN — could not audit ingress rules on $2 ($1)." >&2
|
||||
return 0
|
||||
fi
|
||||
# `|| echo` keeps a parse hiccup non-fatal — this audit must never abort a run.
|
||||
_SG_ID="$1" _SG_NAME="$2" _SG_PORT="$3" _SG_EXPECTED="$4" python3 -c "
|
||||
import json, os, sys
|
||||
perms = json.load(sys.stdin) or []
|
||||
port, expected = int(os.environ[\"_SG_PORT\"]), os.environ[\"_SG_EXPECTED\"]
|
||||
extras = []
|
||||
for p in perms:
|
||||
proto, lo, hi = p.get(\"IpProtocol\"), p.get(\"FromPort\"), p.get(\"ToPort\")
|
||||
scope_ok = proto == \"tcp\" and lo == port and hi == port
|
||||
sources = (
|
||||
[r.get(\"CidrIp\", \"?\") for r in p.get(\"IpRanges\", [])]
|
||||
+ [r.get(\"CidrIpv6\", \"?\") for r in p.get(\"Ipv6Ranges\", [])]
|
||||
+ [r.get(\"GroupId\", \"?\") for r in p.get(\"UserIdGroupPairs\", [])]
|
||||
+ [r.get(\"PrefixListId\", \"?\") for r in p.get(\"PrefixListIds\", [])]
|
||||
)
|
||||
extras += [(proto, lo, hi, s) for s in sources if not (scope_ok and s == expected)]
|
||||
if extras:
|
||||
name, gid = os.environ[\"_SG_NAME\"], os.environ[\"_SG_ID\"]
|
||||
print(f\" WARN — security group {name} ({gid}) has ingress beyond the intended rule\", file=sys.stderr)
|
||||
print(f\" (tcp {port} from {expected}) — review it; remove anything you did not add deliberately:\", file=sys.stderr)
|
||||
for proto, lo, hi, src in extras:
|
||||
scope = \"all traffic\" if proto == \"-1\" else (f\"{proto} {lo}\" if lo == hi else f\"{proto} {lo}-{hi}\")
|
||||
print(f\" {scope} from {src}\", file=sys.stderr)
|
||||
" <<<"${perms}" || echo " WARN — could not audit ingress rules on $2 ($1)." >&2
|
||||
}
|
||||
|
||||
secret_arn() {
|
||||
aws secretsmanager describe-secret --secret-id "$1" \
|
||||
--query ARN --output text 2>/dev/null || true
|
||||
}
|
||||
|
||||
# Existence check that fails closed: 0 = exists, 1 = definitively absent
|
||||
# (ResourceNotFoundException), anything else ABORTS the run. Gating on a bare
|
||||
# exit status would let a transient failure (throttle, expired token, network
|
||||
# blip) masquerade as "secret missing" — and the missing-secret branches below
|
||||
# do destructive work (the §3 self-heal resets the DB password), so they must
|
||||
# run only on a definitive not-found.
|
||||
secret_exists() { # <secret-id>
|
||||
local out
|
||||
if out="$(aws secretsmanager describe-secret --secret-id "$1" 2>&1 >/dev/null)"; then
|
||||
return 0
|
||||
elif grep -q 'ResourceNotFoundException' <<<"${out}"; then
|
||||
return 1
|
||||
else
|
||||
echo "ERROR: could not determine whether secret $1 exists (transient AWS error?):" >&2
|
||||
printf '%s\n' "${out}" >&2
|
||||
echo " Refusing to guess — re-run once the call succeeds." >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
# Secret values must never appear on a process argv (argv is world-readable
|
||||
# via /proc and routinely recorded by EDR/auditd), so every aws call that
|
||||
# carries one takes it via --cli-input-json file://<0600 temp file> instead —
|
||||
# explicit flags on the same command line override/merge with the JSON, so
|
||||
# only the secret parameter needs to live in the file. secret_json writes
|
||||
# {"<Key>": "<value>"} to a fresh temp file and returns the path in the named
|
||||
# variable (printf -v, not command substitution — a subshell would lose the
|
||||
# SECRET_TMP_FILES bookkeeping below): the value crosses into python3 via the
|
||||
# environment (never argv) and json.dumps escapes it, so any characters
|
||||
# survive. Callers rm -f the file as soon as the aws call returns; the EXIT
|
||||
# trap sweeps whatever an aborted run leaves.
|
||||
SECRET_TMP_FILES=()
|
||||
cleanup_secret_tmp() { rm -f "${SECRET_TMP_FILES[@]+"${SECRET_TMP_FILES[@]}"}"; }
|
||||
trap cleanup_secret_tmp EXIT
|
||||
secret_json() { # secret_json <outvar> <JsonKey> <value> -> path in <outvar>
|
||||
local file
|
||||
file="$(mktemp)" # mktemp creates 0600
|
||||
chmod 600 "${file}" # belt and braces if TMPDIR overrides umask semantics
|
||||
SECRET_TMP_FILES+=("${file}")
|
||||
_JSON_KEY="$2" _JSON_VALUE="$3" python3 -c \
|
||||
'import json, os; print(json.dumps({os.environ["_JSON_KEY"]: os.environ["_JSON_VALUE"]}))' \
|
||||
> "${file}"
|
||||
printf -v "$1" '%s' "${file}"
|
||||
}
|
||||
|
||||
for required in AWS_REGION ACCOUNT_ID VPC_ID PRIVATE_SUBNETS CORP_CIDR VERSION; do
|
||||
if [[ -z "${!required}" ]]; then
|
||||
echo "ERROR: ${required} is not set." >&2
|
||||
case "${required}" in
|
||||
AWS_REGION) echo " Set it to a region where Bedrock serves the Claude models you need, e.g. export AWS_REGION=us-east-1" >&2 ;;
|
||||
ACCOUNT_ID) echo " Could not resolve it from STS — is the AWS CLI authenticated? (aws sts get-caller-identity)" >&2 ;;
|
||||
VPC_ID) echo " Set it to the VPC from the prerequisites, e.g. export VPC_ID=vpc-..." >&2 ;;
|
||||
PRIVATE_SUBNETS) echo " Set it to two+ private subnet IDs in different AZs, e.g. export PRIVATE_SUBNETS='subnet-a subnet-b'" >&2 ;;
|
||||
CORP_CIDR) echo " Set it to your corporate network CIDR (the ALB's :443 ingress source), e.g. export CORP_CIDR=10.0.0.0/8" >&2 ;;
|
||||
VERSION) echo " Set it to the gateway release version — it tags the image you build and push, e.g. export VERSION=<version>" >&2 ;;
|
||||
esac
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
# The walkthrough (and this bundle) is scoped to commercial US regions: the
|
||||
# task role's Bedrock policy (§2) and the gateway's built-in model catalog
|
||||
# both use the us.anthropic.* geo-prefixed cross-region inference profiles,
|
||||
# which only exist in the commercial US regions — an explicit list, not a
|
||||
# `us-*` prefix match, because GovCloud (us-gov-*) and ISO (us-iso-*) regions
|
||||
# share the prefix but live in different AWS partitions where those profiles
|
||||
# and this bundle's arn:aws: ARNs are wrong. Anywhere else the deploy
|
||||
# provisions fine and then every model call fails. Other-region deploys must
|
||||
# pin region-appropriate inference profiles via a models: block in
|
||||
# gateway.yaml (see the config reference:
|
||||
# https://code.claude.com/docs/en/claude-apps-gateway-config) and adjust the
|
||||
# inference-profile ARN prefix in bedrock-invoke.iam.json below — set
|
||||
# ALLOW_NON_US_REGION=1 once that's done to proceed.
|
||||
case "${AWS_REGION}" in
|
||||
us-east-1|us-east-2|us-west-1|us-west-2) ;;
|
||||
*)
|
||||
if [[ "${ALLOW_NON_US_REGION:-0}" != "1" ]]; then
|
||||
echo "ERROR: AWS_REGION=${AWS_REGION} is not a commercial US region, but this bundle's IAM policy" >&2
|
||||
echo " and model IDs use the US-geo (us.anthropic.*) cross-region inference profiles" >&2
|
||||
echo " (GovCloud/ISO regions are different partitions — the profiles and arn:aws: ARNs" >&2
|
||||
echo " here do not exist there)." >&2
|
||||
echo " Either deploy to us-east-1/us-east-2/us-west-1/us-west-2, or pin region-appropriate" >&2
|
||||
echo " inference profiles in a models: block in gateway.yaml" >&2
|
||||
echo " (https://code.claude.com/docs/en/claude-apps-gateway-config), adjust the" >&2
|
||||
echo " inference-profile ARN in the bedrock-invoke policy, and re-run with" >&2
|
||||
echo " ALLOW_NON_US_REGION=1." >&2
|
||||
exit 1
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
|
||||
if ! command -v python3 >/dev/null 2>&1; then
|
||||
echo "ERROR: python3 is required (it JSON-escapes secret values for --cli-input-json; the AWS CLI itself ships on Python)." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# shellcheck disable=SC2086 # PRIVATE_SUBNETS is intentionally word-split everywhere below
|
||||
set -- ${PRIVATE_SUBNETS}
|
||||
if (( $# < 2 )); then
|
||||
echo "ERROR: PRIVATE_SUBNETS must list at least two subnets in different AZs (the internal ALB requires two)." >&2
|
||||
exit 1
|
||||
fi
|
||||
# Normalize whatever whitespace (spaces, tabs, newlines) separates the list —
|
||||
# `set --` above word-split on IFS, so join those same words with commas
|
||||
# rather than only converting single spaces.
|
||||
SUBNETS_CSV="$(printf '%s,' "$@")"; SUBNETS_CSV="${SUBNETS_CSV%,}"
|
||||
|
||||
log "Account: ${ACCOUNT_ID} Region: ${AWS_REGION} VPC: ${VPC_ID}"
|
||||
|
||||
# ---- 1 Security groups -----------------------------------------------------
|
||||
# Three groups chain the traffic path (walkthrough §1): corp network -> ALB :443,
|
||||
# ALB -> gateway :8080, gateway -> Postgres :5432. Nothing else is reachable.
|
||||
log "Creating security groups (§1)"
|
||||
ALB_SG="$(sg_id "${ALB_SG_NAME}")"
|
||||
if [[ "${ALB_SG}" != "None" ]]; then
|
||||
skip "security group ${ALB_SG_NAME} (${ALB_SG})"
|
||||
else
|
||||
ALB_SG="$(aws ec2 create-security-group --group-name "${ALB_SG_NAME}" \
|
||||
--description "Claude gateway ALB" --vpc-id "${VPC_ID}" \
|
||||
--query GroupId --output text)"
|
||||
fi
|
||||
|
||||
GW_SG="$(sg_id "${GW_SG_NAME}")"
|
||||
if [[ "${GW_SG}" != "None" ]]; then
|
||||
skip "security group ${GW_SG_NAME} (${GW_SG})"
|
||||
else
|
||||
GW_SG="$(aws ec2 create-security-group --group-name "${GW_SG_NAME}" \
|
||||
--description "Claude gateway service" --vpc-id "${VPC_ID}" \
|
||||
--query GroupId --output text)"
|
||||
fi
|
||||
|
||||
DB_SG="$(sg_id "${DB_SG_NAME}")"
|
||||
if [[ "${DB_SG}" != "None" ]]; then
|
||||
skip "security group ${DB_SG_NAME} (${DB_SG})"
|
||||
else
|
||||
DB_SG="$(aws ec2 create-security-group --group-name "${DB_SG_NAME}" \
|
||||
--description "Claude gateway Postgres" --vpc-id "${VPC_ID}" \
|
||||
--query GroupId --output text)"
|
||||
fi
|
||||
|
||||
authorize_ingress --group-id "${ALB_SG}" --protocol tcp --port 443 --cidr "${CORP_CIDR}"
|
||||
authorize_ingress --group-id "${GW_SG}" --protocol tcp --port 8080 --source-group "${ALB_SG}"
|
||||
authorize_ingress --group-id "${DB_SG}" --protocol tcp --port 5432 --source-group "${GW_SG}"
|
||||
|
||||
# Flag any ingress beyond the three rules above (pre-existing groups may carry more).
|
||||
warn_unexpected_ingress "${ALB_SG}" "${ALB_SG_NAME}" 443 "${CORP_CIDR}"
|
||||
warn_unexpected_ingress "${GW_SG}" "${GW_SG_NAME}" 8080 "${ALB_SG}"
|
||||
warn_unexpected_ingress "${DB_SG}" "${DB_SG_NAME}" 5432 "${GW_SG}"
|
||||
|
||||
# ---- 2 IAM roles ------------------------------------------------------------
|
||||
# Task role: the gateway's runtime identity — its ONLY permission is invoking
|
||||
# Claude models on Bedrock (the upstream's `auth: {}` resolves to this role via
|
||||
# the AWS default credential chain). The policy must cover both the cross-region
|
||||
# inference-profile ARNs and the underlying foundation-model ARNs.
|
||||
# Execution role: the ECS agent's identity — pulls the image from ECR and
|
||||
# injects the Secrets Manager values; the gateway never uses it.
|
||||
log "Creating IAM roles ${TASK_ROLE} + ${EXEC_ROLE} (§2)"
|
||||
cat > ecs-trust.iam.json <<'EOF'
|
||||
{
|
||||
"Version": "2012-10-17",
|
||||
"Statement": [{
|
||||
"Effect": "Allow",
|
||||
"Principal": { "Service": "ecs-tasks.amazonaws.com" },
|
||||
"Action": "sts:AssumeRole"
|
||||
}]
|
||||
}
|
||||
EOF
|
||||
cat > bedrock-invoke.iam.json <<EOF
|
||||
{
|
||||
"Version": "2012-10-17",
|
||||
"Statement": [{
|
||||
"Effect": "Allow",
|
||||
"Action": ["bedrock:InvokeModel", "bedrock:InvokeModelWithResponseStream"],
|
||||
"Resource": [
|
||||
"arn:aws:bedrock:${AWS_REGION}:${ACCOUNT_ID}:inference-profile/us.anthropic.*",
|
||||
"arn:aws:bedrock:*::foundation-model/anthropic.*"
|
||||
]
|
||||
}]
|
||||
}
|
||||
EOF
|
||||
# One ARN per secret (never a bare gateway-* wildcard, which would also match
|
||||
# unrelated secrets in a shared account). The trailing -?????? matches exactly
|
||||
# the random 6-character suffix Secrets Manager appends to every secret's ARN
|
||||
# (AWS's documented pattern; a trailing -* would be a plain prefix glob and
|
||||
# also match longer names like ${SECRET_NAME}-prod) — the exact ARNs aren't
|
||||
# knowable here because the role is created before the secrets are.
|
||||
cat > secrets-read.iam.json <<EOF
|
||||
{
|
||||
"Version": "2012-10-17",
|
||||
"Statement": [{
|
||||
"Effect": "Allow",
|
||||
"Action": "secretsmanager:GetSecretValue",
|
||||
"Resource": [
|
||||
"arn:aws:secretsmanager:${AWS_REGION}:${ACCOUNT_ID}:secret:${JWT_SECRET_NAME}-??????",
|
||||
"arn:aws:secretsmanager:${AWS_REGION}:${ACCOUNT_ID}:secret:${OIDC_SECRET_NAME}-??????",
|
||||
"arn:aws:secretsmanager:${AWS_REGION}:${ACCOUNT_ID}:secret:${SECRET_NAME}-??????"
|
||||
]
|
||||
}]
|
||||
}
|
||||
EOF
|
||||
|
||||
if aws iam get-role --role-name "${TASK_ROLE}" >/dev/null 2>&1; then
|
||||
skip "role ${TASK_ROLE}"
|
||||
else
|
||||
aws iam create-role --role-name "${TASK_ROLE}" \
|
||||
--assume-role-policy-document file://ecs-trust.iam.json >/dev/null
|
||||
fi
|
||||
# put-role-policy is an upsert — safe to re-run (it also picks up region changes).
|
||||
aws iam put-role-policy --role-name "${TASK_ROLE}" \
|
||||
--policy-name bedrock-invoke --policy-document file://bedrock-invoke.iam.json
|
||||
|
||||
if aws iam get-role --role-name "${EXEC_ROLE}" >/dev/null 2>&1; then
|
||||
skip "role ${EXEC_ROLE}"
|
||||
else
|
||||
aws iam create-role --role-name "${EXEC_ROLE}" \
|
||||
--assume-role-policy-document file://ecs-trust.iam.json >/dev/null
|
||||
fi
|
||||
# attach-role-policy is idempotent (re-attaching is a no-op).
|
||||
aws iam attach-role-policy --role-name "${EXEC_ROLE}" \
|
||||
--policy-arn arn:aws:iam::aws:policy/service-role/AmazonECSTaskExecutionRolePolicy
|
||||
aws iam put-role-policy --role-name "${EXEC_ROLE}" \
|
||||
--policy-name read-gateway-secrets --policy-document file://secrets-read.iam.json
|
||||
|
||||
echo " NOTE: Bedrock model access is console-only — enable it for the Claude models"
|
||||
echo " you need (Bedrock console -> Model access), and submit the one-time use"
|
||||
echo " case form for the account. Cross-region inference profiles"
|
||||
echo " (us.anthropic.*) need access in EACH region the profile spans."
|
||||
|
||||
# ---- 6 Build & push image to Amazon ECR (config baked in — §6 + §4) ---------
|
||||
log "Ensuring ECR repository and image (§6)"
|
||||
if aws ecr describe-repositories --repository-names "${ECR_REPO}" >/dev/null 2>&1; then
|
||||
skip "ECR repository ${ECR_REPO}"
|
||||
# Integrity-critical settings: converge on re-runs so a pre-existing MUTABLE repo can't slip through.
|
||||
aws ecr put-image-tag-mutability --repository-name "${ECR_REPO}" \
|
||||
--image-tag-mutability IMMUTABLE >/dev/null
|
||||
aws ecr put-image-scanning-configuration --repository-name "${ECR_REPO}" \
|
||||
--image-scanning-configuration scanOnPush=true >/dev/null
|
||||
else
|
||||
# IMMUTABLE tags + scan-on-push: the ECS service pulls whatever this repo
|
||||
# serves under the deployed tag, so a pushed tag must never be silently
|
||||
# re-pointed. For production, also restrict push rights on this repo to your
|
||||
# CI / image-promotion pipeline rather than operator credentials — this
|
||||
# walkthrough pushes directly for simplicity.
|
||||
aws ecr create-repository --repository-name "${ECR_REPO}" \
|
||||
--image-tag-mutability IMMUTABLE \
|
||||
--image-scanning-configuration scanOnPush=true >/dev/null
|
||||
fi
|
||||
|
||||
# The config is baked into the image, so the build is gated the way the GCP
|
||||
# example gates its config-secret publish: gateway.yaml must exist and be fully
|
||||
# filled in (REPLACE_ME checked on non-comment lines so commented examples and
|
||||
# the file's header don't trip the guard). The tag carries a hash of the config
|
||||
# so an edit produces a NEW tag (required by tag immutability) and a re-run
|
||||
# rebuilds automatically.
|
||||
IMAGE=""
|
||||
if [[ ! -f "${GATEWAY_YAML}" ]]; then
|
||||
echo " (skip) ${GATEWAY_YAML} not found — run 'cp gateway.yaml.example gateway.yaml', fill it in, then re-run (§4)."
|
||||
elif grep -vE '^[[:space:]]*#' "${GATEWAY_YAML}" | grep -q 'REPLACE_ME'; then
|
||||
echo " (skip) ${GATEWAY_YAML} still has REPLACE_ME placeholders to fill:"
|
||||
grep -nE 'REPLACE_ME' "${GATEWAY_YAML}" | grep -vE '^[0-9]+:[[:space:]]*#' | sed 's/^/ /'
|
||||
echo " Fill them in, then re-run to build the image (the config is baked in)."
|
||||
else
|
||||
CONFIG_SHA="$(sha_of "${GATEWAY_YAML}" | cut -c1-8)"
|
||||
IMAGE_TAG="${IMAGE_TAG:-${VERSION}-cfg${CONFIG_SHA}}"
|
||||
IMAGE="${REGISTRY}/${ECR_REPO}:${IMAGE_TAG}"
|
||||
|
||||
# Image is the expensive, already-done step: skip the build+push entirely if
|
||||
# the tag already exists in the registry.
|
||||
if aws ecr describe-images --repository-name "${ECR_REPO}" \
|
||||
--image-ids "imageTag=${IMAGE_TAG}" >/dev/null 2>&1; then
|
||||
skip "image ${IMAGE}"
|
||||
else
|
||||
# When the expected checksum is known, verify a PRE-EXISTING binary too:
|
||||
# the [[ ! -f ]] guard below otherwise trusts whatever is on disk, so a
|
||||
# stale binary from an earlier VERSION (or a tampered one) would be baked
|
||||
# into the image silently. On mismatch, set it aside (never delete — the
|
||||
# mismatch may be a typo'd DIST_SHA256, not a bad binary) and fall through
|
||||
# to the fail-closed download path. Without DIST_SHA256 the operator-
|
||||
# provided-binary flow is unchanged — no checksum was declared, so none is
|
||||
# checked.
|
||||
QUARANTINED_SHA=""
|
||||
if [[ -n "${DIST_SHA256}" && -f "${CLAUDE_BINARY}" ]]; then
|
||||
existing_sha="$(sha_of "${CLAUDE_BINARY}")"
|
||||
if [[ "${existing_sha}" != "${DIST_SHA256}" ]]; then
|
||||
log "Existing ${CLAUDE_BINARY} sha256 ${existing_sha} does not match DIST_SHA256 — setting it aside as ${CLAUDE_BINARY}.bad"
|
||||
mv -f "${CLAUDE_BINARY}" "${CLAUDE_BINARY}.bad"
|
||||
QUARANTINED_SHA="${existing_sha}"
|
||||
fi
|
||||
fi
|
||||
if [[ ! -f "${CLAUDE_BINARY}" ]]; then
|
||||
if [[ -n "${DIST_URL}" ]]; then
|
||||
# Fail closed: never download an executable we can't verify.
|
||||
if [[ -z "${DIST_SHA256}" ]]; then
|
||||
echo "ERROR: DIST_SHA256 must be set when DIST_URL is used — refusing to download an unverified binary." >&2
|
||||
echo " Set DIST_SHA256 to the expected sha256 of the binary at DIST_URL, obtained out-of-band:" >&2
|
||||
echo " for standard-release binaries, from the release's GPG-signed manifest.json (verify the" >&2
|
||||
echo " manifest signature first — see code.claude.com/docs/en/setup#binary-integrity-and-code-signing);" >&2
|
||||
echo " otherwise from the channel that published the download link, never from the download server." >&2
|
||||
exit 1
|
||||
fi
|
||||
log "Downloading gateway binary from ${DIST_URL}"
|
||||
# Download to a temp path and only mv into place after the checksum
|
||||
# verifies, so an interrupted download can't leave a partial CLAUDE_BINARY
|
||||
# that the [[ ! -f ]] guard above would skip — and silently push — on re-run.
|
||||
# Refuse plaintext/protocol-downgrade; only follow HTTPS redirects.
|
||||
dl_tmp="${CLAUDE_BINARY}.download"
|
||||
rm -f "${dl_tmp}"
|
||||
curl_https -fL -o "${dl_tmp}" "${DIST_URL}"
|
||||
actual_sha="$(sha_of "${dl_tmp}")"
|
||||
if [[ "${actual_sha}" != "${DIST_SHA256}" ]]; then
|
||||
echo "ERROR: checksum mismatch for ${dl_tmp} (expected ${DIST_SHA256}, got ${actual_sha}) — refusing to build." >&2
|
||||
rm -f "${dl_tmp}"
|
||||
exit 1
|
||||
fi
|
||||
log "Verified binary sha256 ${actual_sha}"
|
||||
chmod +x "${dl_tmp}"
|
||||
mv -f "${dl_tmp}" "${CLAUDE_BINARY}"
|
||||
else
|
||||
echo "ERROR: build binary not found at ${CLAUDE_BINARY} and DIST_URL is not set." >&2
|
||||
if [[ -n "${QUARANTINED_SHA}" ]]; then
|
||||
echo " The binary that WAS there had sha256 ${QUARANTINED_SHA}, which does not match" >&2
|
||||
echo " DIST_SHA256=${DIST_SHA256} — it was preserved as ${CLAUDE_BINARY}.bad." >&2
|
||||
echo " If DIST_SHA256 was a typo, fix it and move the file back:" >&2
|
||||
echo " mv '${CLAUDE_BINARY}.bad' '${CLAUDE_BINARY}'" >&2
|
||||
echo " Otherwise treat that file as untrusted and obtain a verified binary." >&2
|
||||
fi
|
||||
echo " Provide the prebuilt linux-x64 Claude Code release binary at that path" >&2
|
||||
echo " or set DIST_URL to its download URL (see the walkthrough, §6)." >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
# The RDS CA bundle is baked into the image as the trust anchor for the
|
||||
# connection string's sslmode=verify-full (§3/§5). Fail closed: no bundle,
|
||||
# no build. AWS rotates the bundle, so no checksum is pinned (see the
|
||||
# RDS_CA_BUNDLE_URL comment up top); the sanity check below catches an
|
||||
# error page or truncated download.
|
||||
if [[ ! -f "${RDS_CA_BUNDLE}" ]]; then
|
||||
log "Downloading RDS CA bundle from ${RDS_CA_BUNDLE_URL}"
|
||||
curl_https -fL -o "${RDS_CA_BUNDLE}" "${RDS_CA_BUNDLE_URL}"
|
||||
fi
|
||||
if ! grep -q 'BEGIN CERTIFICATE' "${RDS_CA_BUNDLE}" \
|
||||
|| (( "$(wc -c < "${RDS_CA_BUNDLE}")" < 10000 )); then
|
||||
echo "ERROR: ${RDS_CA_BUNDLE} does not look like the RDS CA bundle (missing PEM blocks or implausibly small) — refusing to build." >&2
|
||||
echo " Delete it and re-run to re-download, or place the bundle from ${RDS_CA_BUNDLE_URL} there yourself." >&2
|
||||
exit 1
|
||||
fi
|
||||
log "Building and pushing ${IMAGE}"
|
||||
aws ecr get-login-password --region "${AWS_REGION}" \
|
||||
| docker login --username AWS --password-stdin "${REGISTRY}"
|
||||
# The task definition below runs linux/amd64 (cpuArchitecture X86_64);
|
||||
# --platform forces it (e.g. when building on an Apple Silicon Mac), and
|
||||
# --provenance=false keeps buildx from wrapping the result in an OCI image
|
||||
# index that some pullers reject. For Fargate on ARM64 (Graviton), build
|
||||
# linux/arm64 with the linux-arm64 binary and set cpuArchitecture to ARM64.
|
||||
docker build --platform=linux/amd64 --provenance=false \
|
||||
-f "${DOCKERFILE}" \
|
||||
--build-arg CLAUDE_BINARY="${CLAUDE_BINARY}" \
|
||||
--build-arg GATEWAY_CONFIG="${GATEWAY_YAML}" \
|
||||
--build-arg RDS_CA_BUNDLE="${RDS_CA_BUNDLE}" \
|
||||
-t "${IMAGE}" .
|
||||
docker push "${IMAGE}"
|
||||
fi
|
||||
fi
|
||||
|
||||
# ---- 3 RDS for PostgreSQL (private subnets, no public address) --------------
|
||||
log "Creating DB subnet group ${DB_SUBNET_GROUP} (§3)"
|
||||
if aws rds describe-db-subnet-groups --db-subnet-group-name "${DB_SUBNET_GROUP}" >/dev/null 2>&1; then
|
||||
skip "DB subnet group ${DB_SUBNET_GROUP}"
|
||||
else
|
||||
# shellcheck disable=SC2086 # subnet IDs are separate arguments by design
|
||||
aws rds create-db-subnet-group --db-subnet-group-name "${DB_SUBNET_GROUP}" \
|
||||
--db-subnet-group-description "Claude gateway" --subnet-ids ${PRIVATE_SUBNETS} >/dev/null
|
||||
fi
|
||||
|
||||
# Parameter group with rds.force_ssl=1: the server side of TLS enforcement —
|
||||
# the client side is sslmode=verify-full in the connection string (§5). The
|
||||
# family must match the engine major version, so it derives from the same
|
||||
# DB_ENGINE_VERSION that create-db-instance pins below.
|
||||
log "Ensuring DB parameter group ${DB_PARAM_GROUP} (rds.force_ssl=1)"
|
||||
PG_FAMILY="postgres${DB_ENGINE_VERSION%%.*}"
|
||||
if aws rds describe-db-parameter-groups --db-parameter-group-name "${DB_PARAM_GROUP}" >/dev/null 2>&1; then
|
||||
skip "DB parameter group ${DB_PARAM_GROUP}"
|
||||
else
|
||||
aws rds create-db-parameter-group --db-parameter-group-name "${DB_PARAM_GROUP}" \
|
||||
--db-parameter-group-family "${PG_FAMILY}" \
|
||||
--description "Claude gateway - require TLS on every connection" >/dev/null
|
||||
fi
|
||||
# modify-db-parameter-group is an upsert — applied every run so a pre-existing
|
||||
# group converges too. rds.force_ssl is dynamic; no reboot needed.
|
||||
aws rds modify-db-parameter-group --db-parameter-group-name "${DB_PARAM_GROUP}" \
|
||||
--parameters "ParameterName=rds.force_ssl,ParameterValue=1,ApplyMethod=immediate" >/dev/null
|
||||
|
||||
# hex (not base64) keeps the password URL-safe for the connection string below.
|
||||
# The password reaches every aws call via --cli-input-json (never argv — see
|
||||
# the secret_json helper); explicit flags merge with (and would override) the
|
||||
# JSON, so only the password lives in the temp file.
|
||||
log "Creating RDS instance ${DB_INSTANCE} (private subnets, --no-publicly-accessible)"
|
||||
DB_PASSWORD=""
|
||||
DB_POSTURE="$(aws rds describe-db-instances --db-instance-identifier "${DB_INSTANCE}" \
|
||||
--query 'DBInstances[0].[PubliclyAccessible,StorageEncrypted]' --output text 2>/dev/null || true)"
|
||||
if [[ -n "${DB_POSTURE}" ]]; then
|
||||
# Name-based reuse: a pre-existing instance may not carry the posture this
|
||||
# script would have created it with. Non-fatal (the operator may be migrating
|
||||
# an existing DB on purpose), but drift from the guide's baseline must be seen.
|
||||
read -r DB_PUBLIC DB_ENCRYPTED <<<"${DB_POSTURE}"
|
||||
if [[ "${DB_PUBLIC}" == "True" ]]; then
|
||||
echo " WARN — RDS instance ${DB_INSTANCE} is PubliclyAccessible; this script would have" >&2
|
||||
echo " created it with --no-publicly-accessible. Fix: aws rds modify-db-instance" >&2
|
||||
echo " --db-instance-identifier ${DB_INSTANCE} --no-publicly-accessible --apply-immediately" >&2
|
||||
fi
|
||||
if [[ "${DB_ENCRYPTED}" == "False" ]]; then
|
||||
echo " WARN — RDS instance ${DB_INSTANCE} has StorageEncrypted=false; this script would" >&2
|
||||
echo " have created it with --storage-encrypted (encryption cannot be enabled in" >&2
|
||||
echo " place — restore an encrypted snapshot copy to migrate)." >&2
|
||||
fi
|
||||
if secret_exists "${SECRET_NAME}"; then
|
||||
skip "instance ${DB_INSTANCE} (password unchanged; secret not rewritten)"
|
||||
else
|
||||
# Self-heal: a previous run died after creating the instance but before
|
||||
# writing the connection-string secret, losing the only copy of the
|
||||
# password. The secret is the password's only consumer, so resetting it is
|
||||
# safe and keeps re-runs able to recover from any partial state.
|
||||
# secret_exists (not a bare exit-status check) gates this: only a
|
||||
# definitive ResourceNotFoundException may trigger a password reset.
|
||||
# ORDERING INVARIANT: the secret write (§5 below) is the heal's commit
|
||||
# point — everything that can fail must happen BEFORE it, so a crash at
|
||||
# any point leaves the secret still missing and the next run simply
|
||||
# repeats the heal. Writing the secret first would invert that: a crash
|
||||
# between secret write and modify-db-instance would leave an existing
|
||||
# secret whose password the DB never received, and every later run would
|
||||
# skip the heal while the gateway can't connect.
|
||||
# NOTE: the parameter group is attached on create only — an instance that
|
||||
# predates it keeps its current group (attach via modify-db-instance
|
||||
# --db-parameter-group-name yourself if you want force_ssl retrofitted).
|
||||
log "Instance ${DB_INSTANCE} exists but secret ${SECRET_NAME} is missing — resetting password"
|
||||
DB_PASSWORD="$(openssl rand -hex 24)"
|
||||
pw_json=""; secret_json pw_json MasterUserPassword "${DB_PASSWORD}"
|
||||
aws rds modify-db-instance --db-instance-identifier "${DB_INSTANCE}" \
|
||||
--cli-input-json "file://${pw_json}" --apply-immediately >/dev/null
|
||||
rm -f "${pw_json}"
|
||||
fi
|
||||
else
|
||||
DB_PASSWORD="$(openssl rand -hex 24)"
|
||||
pw_json=""; secret_json pw_json MasterUserPassword "${DB_PASSWORD}"
|
||||
aws rds create-db-instance --db-instance-identifier "${DB_INSTANCE}" \
|
||||
--engine postgres --engine-version "${DB_ENGINE_VERSION}" \
|
||||
--db-instance-class "${DB_CLASS}" \
|
||||
--allocated-storage "${DB_STORAGE_GB}" --db-name "${DB_NAME}" \
|
||||
--master-username "${DB_USER}" --cli-input-json "file://${pw_json}" \
|
||||
--db-subnet-group-name "${DB_SUBNET_GROUP}" \
|
||||
--db-parameter-group-name "${DB_PARAM_GROUP}" \
|
||||
--vpc-security-group-ids "${DB_SG}" \
|
||||
--no-publicly-accessible \
|
||||
--storage-encrypted >/dev/null
|
||||
rm -f "${pw_json}"
|
||||
fi
|
||||
|
||||
log "Waiting for ${DB_INSTANCE} to become available (first creation takes ~10 min)"
|
||||
aws rds wait db-instance-available --db-instance-identifier "${DB_INSTANCE}"
|
||||
DB_HOST="$(aws rds describe-db-instances --db-instance-identifier "${DB_INSTANCE}" \
|
||||
--query 'DBInstances[0].Endpoint.Address' --output text)"
|
||||
|
||||
# ---- 5 Connection string + JWT secret -> Secrets Manager --------------------
|
||||
# No per-secret IAM grants are needed: the execution role's read-gateway-secrets
|
||||
# policy (§2) names each of the three secrets by its ARN prefix.
|
||||
# Secret values go to aws via --cli-input-json temp files, never argv.
|
||||
if [[ -n "${DB_PASSWORD}" ]]; then
|
||||
# RDS private endpoint (guide §3); the gateway connects directly over the
|
||||
# VPC — the DB security group only admits ${GW_SG_NAME}.
|
||||
# sslmode=verify-full: the gateway's driver honors sslmode from the URL and
|
||||
# verifies the RDS certificate chain AND hostname against the CA bundle the
|
||||
# image trusts via NODE_EXTRA_CA_CERTS (see the Dockerfile). Do NOT add a
|
||||
# libpq-style `sslrootcert=` query param — the driver doesn't read it and
|
||||
# forwards it to Postgres as a startup parameter, which the server rejects.
|
||||
CONN="postgres://${DB_USER}:${DB_PASSWORD}@${DB_HOST}:5432/${DB_NAME}?sslmode=verify-full"
|
||||
log "Storing connection string in Secrets Manager secret ${SECRET_NAME} (§5)"
|
||||
conn_json=""; secret_json conn_json SecretString "${CONN}"
|
||||
if secret_exists "${SECRET_NAME}"; then
|
||||
aws secretsmanager put-secret-value --secret-id "${SECRET_NAME}" \
|
||||
--cli-input-json "file://${conn_json}" >/dev/null
|
||||
else
|
||||
aws secretsmanager create-secret --name "${SECRET_NAME}" \
|
||||
--cli-input-json "file://${conn_json}" >/dev/null
|
||||
fi
|
||||
rm -f "${conn_json}"
|
||||
else
|
||||
log "Skipping postgres-url secret write (instance already existed, password not available this run)"
|
||||
fi
|
||||
|
||||
# JWT signing secret — generated once (re-runs do NOT rotate it).
|
||||
log "Ensuring JWT signing secret ${JWT_SECRET_NAME} (§5)"
|
||||
if secret_exists "${JWT_SECRET_NAME}"; then
|
||||
skip "secret ${JWT_SECRET_NAME}"
|
||||
else
|
||||
jwt_json=""; secret_json jwt_json SecretString "$(openssl rand -base64 32)"
|
||||
aws secretsmanager create-secret --name "${JWT_SECRET_NAME}" \
|
||||
--cli-input-json "file://${jwt_json}" >/dev/null
|
||||
rm -f "${jwt_json}"
|
||||
fi
|
||||
|
||||
# OIDC client secret — operator-created (the script can't generate it; it comes
|
||||
# from the Okta OIDC web application). Checked here so the deploy step below can
|
||||
# gate on it with a clear message instead of a raw ECS secret-injection failure.
|
||||
OIDC_ARN="$(secret_arn "${OIDC_SECRET_NAME}")"
|
||||
|
||||
# ---- 7 ECS Fargate service + internal ALB ----------------------------------
|
||||
# Self-gating: deploy only once its inputs exist (image pushed — i.e.
|
||||
# gateway.yaml was filled in — plus the operator-provided OIDC client secret
|
||||
# and the ACM certificate for the internal hostname). On a first run these are
|
||||
# usually missing and it cleanly skips.
|
||||
ALB_DNS=""
|
||||
missing=""
|
||||
[[ -n "${IMAGE}" ]] || missing="${missing} image(fill ${GATEWAY_YAML})"
|
||||
[[ -n "${OIDC_ARN}" ]] || missing="${missing} ${OIDC_SECRET_NAME}"
|
||||
[[ -n "${ACM_CERT_ARN}" ]] || missing="${missing} ACM_CERT_ARN"
|
||||
SECRET_ARN="$(secret_arn "${SECRET_NAME}")"
|
||||
JWT_ARN="$(secret_arn "${JWT_SECRET_NAME}")"
|
||||
[[ -n "${SECRET_ARN}" ]] || missing="${missing} ${SECRET_NAME}"
|
||||
[[ -n "${JWT_ARN}" ]] || missing="${missing} ${JWT_SECRET_NAME}"
|
||||
|
||||
if [[ "${DEPLOY}" != "1" ]]; then
|
||||
log "Skipping ECS/ALB deploy (DEPLOY=${DEPLOY}) (§7)"
|
||||
elif [[ -n "${missing// }" ]]; then
|
||||
log "Skipping ECS/ALB deploy — missing input(s):${missing} (§7)"
|
||||
echo " Fill ${GATEWAY_YAML} and re-run to build the image; create ${OIDC_SECRET_NAME}"
|
||||
echo " from the Okta client secret; set ACM_CERT_ARN to the certificate for your"
|
||||
echo " internal gateway hostname. Then re-run to deploy."
|
||||
else
|
||||
log "Creating ECS cluster ${CLUSTER} and log group ${LOG_GROUP} (§7)"
|
||||
if [[ "$(aws ecs describe-clusters --clusters "${CLUSTER}" \
|
||||
--query 'clusters[0].status' --output text 2>/dev/null)" == "ACTIVE" ]]; then
|
||||
skip "cluster ${CLUSTER}"
|
||||
else
|
||||
aws ecs create-cluster --cluster-name "${CLUSTER}" >/dev/null
|
||||
fi
|
||||
# The gateway's stderr carries both its audit events and operational logs.
|
||||
if aws logs describe-log-groups --log-group-name-prefix "${LOG_GROUP}" \
|
||||
--query 'logGroups[?logGroupName==`'"${LOG_GROUP}"'`]' --output text 2>/dev/null | grep -q .; then
|
||||
skip "log group ${LOG_GROUP}"
|
||||
else
|
||||
aws logs create-log-group --log-group-name "${LOG_GROUP}"
|
||||
fi
|
||||
# Retention is a separate API (create-log-group has no retention flag) and an
|
||||
# upsert — applied every run so pre-existing groups converge too. Without it
|
||||
# the group keeps logs forever and cost grows unbounded.
|
||||
aws logs put-retention-policy --log-group-name "${LOG_GROUP}" \
|
||||
--retention-in-days "${LOG_RETENTION_DAYS}"
|
||||
|
||||
# Task definition: the task role carries the Bedrock permission; the
|
||||
# execution role injects the secrets. Registering is an append (a new
|
||||
# revision) — the service below always points at the latest.
|
||||
log "Registering task definition ${TASK_FAMILY}"
|
||||
taskdef_tmp="$(mktemp)"
|
||||
cat > "${taskdef_tmp}" <<EOF
|
||||
{
|
||||
"family": "${TASK_FAMILY}",
|
||||
"networkMode": "awsvpc",
|
||||
"requiresCompatibilities": ["FARGATE"],
|
||||
"cpu": "${TASK_CPU}",
|
||||
"memory": "${TASK_MEMORY}",
|
||||
"runtimePlatform": { "cpuArchitecture": "X86_64", "operatingSystemFamily": "LINUX" },
|
||||
"executionRoleArn": "arn:aws:iam::${ACCOUNT_ID}:role/${EXEC_ROLE}",
|
||||
"taskRoleArn": "arn:aws:iam::${ACCOUNT_ID}:role/${TASK_ROLE}",
|
||||
"containerDefinitions": [
|
||||
{
|
||||
"name": "gateway",
|
||||
"image": "${IMAGE}",
|
||||
"portMappings": [{ "containerPort": 8080 }],
|
||||
"secrets": [
|
||||
{ "name": "GATEWAY_JWT_SECRET", "valueFrom": "${JWT_ARN}" },
|
||||
{ "name": "OIDC_CLIENT_SECRET", "valueFrom": "${OIDC_ARN}" },
|
||||
{ "name": "GATEWAY_POSTGRES_URL", "valueFrom": "${SECRET_ARN}" }
|
||||
],
|
||||
"logConfiguration": {
|
||||
"logDriver": "awslogs",
|
||||
"options": {
|
||||
"awslogs-group": "${LOG_GROUP}",
|
||||
"awslogs-region": "${AWS_REGION}",
|
||||
"awslogs-stream-prefix": "gateway"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
EOF
|
||||
aws ecs register-task-definition --cli-input-json "file://${taskdef_tmp}" >/dev/null
|
||||
rm -f "${taskdef_tmp}"
|
||||
|
||||
# Internal ALB. --ip-address-type ipv4: an internal dual-stack ALB publishes
|
||||
# public-range AAAA records, which the CLI's /login private-network check
|
||||
# rejects.
|
||||
log "Creating internal ALB ${ALB_NAME} + target group + HTTPS listener"
|
||||
read -r ALB_ARN ALB_SCHEME ALB_VPC ALB_IP_TYPE <<<"$(aws elbv2 describe-load-balancers --names "${ALB_NAME}" \
|
||||
--query 'LoadBalancers[0].[LoadBalancerArn,Scheme,VpcId,IpAddressType]' --output text 2>/dev/null || true)"
|
||||
if [[ -n "${ALB_ARN}" && "${ALB_ARN}" != "None" ]]; then
|
||||
# Reuse is by name, and scheme/VPC are immutable on an ALB — so posture is
|
||||
# asserted, fail-closed: attaching the gateway to an internet-facing or
|
||||
# wrong-VPC load balancer would change the exposure model, not just drift.
|
||||
if [[ "${ALB_SCHEME}" != "internal" || "${ALB_VPC}" != "${VPC_ID}" ]]; then
|
||||
echo "ERROR: load balancer ${ALB_NAME} exists but is not the internal ALB this script expects:" >&2
|
||||
echo " scheme=${ALB_SCHEME} (need internal), vpc=${ALB_VPC} (need ${VPC_ID})." >&2
|
||||
echo " Refusing to deploy the gateway behind it. Delete that load balancer, or set" >&2
|
||||
echo " ALB_NAME to an unused name, then re-run." >&2
|
||||
exit 1
|
||||
fi
|
||||
skip "load balancer ${ALB_NAME} (internal, ${ALB_VPC})"
|
||||
# ip-address-type IS mutable (unlike scheme/VPC) — converge a reused
|
||||
# dualstack ALB back to ipv4, matching the Terraform sibling: dual-stack
|
||||
# publishes public-range AAAA records that /login rejects (see above).
|
||||
if [[ "${ALB_IP_TYPE}" != "ipv4" ]]; then
|
||||
aws elbv2 set-ip-address-type --load-balancer-arn "${ALB_ARN}" \
|
||||
--ip-address-type ipv4 >/dev/null
|
||||
fi
|
||||
else
|
||||
# shellcheck disable=SC2086
|
||||
ALB_ARN="$(aws elbv2 create-load-balancer --name "${ALB_NAME}" \
|
||||
--scheme internal --type application --ip-address-type ipv4 \
|
||||
--subnets ${PRIVATE_SUBNETS} --security-groups "${ALB_SG}" \
|
||||
--query 'LoadBalancers[0].LoadBalancerArn' --output text)"
|
||||
fi
|
||||
|
||||
# The ALB closes a connection after 60 seconds with no data by default, which
|
||||
# cuts off streams during quiet periods (long prompt processing before the
|
||||
# first token, extended thinking). Attribute setting is idempotent.
|
||||
aws elbv2 modify-load-balancer-attributes --load-balancer-arn "${ALB_ARN}" \
|
||||
--attributes Key=idle_timeout.timeout_seconds,Value=3600 >/dev/null
|
||||
|
||||
read -r TG_ARN TG_VPC <<<"$(aws elbv2 describe-target-groups --names "${TG_NAME}" \
|
||||
--query 'TargetGroups[0].[TargetGroupArn,VpcId]' --output text 2>/dev/null || true)"
|
||||
if [[ -n "${TG_ARN}" && "${TG_ARN}" != "None" ]]; then
|
||||
skip "target group ${TG_NAME}"
|
||||
# VPC is immutable on a target group; a wrong-VPC one can't reach the tasks.
|
||||
if [[ "${TG_VPC}" != "${VPC_ID}" ]]; then
|
||||
echo " WARN — target group ${TG_NAME} is in ${TG_VPC}, not ${VPC_ID}; the service's tasks" >&2
|
||||
echo " will not become healthy behind it. Delete it or set TG_NAME to an unused" >&2
|
||||
echo " name, then re-run." >&2
|
||||
fi
|
||||
else
|
||||
# /readyz verifies the store is reachable, so a task that can't reach
|
||||
# Postgres never enters rotation (the gateway also serves liveness-only
|
||||
# /healthz — see the deploy guide's outage-behavior tradeoff).
|
||||
TG_ARN="$(aws elbv2 create-target-group --name "${TG_NAME}" \
|
||||
--protocol HTTP --port 8080 --vpc-id "${VPC_ID}" --target-type ip \
|
||||
--health-check-path /readyz \
|
||||
--query 'TargetGroups[0].TargetGroupArn' --output text)"
|
||||
fi
|
||||
|
||||
# Select the HTTPS:443 listener specifically — a reused ALB may carry other
|
||||
# listeners (say HTTP:80); those stay untouched, and the 443 listener is
|
||||
# still created when it's the one that's missing.
|
||||
# shellcheck disable=SC2016 # backticks are JMESPath literals, not expansion
|
||||
LISTENER_ARN="$(aws elbv2 describe-listeners --load-balancer-arn "${ALB_ARN}" \
|
||||
--query 'Listeners[?Port==`443`]|[0].ListenerArn' --output text 2>/dev/null || true)"
|
||||
if [[ -n "${LISTENER_ARN}" && "${LISTENER_ARN}" != "None" ]]; then
|
||||
skip "HTTPS:443 listener on ${ALB_NAME}"
|
||||
# Converge everything this script owns on pre-existing listeners
|
||||
# (modify-listener is an upsert): the TLS policy (so re-runs pick up an
|
||||
# ALB_SSL_POLICY change, and listeners created before this script pinned
|
||||
# one lose the legacy default), the certificate (so a changed ACM_CERT_ARN
|
||||
# — e.g. a renewal under a new ARN — is not silently ignored), and the
|
||||
# default action (so the listener always forwards to this target group).
|
||||
aws elbv2 modify-listener --listener-arn "${LISTENER_ARN}" \
|
||||
--ssl-policy "${ALB_SSL_POLICY}" \
|
||||
--certificates "CertificateArn=${ACM_CERT_ARN}" \
|
||||
--default-actions "Type=forward,TargetGroupArn=${TG_ARN}" >/dev/null
|
||||
else
|
||||
aws elbv2 create-listener --load-balancer-arn "${ALB_ARN}" \
|
||||
--protocol HTTPS --port 443 \
|
||||
--ssl-policy "${ALB_SSL_POLICY}" \
|
||||
--certificates "CertificateArn=${ACM_CERT_ARN}" \
|
||||
--default-actions "Type=forward,TargetGroupArn=${TG_ARN}" >/dev/null
|
||||
fi
|
||||
|
||||
# Service: created once, then rolled forward — a re-run points it at the
|
||||
# latest task-definition revision (which carries the current image tag, and
|
||||
# therefore the current gateway.yaml) and forces a new deployment.
|
||||
log "Creating/updating ECS service ${SERVICE} (Fargate, private subnets, no public IP)"
|
||||
svc_status="$(aws ecs describe-services --cluster "${CLUSTER}" --services "${SERVICE}" \
|
||||
--query 'services[0].status' --output text 2>/dev/null || true)"
|
||||
if [[ "${svc_status}" == "ACTIVE" ]]; then
|
||||
aws ecs update-service --cluster "${CLUSTER}" --service "${SERVICE}" \
|
||||
--task-definition "${TASK_FAMILY}" --desired-count "${DESIRED_COUNT}" \
|
||||
--deployment-configuration "deploymentCircuitBreaker={enable=true,rollback=true}" \
|
||||
--health-check-grace-period-seconds 60 \
|
||||
--force-new-deployment >/dev/null
|
||||
echo " service updated to the latest task-definition revision."
|
||||
else
|
||||
# All egress (Bedrock, the IdP, Secrets Manager, ECR, CloudWatch Logs) goes
|
||||
# through the NAT gateway — assignPublicIp stays DISABLED.
|
||||
# The deployment circuit breaker stops a rollout whose tasks keep failing
|
||||
# (bad image, unbootable config) and rolls back to the last steady state
|
||||
# instead of relaunching failing tasks forever. The health-check grace
|
||||
# period gives a cold task (image pull + store connect + first /readyz)
|
||||
# time before ECS counts it unhealthy — without it the circuit breaker can
|
||||
# declare the very first rollout failed (matches terraform/'s
|
||||
# health_check_grace_period_seconds).
|
||||
aws ecs create-service --cluster "${CLUSTER}" --service-name "${SERVICE}" \
|
||||
--task-definition "${TASK_FAMILY}" --desired-count "${DESIRED_COUNT}" \
|
||||
--launch-type FARGATE \
|
||||
--deployment-configuration "deploymentCircuitBreaker={enable=true,rollback=true}" \
|
||||
--health-check-grace-period-seconds 60 \
|
||||
--network-configuration "awsvpcConfiguration={subnets=[${SUBNETS_CSV}],securityGroups=[${GW_SG}],assignPublicIp=DISABLED}" \
|
||||
--load-balancers "targetGroupArn=${TG_ARN},containerName=gateway,containerPort=8080" >/dev/null
|
||||
fi
|
||||
|
||||
ALB_DNS="$(aws elbv2 describe-load-balancers --load-balancer-arns "${ALB_ARN}" \
|
||||
--query 'LoadBalancers[0].DNSName' --output text)"
|
||||
log "Internal ALB DNS: ${ALB_DNS}"
|
||||
|
||||
# Post-deploy smoke check: the ALB is internal (unreachable from this
|
||||
# machine), but target health is visible through the API — poll until the
|
||||
# /readyz health check passes. Non-fatal; a cold task needs a minute or two
|
||||
# (image pull + store connect).
|
||||
log "Smoke check: polling target health on ${TG_NAME} (health check: GET /readyz)"
|
||||
tg_state="unknown"
|
||||
for _ in $(seq 1 24); do
|
||||
tg_state="$(aws elbv2 describe-target-health --target-group-arn "${TG_ARN}" \
|
||||
--query 'TargetHealthDescriptions[0].TargetHealth.State' --output text 2>/dev/null || true)"
|
||||
[[ "${tg_state}" == "healthy" ]] && break
|
||||
sleep 10
|
||||
done
|
||||
if [[ "${tg_state}" == "healthy" ]]; then
|
||||
echo " OK — a gateway task is healthy behind the ALB (store reachable)."
|
||||
else
|
||||
echo " WARN — last target state: ${tg_state:-none}; the task may still be starting."
|
||||
echo " Check the service events and the gateway's logs:"
|
||||
echo " aws ecs describe-services --cluster ${CLUSTER} --services ${SERVICE} --query 'services[0].events[:5]'"
|
||||
echo " aws logs tail ${LOG_GROUP} --since 10m"
|
||||
fi
|
||||
|
||||
# public_url is baked into the image, so verify the operator's chosen
|
||||
# hostname is in place (the redirect URI and discovery doc derive from it).
|
||||
CFG_PUBLIC_URL="$(grep -E '^[[:space:]]*public_url:' "${GATEWAY_YAML}" 2>/dev/null \
|
||||
| head -1 \
|
||||
| sed -E 's/^[[:space:]]*public_url:[[:space:]]*//; s/[[:space:]]+#.*$//; s/[[:space:]]*$//' \
|
||||
|| true)"
|
||||
CFG_PUBLIC_URL="${CFG_PUBLIC_URL#[\'\"]}"; CFG_PUBLIC_URL="${CFG_PUBLIC_URL%[\'\"]}"
|
||||
CFG_PUBLIC_URL="${CFG_PUBLIC_URL%/}"
|
||||
echo " 1. In your Route 53 private hosted zone, alias the host of"
|
||||
echo " ${CFG_PUBLIC_URL:-<public_url>} to the ALB: ${ALB_DNS}"
|
||||
echo " (the ALB's own *.elb.amazonaws.com name can't carry your ACM certificate)."
|
||||
echo " 2. Register this redirect URI on the Okta OIDC web app: ${CFG_PUBLIC_URL:-<public_url>}/oauth/callback"
|
||||
echo " 3. Verify from inside your corporate network:"
|
||||
echo " curl -s ${CFG_PUBLIC_URL:-<public_url>}/.well-known/oauth-authorization-server"
|
||||
fi
|
||||
|
||||
# ---- summary ----------------------------------------------------------------
|
||||
cat <<EOF
|
||||
|
||||
==> Done.
|
||||
|
||||
Security groups ${ALB_SG_NAME}=${ALB_SG} ${GW_SG_NAME}=${GW_SG} ${DB_SG_NAME}=${DB_SG}
|
||||
IAM roles ${TASK_ROLE} (bedrock-invoke), ${EXEC_ROLE} (pull + secrets)
|
||||
Image ${IMAGE:-(not built yet — fill ${GATEWAY_YAML})}
|
||||
RDS instance ${DB_INSTANCE} -> ${DB_HOST}
|
||||
Database / user ${DB_NAME} / ${DB_USER}
|
||||
Secrets ${SECRET_NAME}, ${JWT_SECRET_NAME}, ${OIDC_SECRET_NAME}$( [[ -n "${OIDC_ARN}" ]] || printf ' (MISSING — create it)' )
|
||||
ECS service ${CLUSTER}/${SERVICE} behind ${ALB_DNS:-(not deployed yet)}
|
||||
|
||||
Next steps (see https://code.claude.com/docs/en/claude-apps-gateway-on-aws):
|
||||
- Create the one operator-provided secret (from the Okta OIDC web app). Put the
|
||||
client secret in a 0600 file first — passing it as a literal argument would
|
||||
leave it readable in the process table and in audit/EDR logs:
|
||||
aws secretsmanager create-secret --name ${OIDC_SECRET_NAME} \\
|
||||
--secret-string file:///path/to/okta-client-secret.txt
|
||||
- Fill in the REPLACE_ME values in ${GATEWAY_YAML}, then re-run: setup.sh builds the
|
||||
image (config baked in) and deploys once the secret and ACM_CERT_ARN exist.
|
||||
- Enable Bedrock model access in the console for the Claude models you need (per
|
||||
region the us.anthropic.* profiles span) and submit the one-time use case form.
|
||||
- Alias your internal hostname (gateway.yaml public_url) to the ALB in a Route 53
|
||||
private hosted zone, and register <public_url>/oauth/callback on the Okta app.
|
||||
- The gateway runs its own schema migrations at boot, so ${DB_USER} needs CREATE TABLE.
|
||||
EOF
|
||||
19
examples/gateway/aws/terraform/.gitignore
vendored
Normal file
19
examples/gateway/aws/terraform/.gitignore
vendored
Normal file
@@ -0,0 +1,19 @@
|
||||
# Never commit state (contains secrets) or local var files
|
||||
*.tfstate
|
||||
*.tfstate.*
|
||||
.terraform/
|
||||
terraform.tfvars
|
||||
*.auto.tfvars
|
||||
crash.log
|
||||
|
||||
# The lock file holds no secrets. It's ignored here so consumers who copy this
|
||||
# example into their own repo generate (and commit) their own platform-complete
|
||||
# lock at first init — committing one from this repo would carry only one
|
||||
# platform's provider hashes. In your copy, drop this line and commit the lock
|
||||
# produced by:
|
||||
# terraform providers lock -platform=linux_amd64 -platform=linux_arm64 \
|
||||
# -platform=darwin_amd64 -platform=darwin_arm64 -platform=windows_amd64
|
||||
# versions.tf pins by range only, so without a committed lock the registry
|
||||
# serves the newest in-range build; a platform-complete lock gives hash
|
||||
# continuity across machines/CI and makes provider upgrades reviewable diffs.
|
||||
.terraform.lock.hcl
|
||||
182
examples/gateway/aws/terraform/README.md
Normal file
182
examples/gateway/aws/terraform/README.md
Normal file
@@ -0,0 +1,182 @@
|
||||
# Claude apps gateway — Terraform (ECS Fargate)
|
||||
|
||||
Terraform equivalent of `../setup.sh`. Lets end-users provision and manage
|
||||
the gateway with `terraform apply`. Covers the same scope ([walkthrough](https://code.claude.com/docs/en/claude-apps-gateway-on-aws) §1–7,
|
||||
ECS track): security groups → task + execution IAM roles → ECR repository →
|
||||
private-subnet RDS for PostgreSQL → Secrets Manager secrets → ECS Fargate
|
||||
service behind an internal ALB. The VPC and private subnets are walkthrough
|
||||
prerequisites, passed in as variables — unlike the GCP example, no network is
|
||||
created here.
|
||||
|
||||
## Files
|
||||
|
||||
| File | Purpose |
|
||||
|------|---------|
|
||||
| `versions.tf` | Provider pins (aws, random) |
|
||||
| `variables.tf` | All inputs (defaults match `setup.sh`'s) |
|
||||
| `main.tf` | Resources |
|
||||
| `outputs.tf` | ALB DNS name + zone ID, image, roles, DB endpoint |
|
||||
| `terraform.tfvars.example` | Copy to `terraform.tfvars` and edit |
|
||||
|
||||
## Prerequisites
|
||||
|
||||
1. **`../gateway.yaml` created and FULLY filled in** — copy the template first:
|
||||
`cp ../gateway.yaml.example ../gateway.yaml`, then replace every `REPLACE_ME`
|
||||
(Terraform reads this file and enforces no `REPLACE_ME` via a precondition).
|
||||
Unlike the GCP example there is no placeholder-first-pass: the config is
|
||||
**baked into the image**, and `public_url` is your own internal hostname,
|
||||
which you choose up front (you already hold its ACM certificate).
|
||||
`gateway.yaml` is gitignored; the committed template is `gateway.yaml.example`.
|
||||
2. The **prebuilt linux-x64 `claude` binary at `../claude`** — the Claude Code
|
||||
release binary, which includes the `gateway` subcommand (see the
|
||||
[walkthrough](https://code.claude.com/docs/en/claude-apps-gateway-on-aws)).
|
||||
See `../setup.sh`'s `DIST_URL`/`DIST_SHA256` download path for a
|
||||
checksum-verified fetch.
|
||||
3. A **VPC with two+ private subnets** in different AZs and NAT egress, an **ACM
|
||||
certificate** for your internal gateway hostname, and **Bedrock model access**
|
||||
enabled in the console (cross-region `us.anthropic.*` profiles need it in each
|
||||
region the profile spans), with the one-time use case form submitted.
|
||||
4. A **remote backend** for shared use (see below). State holds secrets — never commit it.
|
||||
|
||||
## Deploy
|
||||
|
||||
Terraform creates the ECR repository but does **not** build/push the image, so
|
||||
the apply is two passes: a targeted apply to create the repo, then build/push,
|
||||
then the full apply.
|
||||
|
||||
```bash
|
||||
cp terraform.tfvars.example terraform.tfvars # edit it
|
||||
terraform init
|
||||
|
||||
# Pin providers in your copy (once, then commit .terraform.lock.hcl and drop
|
||||
# its .gitignore line): versions.tf pins by range only, so without a committed
|
||||
# lock the registry serves the newest in-range build — a platform-complete
|
||||
# lock gives hash continuity across machines/CI and makes provider upgrades
|
||||
# reviewable diffs.
|
||||
terraform providers lock -platform=linux_amd64 -platform=linux_arm64 \
|
||||
-platform=darwin_amd64 -platform=darwin_arm64 -platform=windows_amd64
|
||||
|
||||
# 1. Create just the ECR repository (the -target warning is expected):
|
||||
terraform apply -target=aws_ecr_repository.repo
|
||||
|
||||
# 2. Build and push the image (gateway.yaml and the RDS CA bundle are baked in;
|
||||
# the COPY sources are context-relative — the build context `..` is aws/, so
|
||||
# `claude`, `gateway.yaml`, and `rds-global-bundle.pem`).
|
||||
# The CA bundle is the trust anchor for the connection string's
|
||||
# sslmode=verify-full (AWS rotates it; download it when absent — don't commit it):
|
||||
curl -fL --proto '=https' -o ../rds-global-bundle.pem \
|
||||
https://truststore.pki.rds.amazonaws.com/global/global-bundle.pem
|
||||
aws ecr get-login-password --region us-east-1 \
|
||||
| docker login --username AWS --password-stdin <account-id>.dkr.ecr.us-east-1.amazonaws.com
|
||||
docker build --platform=linux/amd64 --provenance=false \
|
||||
-f ../Dockerfile --build-arg CLAUDE_BINARY=claude --build-arg GATEWAY_CONFIG=gateway.yaml \
|
||||
-t <account-id>.dkr.ecr.us-east-1.amazonaws.com/claude-gateway:<version> ..
|
||||
docker push <account-id>.dkr.ecr.us-east-1.amazonaws.com/claude-gateway:<version>
|
||||
|
||||
# 3. Full apply:
|
||||
terraform apply
|
||||
```
|
||||
|
||||
Set in `terraform.tfvars`:
|
||||
|
||||
- `region`, `vpc_id`, `private_subnet_ids`, `corporate_cidr`
|
||||
- `acm_certificate_arn` — the certificate for your internal gateway hostname
|
||||
(`gateway.yaml`'s `public_url` host), served by the ALB's HTTPS listener
|
||||
- `image_tag` (after building/pushing — step 2 above). The repo enforces
|
||||
**immutable tags**, so a `gateway.yaml` edit means a rebuild under a **new**
|
||||
tag and an `image_tag` bump (`../setup.sh` automates this by tagging
|
||||
`<version>-cfg<sha8-of-gateway.yaml>`)
|
||||
- **`oidc_client_secret`** — required (the ECS tasks inject `latest` of this
|
||||
secret at start; with no version they fail with
|
||||
`ResourceInitializationError`). Terraform creates the secret + version from it.
|
||||
|
||||
## Tear down
|
||||
|
||||
Tear down a trial with `terraform destroy`: set `deletion_protection = false`,
|
||||
run `terraform apply` to record that on RDS and the ALB (and to flip RDS to
|
||||
`skip_final_snapshot` — the provider checks the value in **state**, not config,
|
||||
so destroy would still refuse otherwise), then `terraform destroy`.
|
||||
|
||||
The same switch drives the Secrets Manager recovery window: the three secrets
|
||||
have **fixed names**, and a secret deleted with the default 30-day recovery
|
||||
window keeps its name reserved — a later `terraform apply` would fail with a
|
||||
name conflict until the window elapses. With `deletion_protection = false` the
|
||||
destroy deletes them immediately (`recovery_window_in_days = 0`). If you
|
||||
destroyed a deployment that still had `deletion_protection = true` (or tore
|
||||
down an older copy of this module), clear the scheduled deletions before
|
||||
re-applying:
|
||||
|
||||
```bash
|
||||
for s in gateway-postgres-url gateway-jwt-secret gateway-oidc-client-secret; do
|
||||
aws secretsmanager delete-secret --secret-id "$s" --force-delete-without-recovery
|
||||
done
|
||||
```
|
||||
|
||||
## Guard rails
|
||||
|
||||
Tuned so accidental deletion is hard but greenfield teardown stays easy:
|
||||
|
||||
- `deletion_protection = true` (variable, default true) on RDS and the ALB —
|
||||
blocks accidental deletion; set `false` when you intend to `terraform destroy`.
|
||||
The same switch controls RDS `skip_final_snapshot`, so a protected instance
|
||||
always leaves a final snapshot.
|
||||
- ECR tags are **immutable** and **scanned on push** — a deployed tag can never
|
||||
be silently re-pointed at different bytes. For production, also restrict push
|
||||
rights on the repo to your CI / image-promotion pipeline rather than operator
|
||||
credentials.
|
||||
- The IAM roles carry only the walkthrough's least-privilege documents: Bedrock
|
||||
invoke on the Anthropic model ARNs (task role) and `secretsmanager:GetSecretValue`
|
||||
on exactly the three secrets this module creates (by ARN) plus the AWS-managed
|
||||
ECS execution policy (execution role). Inline policies are scoped to these
|
||||
roles, so nothing else in the account is touched.
|
||||
- TLS everywhere it terminates: the ALB listener pins
|
||||
`ELBSecurityPolicy-TLS13-1-2-2021-06` (no TLS 1.0/1.1), and the store
|
||||
connection uses `sslmode=verify-full` against the RDS CA bundle baked into
|
||||
the image, with `rds.force_ssl=1` enforcing TLS server-side.
|
||||
|
||||
## Private access
|
||||
|
||||
The ALB is **internal** with `ip_address_type = "ipv4"` (a dual-stack internal
|
||||
ALB publishes public-range AAAA records, which the CLI's `/login`
|
||||
private-network check rejects), and its security group admits only
|
||||
`corporate_cidr` on 443. Reaching it from on-prem requires your existing
|
||||
routing into the VPC (Direct Connect / VPN) — **operator / network-team-owned**
|
||||
plumbing this module does not create.
|
||||
|
||||
After the apply, give developers a privately resolvable hostname: in a Route 53
|
||||
private hosted zone, alias the host of `gateway.yaml`'s `public_url` to the ALB
|
||||
(`alb_dns_name` / `alb_zone_id` outputs). The ALB's own `*.elb.amazonaws.com`
|
||||
name can't carry your ACM certificate, so use your own name.
|
||||
|
||||
The tasks run in the private subnets with no public IP; all egress (Bedrock,
|
||||
the IdP, Secrets Manager, ECR, CloudWatch Logs) goes through the NAT gateway.
|
||||
To keep Bedrock traffic off the public path, create a `bedrock-runtime`
|
||||
interface VPC endpoint and point the upstream's `base_url` at it (see
|
||||
`../gateway.yaml.example`); the IdP still needs internet egress.
|
||||
|
||||
## Remote state (recommended for teams)
|
||||
|
||||
Add a backend so state is shared and locked (and out of git):
|
||||
|
||||
```hcl
|
||||
# backend.tf
|
||||
terraform {
|
||||
backend "s3" {
|
||||
bucket = "<your-tf-state-bucket>"
|
||||
key = "claude-gateway/ecs"
|
||||
region = "us-east-1"
|
||||
use_lockfile = true # S3-native locking (Terraform >= 1.10); or set dynamodb_table
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## After deploy
|
||||
|
||||
- `terraform output alb_dns_name` / `alb_zone_id` — create the Route 53 alias.
|
||||
- Register `<public_url>/oauth/callback` on the Okta OIDC web app and make sure
|
||||
`../gateway.yaml` `public_url` matches the host you aliased.
|
||||
- Notes: Terraform does not build the image. To ship a new gateway version **or
|
||||
a config edit**, rerun the docker build/push under a new tag and bump
|
||||
`image_tag` — secrets-only rotations roll the service without a rebuild (the
|
||||
task definition stamps a hash of the managed secret values), but a
|
||||
`gateway.yaml` edit reaches the container only through the rebuilt image.
|
||||
510
examples/gateway/aws/terraform/main.tf
Normal file
510
examples/gateway/aws/terraform/main.tf
Normal file
@@ -0,0 +1,510 @@
|
||||
# Claude apps gateway on ECS Fargate — Terraform equivalent of setup.sh.
|
||||
# Section markers (§N) map to setup.sh and the walkthrough:
|
||||
# https://code.claude.com/docs/en/claude-apps-gateway-on-aws
|
||||
#
|
||||
# Unlike the GCP example this module does NOT create the network — the VPC and
|
||||
# private subnets are walkthrough prerequisites, passed in as variables.
|
||||
|
||||
data "aws_caller_identity" "current" {}
|
||||
data "aws_region" "current" {}
|
||||
|
||||
# Read (not created) so a typo'd VPC or subnet ID fails the plan up front
|
||||
# instead of half-applying.
|
||||
data "aws_vpc" "this" {
|
||||
id = var.vpc_id
|
||||
}
|
||||
|
||||
data "aws_subnet" "private" {
|
||||
for_each = toset(var.private_subnet_ids)
|
||||
id = each.value
|
||||
}
|
||||
|
||||
locals {
|
||||
config_path = var.gateway_config_path != "" ? var.gateway_config_path : "${path.module}/../gateway.yaml"
|
||||
gateway_config = file(local.config_path)
|
||||
image = "${aws_ecr_repository.repo.repository_url}:${var.image_tag}"
|
||||
}
|
||||
|
||||
# ── 1 Security groups ───────────────────────────────────────────────────────
|
||||
# Three groups chain the traffic path: corp network -> ALB :443, ALB ->
|
||||
# gateway :8080, gateway -> Postgres :5432. Nothing else is reachable.
|
||||
# Rules are separate resources (not inline) so they never fight other tooling.
|
||||
resource "aws_security_group" "alb" {
|
||||
name = "claude-gateway-alb"
|
||||
description = "Claude gateway ALB"
|
||||
vpc_id = var.vpc_id
|
||||
}
|
||||
|
||||
resource "aws_security_group" "gateway" {
|
||||
name = "claude-gateway-svc"
|
||||
description = "Claude gateway service"
|
||||
vpc_id = var.vpc_id
|
||||
}
|
||||
|
||||
resource "aws_security_group" "db" {
|
||||
name = "claude-gateway-db"
|
||||
description = "Claude gateway Postgres"
|
||||
vpc_id = var.vpc_id
|
||||
}
|
||||
|
||||
resource "aws_vpc_security_group_ingress_rule" "alb_https" {
|
||||
security_group_id = aws_security_group.alb.id
|
||||
description = "HTTPS from the corporate network"
|
||||
ip_protocol = "tcp"
|
||||
from_port = 443
|
||||
to_port = 443
|
||||
cidr_ipv4 = var.corporate_cidr
|
||||
}
|
||||
|
||||
resource "aws_vpc_security_group_ingress_rule" "gateway_from_alb" {
|
||||
security_group_id = aws_security_group.gateway.id
|
||||
description = "Gateway port from the ALB"
|
||||
ip_protocol = "tcp"
|
||||
from_port = 8080
|
||||
to_port = 8080
|
||||
referenced_security_group_id = aws_security_group.alb.id
|
||||
}
|
||||
|
||||
resource "aws_vpc_security_group_ingress_rule" "db_from_gateway" {
|
||||
security_group_id = aws_security_group.db.id
|
||||
description = "Postgres from the gateway"
|
||||
ip_protocol = "tcp"
|
||||
from_port = 5432
|
||||
to_port = 5432
|
||||
referenced_security_group_id = aws_security_group.gateway.id
|
||||
}
|
||||
|
||||
# Egress: the ALB only needs to reach its targets; the gateway needs the NAT
|
||||
# path out (Bedrock, the IdP, Secrets Manager, ECR, CloudWatch Logs) plus
|
||||
# Postgres. The DB group needs no egress (security groups are stateful).
|
||||
resource "aws_vpc_security_group_egress_rule" "alb_to_gateway" {
|
||||
security_group_id = aws_security_group.alb.id
|
||||
description = "Health checks + forwarding to gateway tasks"
|
||||
ip_protocol = "tcp"
|
||||
from_port = 8080
|
||||
to_port = 8080
|
||||
referenced_security_group_id = aws_security_group.gateway.id
|
||||
}
|
||||
|
||||
resource "aws_vpc_security_group_egress_rule" "gateway_all" {
|
||||
security_group_id = aws_security_group.gateway.id
|
||||
description = "Egress to Bedrock, the IdP, Secrets Manager, ECR, CloudWatch Logs, Postgres"
|
||||
ip_protocol = "-1"
|
||||
cidr_ipv4 = "0.0.0.0/0"
|
||||
}
|
||||
|
||||
# ── 2 IAM roles (least-privilege) ───────────────────────────────────────────
|
||||
# Task role: the gateway's runtime identity. Its ONLY permission is invoking
|
||||
# Claude models on Bedrock — the upstream's `auth: {}` resolves to this role
|
||||
# via the AWS default credential chain. The policy must cover both the
|
||||
# cross-region inference-profile ARNs and the underlying foundation-model ARNs.
|
||||
data "aws_iam_policy_document" "ecs_trust" {
|
||||
statement {
|
||||
effect = "Allow"
|
||||
actions = ["sts:AssumeRole"]
|
||||
principals {
|
||||
type = "Service"
|
||||
identifiers = ["ecs-tasks.amazonaws.com"]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
resource "aws_iam_role" "task" {
|
||||
name = var.task_role_name
|
||||
assume_role_policy = data.aws_iam_policy_document.ecs_trust.json
|
||||
}
|
||||
|
||||
resource "aws_iam_role_policy" "bedrock_invoke" {
|
||||
name = "bedrock-invoke"
|
||||
role = aws_iam_role.task.id
|
||||
policy = jsonencode({
|
||||
Version = "2012-10-17"
|
||||
Statement = [{
|
||||
Effect = "Allow"
|
||||
Action = ["bedrock:InvokeModel", "bedrock:InvokeModelWithResponseStream"]
|
||||
Resource = [
|
||||
"arn:aws:bedrock:${data.aws_region.current.region}:${data.aws_caller_identity.current.account_id}:inference-profile/us.anthropic.*",
|
||||
"arn:aws:bedrock:*::foundation-model/anthropic.*",
|
||||
]
|
||||
}]
|
||||
})
|
||||
|
||||
# The walkthrough is scoped to commercial US regions: this policy and the
|
||||
# gateway's built-in model catalog both use the us.anthropic.* geo-prefixed
|
||||
# cross-region inference profiles, which only exist in the commercial US
|
||||
# regions — an explicit list, not a `us-` prefix match, because GovCloud
|
||||
# (us-gov-*) and ISO (us-iso-*) regions share the prefix but live in
|
||||
# different AWS partitions where those profiles and this module's arn:aws:
|
||||
# ARNs are wrong. Anywhere else the deploy provisions fine and then every
|
||||
# model call fails. Other-region deploys must pin region-appropriate
|
||||
# profiles via a models: block in gateway.yaml (see the config reference's
|
||||
# models: guidance: https://code.claude.com/docs/en/claude-apps-gateway-config),
|
||||
# widen the inference-profile ARN geo prefix above, and set
|
||||
# allow_non_us_region = true.
|
||||
lifecycle {
|
||||
precondition {
|
||||
condition = var.allow_non_us_region || contains(["us-east-1", "us-east-2", "us-west-1", "us-west-2"], var.region)
|
||||
error_message = "region is not a commercial US region (GovCloud/ISO share the us- prefix but are different partitions), and this module's IAM policy and the built-in model catalog use the US-geo (us.anthropic.*) inference profiles. Pin your region's inference profiles in a models: block in gateway.yaml, adjust the bedrock-invoke ARN prefix, then set allow_non_us_region = true."
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Execution role: the ECS agent's identity — pulls the image from ECR and
|
||||
# injects the Secrets Manager values into the container; the gateway never
|
||||
# uses it. AmazonECSTaskExecutionRolePolicy covers the ECR pull + awslogs;
|
||||
# the inline policy adds read on exactly the three secrets this module
|
||||
# creates — their full ARNs, not a name-prefix wildcard, so nothing else
|
||||
# in a shared account (present or future) is readable through this role.
|
||||
resource "aws_iam_role" "execution" {
|
||||
name = var.execution_role_name
|
||||
assume_role_policy = data.aws_iam_policy_document.ecs_trust.json
|
||||
}
|
||||
|
||||
resource "aws_iam_role_policy_attachment" "execution_managed" {
|
||||
role = aws_iam_role.execution.name
|
||||
policy_arn = "arn:aws:iam::aws:policy/service-role/AmazonECSTaskExecutionRolePolicy"
|
||||
}
|
||||
|
||||
resource "aws_iam_role_policy" "secrets_read" {
|
||||
name = "read-gateway-secrets"
|
||||
role = aws_iam_role.execution.id
|
||||
policy = jsonencode({
|
||||
Version = "2012-10-17"
|
||||
Statement = [{
|
||||
Effect = "Allow"
|
||||
Action = "secretsmanager:GetSecretValue"
|
||||
Resource = [
|
||||
aws_secretsmanager_secret.jwt.arn,
|
||||
aws_secretsmanager_secret.oidc.arn,
|
||||
aws_secretsmanager_secret.postgres_url.arn,
|
||||
]
|
||||
}]
|
||||
})
|
||||
}
|
||||
|
||||
# ── 6 ECR repository ────────────────────────────────────────────────────────
|
||||
# NOTE: image build/push is a separate step (see README) — Terraform only makes
|
||||
# the repo. IMMUTABLE tags + scan-on-push: the ECS service pulls whatever this
|
||||
# repo serves under the deployed tag, so a pushed tag must never be silently
|
||||
# re-pointed. For production, also restrict push rights on this repo to your
|
||||
# CI / image-promotion pipeline rather than operator credentials.
|
||||
resource "aws_ecr_repository" "repo" {
|
||||
name = var.ecr_repo
|
||||
image_tag_mutability = "IMMUTABLE"
|
||||
image_scanning_configuration {
|
||||
scan_on_push = true
|
||||
}
|
||||
}
|
||||
|
||||
# ── 3 RDS for PostgreSQL (private subnets, no public address) ───────────────
|
||||
resource "aws_db_subnet_group" "db" {
|
||||
name = var.db_instance
|
||||
description = "Claude gateway"
|
||||
subnet_ids = var.private_subnet_ids
|
||||
}
|
||||
|
||||
# rds.force_ssl: reject plaintext connections server-side — the client-side
|
||||
# counterpart is sslmode=verify-full in the connection string (§5). The family
|
||||
# tracks the major version in var.db_engine_version.
|
||||
#
|
||||
# name_prefix + create_before_destroy: a major engine bump changes `family`,
|
||||
# which forces replacement — with a static name that deadlocks (the new group
|
||||
# can't be created under the taken name; the old can't be destroyed while the
|
||||
# live instance uses it: "parameter group is currently in use"). With this
|
||||
# shape the replacement group gets a fresh unique name, the instance is
|
||||
# repointed, then the old group is destroyed. (The subnet group above needs
|
||||
# neither: subnet_ids update in place and an engine bump never touches it.)
|
||||
resource "aws_db_parameter_group" "db" {
|
||||
name_prefix = "${var.db_instance}-"
|
||||
family = "postgres${split(".", var.db_engine_version)[0]}"
|
||||
description = "Claude gateway - require TLS on every connection"
|
||||
|
||||
parameter {
|
||||
name = "rds.force_ssl"
|
||||
value = "1"
|
||||
}
|
||||
|
||||
lifecycle {
|
||||
create_before_destroy = true
|
||||
}
|
||||
}
|
||||
|
||||
# URL-safe (alphanumeric) so it drops cleanly into the connection string.
|
||||
# nosemgrep: terraform-generic-secrets-in-state -- secrets in tfstate are inherent to TF; mitigated by the documented remote S3 backend (see README "Remote state")
|
||||
resource "random_password" "db" {
|
||||
length = 32
|
||||
special = false
|
||||
}
|
||||
|
||||
# nosemgrep: terraform-aws-secrets-in-state -- secrets in tfstate are inherent to TF; mitigated by the documented remote S3 backend (see README "Remote state")
|
||||
resource "aws_db_instance" "db" {
|
||||
identifier = var.db_instance
|
||||
engine = "postgres"
|
||||
engine_version = var.db_engine_version
|
||||
instance_class = var.db_instance_class
|
||||
allocated_storage = var.db_allocated_storage
|
||||
db_name = var.db_name
|
||||
username = var.db_user
|
||||
password = random_password.db.result
|
||||
db_subnet_group_name = aws_db_subnet_group.db.name
|
||||
parameter_group_name = aws_db_parameter_group.db.name
|
||||
vpc_security_group_ids = [aws_security_group.db.id]
|
||||
publicly_accessible = false
|
||||
storage_encrypted = true
|
||||
deletion_protection = var.deletion_protection
|
||||
# Greenfield teardown: skip the final snapshot only once deletion protection
|
||||
# is deliberately turned off (the same switch — see README "Tear down").
|
||||
skip_final_snapshot = !var.deletion_protection
|
||||
final_snapshot_identifier = "${var.db_instance}-final"
|
||||
}
|
||||
|
||||
# ── 5 Secrets Manager ───────────────────────────────────────────────────────
|
||||
# postgres-url: connection string built from the instance's private endpoint.
|
||||
# The execution role's policy (§2) grants read on these three secrets' ARNs
|
||||
# and nothing else.
|
||||
#
|
||||
# recovery_window_in_days rides the same switch as skip_final_snapshot: the
|
||||
# secrets have fixed names, so a destroy that leaves them in the default
|
||||
# 30-day scheduled-deletion state makes the next apply fail with a name
|
||||
# conflict. Greenfield teardown (deletion_protection = false) deletes them
|
||||
# immediately; a protected deployment keeps the 30-day recovery window.
|
||||
resource "aws_secretsmanager_secret" "postgres_url" {
|
||||
name = var.secret_name
|
||||
recovery_window_in_days = var.deletion_protection ? 30 : 0
|
||||
}
|
||||
|
||||
# sslmode=verify-full: the gateway's driver (Bun.SQL) honors sslmode from the
|
||||
# URL and verifies the server certificate chain AND hostname. The trust anchor
|
||||
# is the AWS RDS CA bundle baked into the image at /etc/claude/rds-global-bundle.pem
|
||||
# and loaded via NODE_EXTRA_CA_CERTS (see ../Dockerfile) — do NOT add a
|
||||
# libpq-style `sslrootcert=` query param: the driver doesn't read it and
|
||||
# forwards it to Postgres as a startup parameter, which the server rejects.
|
||||
# nosemgrep: terraform-aws-secrets-in-state -- secrets in tfstate are inherent to TF; mitigated by the documented remote S3 backend (see README "Remote state")
|
||||
resource "aws_secretsmanager_secret_version" "postgres_url" {
|
||||
secret_id = aws_secretsmanager_secret.postgres_url.id
|
||||
secret_string = "postgres://${var.db_user}:${random_password.db.result}@${aws_db_instance.db.address}:5432/${var.db_name}?sslmode=verify-full"
|
||||
}
|
||||
|
||||
# jwt: session signing key.
|
||||
# nosemgrep: terraform-generic-secrets-in-state -- secrets in tfstate are inherent to TF; mitigated by the documented remote S3 backend (see README "Remote state")
|
||||
resource "random_password" "jwt" {
|
||||
length = 48
|
||||
special = false
|
||||
}
|
||||
|
||||
resource "aws_secretsmanager_secret" "jwt" {
|
||||
name = var.jwt_secret_name
|
||||
recovery_window_in_days = var.deletion_protection ? 30 : 0 # see postgres_url
|
||||
}
|
||||
|
||||
# nosemgrep: terraform-aws-secrets-in-state -- secrets in tfstate are inherent to TF; mitigated by the documented remote S3 backend (see README "Remote state")
|
||||
resource "aws_secretsmanager_secret_version" "jwt" {
|
||||
secret_id = aws_secretsmanager_secret.jwt.id
|
||||
secret_string = random_password.jwt.result
|
||||
}
|
||||
|
||||
# oidc client secret: operator-provided (from the Okta OIDC web app).
|
||||
resource "aws_secretsmanager_secret" "oidc" {
|
||||
name = var.oidc_secret_name
|
||||
recovery_window_in_days = var.deletion_protection ? 30 : 0 # see postgres_url
|
||||
}
|
||||
|
||||
# nosemgrep: terraform-aws-secrets-in-state -- secrets in tfstate are inherent to TF; mitigated by the documented remote S3 backend (see README "Remote state")
|
||||
resource "aws_secretsmanager_secret_version" "oidc" {
|
||||
count = var.oidc_client_secret != "" ? 1 : 0
|
||||
secret_id = aws_secretsmanager_secret.oidc.id
|
||||
secret_string = var.oidc_client_secret
|
||||
}
|
||||
|
||||
# Warn (not block) at plan time when the OIDC secret value isn't set: the task
|
||||
# definition references the secret unconditionally, so an empty value with no
|
||||
# out-of-band version means the tasks fail to start late, at container init
|
||||
# (ResourceInitializationError). A warning (not a precondition) keeps the
|
||||
# documented out-of-band-version mode usable.
|
||||
check "oidc_client_secret_set" {
|
||||
assert {
|
||||
condition = var.oidc_client_secret != ""
|
||||
error_message = "oidc_client_secret is empty — set it in terraform.tfvars, or add a version to the gateway-oidc-client-secret secret out-of-band before applying (the ECS tasks inject it at start and will fail without one)."
|
||||
}
|
||||
}
|
||||
|
||||
# ── 7 ECS Fargate service + internal ALB ────────────────────────────────────
|
||||
resource "aws_ecs_cluster" "cluster" {
|
||||
name = var.cluster_name
|
||||
}
|
||||
|
||||
# The gateway's stderr carries both its audit events and operational logs.
|
||||
# Bounded retention — without it the group keeps logs forever and cost grows
|
||||
# unbounded; the default (90 days) is sized for audit-trail review windows.
|
||||
resource "aws_cloudwatch_log_group" "gateway" {
|
||||
name = var.log_group_name
|
||||
retention_in_days = var.log_retention_days
|
||||
}
|
||||
|
||||
# Task definition. gateway.yaml ships INSIDE the image (unlike the GCP example,
|
||||
# which mounts it from Secret Manager), so Terraform reads ../gateway.yaml only
|
||||
# to (a) enforce the no-REPLACE_ME guard before a deploy and (b) stamp a hash
|
||||
# of the config + every managed secret value into the container environment —
|
||||
# secrets are injected at task start, so rotating one (tainting
|
||||
# random_password.db ALTERs the DB password; a new oidc_client_secret) would
|
||||
# otherwise leave running tasks on stale values with nothing forcing a roll.
|
||||
# NOTE the hash only forces a roll; a config EDIT still reaches the container
|
||||
# only via a rebuilt image — push under a new tag (the repo enforces
|
||||
# immutability) and bump image_tag, or the roll redeploys the old config.
|
||||
resource "aws_ecs_task_definition" "gateway" {
|
||||
family = var.service_name
|
||||
network_mode = "awsvpc"
|
||||
requires_compatibilities = ["FARGATE"]
|
||||
cpu = tostring(var.task_cpu)
|
||||
memory = tostring(var.task_memory)
|
||||
execution_role_arn = aws_iam_role.execution.arn
|
||||
task_role_arn = aws_iam_role.task.arn
|
||||
|
||||
runtime_platform {
|
||||
cpu_architecture = "X86_64" # build the image linux/amd64; ARM64 for Graviton (see ../Dockerfile)
|
||||
operating_system_family = "LINUX"
|
||||
}
|
||||
|
||||
container_definitions = jsonencode([
|
||||
{
|
||||
name = "gateway"
|
||||
image = local.image
|
||||
portMappings = [{ containerPort = 8080 }]
|
||||
environment = [
|
||||
{
|
||||
name = "GATEWAY_CONFIG_SHA"
|
||||
value = substr(sha256(join("", [
|
||||
local.gateway_config,
|
||||
random_password.db.result,
|
||||
random_password.jwt.result,
|
||||
var.oidc_client_secret,
|
||||
])), 0, 16)
|
||||
},
|
||||
]
|
||||
secrets = [
|
||||
{ name = "GATEWAY_JWT_SECRET", valueFrom = aws_secretsmanager_secret.jwt.arn },
|
||||
{ name = "OIDC_CLIENT_SECRET", valueFrom = aws_secretsmanager_secret.oidc.arn },
|
||||
{ name = "GATEWAY_POSTGRES_URL", valueFrom = aws_secretsmanager_secret.postgres_url.arn },
|
||||
]
|
||||
logConfiguration = {
|
||||
logDriver = "awslogs"
|
||||
options = {
|
||||
awslogs-group = aws_cloudwatch_log_group.gateway.name
|
||||
awslogs-region = data.aws_region.current.region
|
||||
awslogs-stream-prefix = "gateway"
|
||||
}
|
||||
}
|
||||
}
|
||||
])
|
||||
|
||||
# Guard mirrors setup.sh's REPLACE_ME check (non-comment lines): the config
|
||||
# is baked into the image this task definition deploys, so a half-filled
|
||||
# gateway.yaml at apply time means the pushed image is half-filled too.
|
||||
lifecycle {
|
||||
precondition {
|
||||
condition = length([
|
||||
for line in split("\n", local.gateway_config) :
|
||||
line
|
||||
if !startswith(trimspace(line), "#") && strcontains(line, "REPLACE_ME")
|
||||
]) == 0
|
||||
error_message = "gateway.yaml still has REPLACE_ME on a non-comment line — fill it in (and rebuild/push the image) before applying."
|
||||
}
|
||||
}
|
||||
|
||||
depends_on = [
|
||||
aws_secretsmanager_secret_version.postgres_url,
|
||||
aws_secretsmanager_secret_version.jwt,
|
||||
]
|
||||
}
|
||||
|
||||
# Internal ALB. ip_address_type ipv4: an internal dual-stack ALB publishes
|
||||
# public-range AAAA records, which the CLI's /login private-network check
|
||||
# rejects. idle_timeout 3600: the 60-second default closes a streaming
|
||||
# response at the first quiet period (long prompt processing before the first
|
||||
# token, extended thinking with no streamed output).
|
||||
resource "aws_lb" "gateway" {
|
||||
name = var.service_name
|
||||
internal = true
|
||||
load_balancer_type = "application"
|
||||
ip_address_type = "ipv4"
|
||||
subnets = var.private_subnet_ids
|
||||
security_groups = [aws_security_group.alb.id]
|
||||
idle_timeout = 3600
|
||||
enable_deletion_protection = var.deletion_protection
|
||||
}
|
||||
|
||||
# /readyz verifies the store is reachable, so a task that can't reach Postgres
|
||||
# never enters rotation (the gateway also serves liveness-only /healthz — see
|
||||
# the deploy guide's outage-behavior tradeoff).
|
||||
resource "aws_lb_target_group" "gateway" {
|
||||
name = var.service_name
|
||||
protocol = "HTTP"
|
||||
port = 8080
|
||||
vpc_id = var.vpc_id
|
||||
target_type = "ip"
|
||||
|
||||
health_check {
|
||||
path = "/readyz"
|
||||
}
|
||||
}
|
||||
|
||||
resource "aws_lb_listener" "https" {
|
||||
load_balancer_arn = aws_lb.gateway.arn
|
||||
protocol = "HTTPS"
|
||||
port = 443
|
||||
# Explicit modern policy — omitting ssl_policy falls back to the legacy
|
||||
# ELBSecurityPolicy-2016-08 default, which still accepts TLS 1.0/1.1.
|
||||
ssl_policy = "ELBSecurityPolicy-TLS13-1-2-2021-06"
|
||||
certificate_arn = var.acm_certificate_arn
|
||||
|
||||
default_action {
|
||||
type = "forward"
|
||||
target_group_arn = aws_lb_target_group.gateway.arn
|
||||
}
|
||||
}
|
||||
|
||||
resource "aws_ecs_service" "gateway" {
|
||||
name = var.service_name
|
||||
cluster = aws_ecs_cluster.cluster.id
|
||||
task_definition = aws_ecs_task_definition.gateway.arn
|
||||
desired_count = var.desired_count
|
||||
launch_type = "FARGATE"
|
||||
|
||||
# Stop a rollout whose tasks keep failing (bad image, unbootable config) and
|
||||
# roll back to the last steady state instead of relaunching failing tasks
|
||||
# forever.
|
||||
deployment_circuit_breaker {
|
||||
enable = true
|
||||
rollback = true
|
||||
}
|
||||
|
||||
network_configuration {
|
||||
subnets = var.private_subnet_ids
|
||||
security_groups = [aws_security_group.gateway.id]
|
||||
# All egress (Bedrock, the IdP, Secrets Manager, ECR, CloudWatch Logs)
|
||||
# goes through the NAT gateway — tasks get no public IP.
|
||||
assign_public_ip = false
|
||||
}
|
||||
|
||||
load_balancer {
|
||||
target_group_arn = aws_lb_target_group.gateway.arn
|
||||
container_name = "gateway"
|
||||
container_port = 8080
|
||||
}
|
||||
|
||||
# Tasks register with the ALB at start — give a cold task (image pull +
|
||||
# store connect + first /readyz) time before ECS replaces it as unhealthy.
|
||||
health_check_grace_period_seconds = 60
|
||||
|
||||
# The listener must exist before targets register; the secrets must be
|
||||
# readable before the first task starts.
|
||||
depends_on = [
|
||||
aws_lb_listener.https,
|
||||
aws_iam_role_policy.secrets_read,
|
||||
aws_iam_role_policy_attachment.execution_managed,
|
||||
aws_secretsmanager_secret_version.postgres_url,
|
||||
aws_secretsmanager_secret_version.jwt,
|
||||
aws_secretsmanager_secret_version.oidc,
|
||||
aws_db_instance.db,
|
||||
]
|
||||
}
|
||||
34
examples/gateway/aws/terraform/outputs.tf
Normal file
34
examples/gateway/aws/terraform/outputs.tf
Normal file
@@ -0,0 +1,34 @@
|
||||
output "alb_dns_name" {
|
||||
description = "Internal ALB DNS name. Alias your gateway hostname (the host in gateway.yaml's public_url) to this in a Route 53 private hosted zone — the *.elb.amazonaws.com name itself can't carry your ACM certificate."
|
||||
value = aws_lb.gateway.dns_name
|
||||
}
|
||||
|
||||
output "alb_zone_id" {
|
||||
description = "ALB hosted zone ID, for the Route 53 alias record."
|
||||
value = aws_lb.gateway.zone_id
|
||||
}
|
||||
|
||||
output "image" {
|
||||
description = "Image the service runs (build/push this separately — see README)."
|
||||
value = local.image
|
||||
}
|
||||
|
||||
output "ecr_repository_url" {
|
||||
description = "ECR repository URL to push the gateway image to."
|
||||
value = aws_ecr_repository.repo.repository_url
|
||||
}
|
||||
|
||||
output "task_role_arn" {
|
||||
description = "Gateway runtime task role (Bedrock invoke)."
|
||||
value = aws_iam_role.task.arn
|
||||
}
|
||||
|
||||
output "execution_role_arn" {
|
||||
description = "ECS execution role (image pull + secret injection)."
|
||||
value = aws_iam_role.execution.arn
|
||||
}
|
||||
|
||||
output "db_endpoint" {
|
||||
description = "RDS private endpoint (host only; the connection string lives in the gateway-postgres-url secret)."
|
||||
value = aws_db_instance.db.address
|
||||
}
|
||||
27
examples/gateway/aws/terraform/terraform.tfvars.example
Normal file
27
examples/gateway/aws/terraform/terraform.tfvars.example
Normal file
@@ -0,0 +1,27 @@
|
||||
# Copy to terraform.tfvars and edit. terraform.tfvars is gitignored (see .gitignore).
|
||||
|
||||
region = "us-east-1" # a region where Bedrock serves the Claude models you need
|
||||
|
||||
# Prerequisite networking (NOT created by this module): the VPC and two+ private
|
||||
# subnets in different AZs with outbound internet via a NAT gateway.
|
||||
vpc_id = "vpc-..."
|
||||
private_subnet_ids = ["subnet-...a", "subnet-...b"]
|
||||
|
||||
# The only source the ALB admits on 443. Must not overlap the private subnets
|
||||
# above — hosts in the ALB subnets are trusted_proxies (gateway.yaml) and could
|
||||
# spoof client IPs via X-Forwarded-For.
|
||||
corporate_cidr = "10.0.0.0/8"
|
||||
|
||||
# ACM certificate for your internal gateway hostname (the host in gateway.yaml's
|
||||
# public_url), imported or issued by AWS Private CA.
|
||||
acm_certificate_arn = "arn:aws:acm:..."
|
||||
|
||||
image_tag = "<version>" # REQUIRED — the tag you build and push as linux/amd64 with
|
||||
# gateway.yaml baked in (setup.sh tags <version>-cfg<sha8>;
|
||||
# see README Deploy)
|
||||
|
||||
# Okta OIDC client secret: REQUIRED — uncomment and set it (Terraform creates the
|
||||
# secret version; the ECS tasks inject `gateway-oidc-client-secret` at start, so
|
||||
# without a version they fail with ResourceInitializationError). Leave empty only
|
||||
# if you add the secret version out-of-band.
|
||||
# oidc_client_secret = "..."
|
||||
189
examples/gateway/aws/terraform/variables.tf
Normal file
189
examples/gateway/aws/terraform/variables.tf
Normal file
@@ -0,0 +1,189 @@
|
||||
# Inputs — mirror the env-overridable knobs in setup.sh (same defaults).
|
||||
|
||||
variable "region" {
|
||||
description = "AWS region for everything this module creates. Pick one where Bedrock serves the Claude models you need. (The Bedrock region the gateway calls is set separately inside gateway.yaml — keep the two equal.) The walkthrough is scoped to the commercial US regions (us-east-1/us-east-2/us-west-1/us-west-2 — GovCloud and ISO regions are different partitions); see allow_non_us_region."
|
||||
type = string
|
||||
default = "us-east-1"
|
||||
}
|
||||
|
||||
variable "allow_non_us_region" {
|
||||
description = "The bedrock-invoke IAM policy and the gateway's built-in model catalog use the US-geo (us.anthropic.*) cross-region inference profiles, so any region outside the commercial US four (including GovCloud/ISO, which are different partitions) fails a plan-time precondition. Set true ONLY after pinning region-appropriate inference profiles via a models: block in gateway.yaml (see the config reference) and widening the ARN geo prefix in main.tf's bedrock-invoke policy."
|
||||
type = bool
|
||||
default = false
|
||||
}
|
||||
|
||||
# ── Networking inputs (prerequisites — NOT created here) ────────────────────
|
||||
variable "vpc_id" {
|
||||
description = "Existing VPC ID (the walkthrough's prerequisite VPC). Unlike the GCP example, this module does not create the network."
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "private_subnet_ids" {
|
||||
description = "Two+ private subnet IDs in different AZs, with outbound internet via a NAT gateway. The internal ALB, the ECS tasks, and the RDS subnet group all attach here."
|
||||
type = list(string)
|
||||
validation {
|
||||
condition = length(var.private_subnet_ids) >= 2
|
||||
error_message = "private_subnet_ids needs at least two subnets in different AZs (the internal ALB requires two)."
|
||||
}
|
||||
}
|
||||
|
||||
variable "corporate_cidr" {
|
||||
description = "Your corporate network CIDR — the only source the ALB security group admits on 443. Must not overlap private_subnet_ids: hosts there are trusted_proxies (gateway.yaml) and could spoof client IPs via X-Forwarded-For."
|
||||
type = string
|
||||
}
|
||||
|
||||
# ── IAM (§2) ────────────────────────────────────────────────────────────────
|
||||
variable "task_role_name" {
|
||||
description = "ECS task role name (the gateway's runtime identity; its only permission is Bedrock invoke)."
|
||||
type = string
|
||||
default = "claude-gateway-task"
|
||||
}
|
||||
|
||||
variable "execution_role_name" {
|
||||
description = "ECS execution role name (the ECS agent's identity: pulls the image, injects the secrets)."
|
||||
type = string
|
||||
default = "claude-gateway-execution"
|
||||
}
|
||||
|
||||
# ── Image (§6) ──────────────────────────────────────────────────────────────
|
||||
# Terraform creates the ECR repository but does NOT build/push the image (that's
|
||||
# a docker build step — see README). It references the image by tag.
|
||||
variable "ecr_repo" {
|
||||
description = "ECR repository name."
|
||||
type = string
|
||||
default = "claude-gateway"
|
||||
}
|
||||
|
||||
variable "image_tag" {
|
||||
description = "Image tag — the tag you built and pushed (must already exist in the repo as linux/amd64, with gateway.yaml baked in). setup.sh tags as <version>-cfg<sha8 of gateway.yaml>; see the README Deploy section for the build command."
|
||||
type = string
|
||||
validation {
|
||||
condition = can(regex("^[A-Za-z0-9_][A-Za-z0-9._-]{0,127}$", var.image_tag))
|
||||
error_message = "image_tag must be a valid OCI tag — set it to the tag you pushed (the '<version>' in terraform.tfvars.example is a placeholder)."
|
||||
}
|
||||
}
|
||||
|
||||
variable "gateway_config_path" {
|
||||
description = "Path to gateway.yaml. Empty = ../gateway.yaml relative to this module. Read for the REPLACE_ME guard and the config-sha that rolls the service; the file itself ships inside the image."
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
|
||||
# ── RDS (§3) ────────────────────────────────────────────────────────────────
|
||||
variable "db_instance" {
|
||||
description = "RDS instance identifier."
|
||||
type = string
|
||||
default = "claude-gateway-db"
|
||||
}
|
||||
|
||||
variable "db_engine_version" {
|
||||
description = "Postgres major version. The gateway supports PostgreSQL 14 or newer; 16 is the recommended default."
|
||||
type = string
|
||||
default = "16"
|
||||
}
|
||||
|
||||
variable "db_instance_class" {
|
||||
description = "RDS instance class."
|
||||
type = string
|
||||
default = "db.t4g.micro"
|
||||
}
|
||||
|
||||
variable "db_allocated_storage" {
|
||||
description = "RDS allocated storage in GiB."
|
||||
type = number
|
||||
default = 20
|
||||
}
|
||||
|
||||
variable "db_name" {
|
||||
description = "Database name."
|
||||
type = string
|
||||
default = "claude_gateway"
|
||||
}
|
||||
|
||||
variable "db_user" {
|
||||
description = "Database master user (the gateway connects as this role)."
|
||||
type = string
|
||||
default = "gateway"
|
||||
}
|
||||
|
||||
# ── Secrets (§5) ────────────────────────────────────────────────────────────
|
||||
# The execution role's secrets-read policy grants read on exactly these three
|
||||
# secrets' ARNs, so renames are picked up automatically on the next apply.
|
||||
variable "secret_name" {
|
||||
description = "Secrets Manager secret holding the Postgres connection string."
|
||||
type = string
|
||||
default = "gateway-postgres-url"
|
||||
}
|
||||
|
||||
variable "jwt_secret_name" {
|
||||
description = "Secrets Manager secret holding the session JWT signing key."
|
||||
type = string
|
||||
default = "gateway-jwt-secret"
|
||||
}
|
||||
|
||||
variable "oidc_secret_name" {
|
||||
description = "Secrets Manager secret holding the Okta OIDC client secret."
|
||||
type = string
|
||||
default = "gateway-oidc-client-secret"
|
||||
}
|
||||
|
||||
variable "oidc_client_secret" {
|
||||
description = "Okta OIDC client secret value. Leave empty to NOT manage the version via Terraform (only if you add the secret version out-of-band — without one the tasks fail to start)."
|
||||
type = string
|
||||
default = ""
|
||||
sensitive = true
|
||||
}
|
||||
|
||||
# ── ECS + ALB (§7) ──────────────────────────────────────────────────────────
|
||||
variable "cluster_name" {
|
||||
description = "ECS cluster name."
|
||||
type = string
|
||||
default = "claude-gateway"
|
||||
}
|
||||
|
||||
variable "service_name" {
|
||||
description = "ECS service name (also used for the ALB and target group)."
|
||||
type = string
|
||||
default = "claude-gateway"
|
||||
}
|
||||
|
||||
variable "log_group_name" {
|
||||
description = "CloudWatch Logs group for the gateway's stderr (audit events + operational logs)."
|
||||
type = string
|
||||
default = "/ecs/claude-gateway"
|
||||
}
|
||||
|
||||
variable "log_retention_days" {
|
||||
description = "CloudWatch Logs retention in days. The group carries the gateway's audit events, so align with your audit retention policy."
|
||||
type = number
|
||||
default = 90
|
||||
}
|
||||
|
||||
variable "acm_certificate_arn" {
|
||||
description = "ACM certificate ARN for the internal gateway hostname (the host in gateway.yaml's public_url), served by the ALB's HTTPS listener."
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "task_cpu" {
|
||||
description = "Fargate task CPU units."
|
||||
type = number
|
||||
default = 1024
|
||||
}
|
||||
|
||||
variable "task_memory" {
|
||||
description = "Fargate task memory (MiB)."
|
||||
type = number
|
||||
default = 2048
|
||||
}
|
||||
|
||||
variable "desired_count" {
|
||||
description = "ECS service desired task count. Each task opens a Postgres pool of up to 5 connections (the gateway's store.max_connections default) and db.t4g.micro caps at ~80 max_connections — keep desired_count × 5 below the DB class's limit, or raise the class before raising this."
|
||||
type = number
|
||||
default = 1
|
||||
}
|
||||
|
||||
variable "deletion_protection" {
|
||||
description = "Deletion protection on RDS and the ALB (and whether RDS skips the final snapshot on destroy). Keep true to avoid accidental deletion of the running deployment."
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
18
examples/gateway/aws/terraform/versions.tf
Normal file
18
examples/gateway/aws/terraform/versions.tf
Normal file
@@ -0,0 +1,18 @@
|
||||
# Provider + version pins for the Claude apps gateway ECS Fargate deployment.
|
||||
terraform {
|
||||
required_version = ">= 1.5"
|
||||
required_providers {
|
||||
aws = {
|
||||
source = "hashicorp/aws"
|
||||
version = ">= 6.0, < 7.0" # 6.0 renames data.aws_region's attribute to `region`
|
||||
}
|
||||
random = {
|
||||
source = "hashicorp/random"
|
||||
version = ">= 3.5"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
provider "aws" {
|
||||
region = var.region
|
||||
}
|
||||
Reference in New Issue
Block a user