From da2640bee3d49f341680c8af161c7afa63f76c5a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 21 Sep 2026 12:27:59 +0000 Subject: [PATCH 01/24] feat(aws-nixos): add a NixOS on AWS EC2 template Launches an official NixOS AMI and configures it from a flake the user owns, rather than shipping a configuration inside the template. The instance is handed three things -- the agent token, the agent init script and a flake reference -- and applies the flake itself. The flake is treated as foreign code: it is cloned to /etc/nixos and built with a plain `nixos-rebuild switch --flake /etc/nixos#`, with no --override-input, no --impure and no injected inputs, so the command the template runs is reproducible by hand. Notable behaviour: - The agent is started by the boot script after the rebuild finishes, not by systemd, so startup scripts never run against a generation that is about to be replaced. - nixos-rebuild output is streamed verbatim to a "NixOS" log source, filtered of store-path lists and per-derivation output and budgeted well under Coder's 1 MiB per-agent log cap, which latches permanently on overflow. - The checkout at /etc/nixos is the source of truth: it is fast-forwarded when clean, and a dirty tree or local commits are left alone and built as they are. - Per-workspace facts are published to /run/coder/workspace.json as runtime data; the agent token stays in tmpfs at 0600 and never reaches Nix. - A periodic rebuild runs on a cron schedule, either `boot` (staged for the next restart) or `switch` (applied immediately). The Nix-specific parts of the boot and rebuild paths live in modules/nix/ so they can later become a standalone module that manages a flake lifecycle on any Linux host. The companion configuration is github.com/coder/nixos-example-flake. --- .icons/nixos-rainbow.svg | 1 + .icons/nixos-trans.svg | 1 + .icons/nixos.svg | 1 + .../templates/aws-nixos/PREREQUISITES.md | 100 +++++ registry/coder/templates/aws-nixos/README.md | 321 +++++++++++++ registry/coder/templates/aws-nixos/main.tf | 421 ++++++++++++++++++ .../templates/aws-nixos/modules/nix/README.md | 62 +++ .../aws-nixos/modules/nix/lifecycle.sh | 239 ++++++++++ .../aws-nixos/scripts/bootstrap.sh.tftpl | 191 ++++++++ .../coder/templates/aws-nixos/scripts/log.sh | 145 ++++++ .../aws-nixos/scripts/rebuild.sh.tftpl | 76 ++++ 11 files changed, 1558 insertions(+) create mode 100644 .icons/nixos-rainbow.svg create mode 100644 .icons/nixos-trans.svg create mode 100644 .icons/nixos.svg create mode 100644 registry/coder/templates/aws-nixos/PREREQUISITES.md create mode 100644 registry/coder/templates/aws-nixos/README.md create mode 100644 registry/coder/templates/aws-nixos/main.tf create mode 100644 registry/coder/templates/aws-nixos/modules/nix/README.md create mode 100644 registry/coder/templates/aws-nixos/modules/nix/lifecycle.sh create mode 100644 registry/coder/templates/aws-nixos/scripts/bootstrap.sh.tftpl create mode 100644 registry/coder/templates/aws-nixos/scripts/log.sh create mode 100644 registry/coder/templates/aws-nixos/scripts/rebuild.sh.tftpl diff --git a/.icons/nixos-rainbow.svg b/.icons/nixos-rainbow.svg new file mode 100644 index 000000000..440f3f7a0 --- /dev/null +++ b/.icons/nixos-rainbow.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/.icons/nixos-trans.svg b/.icons/nixos-trans.svg new file mode 100644 index 000000000..0d72d1ae0 --- /dev/null +++ b/.icons/nixos-trans.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/.icons/nixos.svg b/.icons/nixos.svg new file mode 100644 index 000000000..a7b94a69d --- /dev/null +++ b/.icons/nixos.svg @@ -0,0 +1 @@ + \ No newline at end of file diff --git a/registry/coder/templates/aws-nixos/PREREQUISITES.md b/registry/coder/templates/aws-nixos/PREREQUISITES.md new file mode 100644 index 000000000..4b2d969be --- /dev/null +++ b/registry/coder/templates/aws-nixos/PREREQUISITES.md @@ -0,0 +1,100 @@ +# Prerequisites + +## Authentication + +This template authenticates to AWS using the provider's default [authentication methods](https://registry.terraform.io/providers/hashicorp/aws/latest/docs#authentication-and-configuration). + +The simplest way, without editing the template, is environment variables (`AWS_ACCESS_KEY_ID`, +`AWS_SECRET_ACCESS_KEY`, `AWS_REGION`) or a [credentials file](https://docs.aws.amazon.com/cli/latest/userguide/cli-configure-files.html#cli-configure-files-format). +If you are running Coder on a VM, that file must be at `/home/coder/aws/credentials`. + +Credentials belong in the environment of the **provisioner process** — `coder server`, or your +external provisioner — and not in Terraform variables. Template variables surface in workspace +parameters and build logs, so a credential passed that way is readable by anyone who can view a +build. Restart the provisioner after changing them. + +Prefer, in order: + +1. An **instance profile** (Coder on EC2) or **IRSA** (Coder on EKS). No long-lived secret exists. +2. A long-lived, low-privilege identity plus `assume_role` in the provider block. +3. Static access keys. + +Avoid `AWS_SESSION_TOKEN` from STS for a provisioner: it expires, and it will expire in the middle +of a build. + +## A default VPC in the selected region + +Like the `aws-linux` template, this one launches into the default VPC of the region chosen by the +`aws_region` parameter and does not take a subnet or security group. Regions without a default VPC +fail at apply with `VPCIdNotSpecified`. Either pick a region that has one, or create one with +`aws ec2 create-default-vpc --region `. + +## Required permissions / policy + +The following sample policy allows Coder to create EC2 instances and modify instances it +provisioned. + +```json +{ + "Version": "2012-10-17", + "Statement": [ + { + "Sid": "VisualEditor0", + "Effect": "Allow", + "Action": [ + "ec2:GetDefaultCreditSpecification", + "ec2:DescribeIamInstanceProfileAssociations", + "ec2:DescribeTags", + "ec2:DescribeInstances", + "ec2:DescribeInstanceTypes", + "ec2:DescribeInstanceStatus", + "ec2:CreateTags", + "ec2:RunInstances", + "ec2:DescribeInstanceCreditSpecifications", + "ec2:DescribeImages", + "ec2:ModifyDefaultCreditSpecification", + "ec2:DescribeVolumes" + ], + "Resource": "*" + }, + { + "Sid": "CoderResources", + "Effect": "Allow", + "Action": [ + "ec2:DescribeInstanceAttribute", + "ec2:UnmonitorInstances", + "ec2:TerminateInstances", + "ec2:StartInstances", + "ec2:StopInstances", + "ec2:DeleteTags", + "ec2:MonitorInstances", + "ec2:CreateTags", + "ec2:RunInstances", + "ec2:ModifyInstanceAttribute", + "ec2:ModifyInstanceCreditSpecification" + ], + "Resource": "arn:aws:ec2:*:*:instance/*", + "Condition": { + "StringEquals": { + "aws:ResourceTag/Coder_Provisioned": "true" + } + } + } + ] +} +``` + +## Network egress from the workspace + +Unlike the stock AWS templates, workspaces here must reach more than the Coder deployment. A NixOS +instance resolves and builds its own configuration on boot, so it needs outbound HTTPS to: + +| Host | Why | +| ------------------------------------ | -------------------------------------------------------- | +| your Coder access URL | agent connection and the log API | +| `cache.nixos.org` | binary cache; without it everything is built from source | +| `github.com` / `codeload.github.com` | fetching your flake and its nixpkgs input | +| any extra substituters you configure | binary caches declared in your flake | + +No inbound rules are required. If you plan to use the SSH rescue path described in the README, open +port 22 from your own address — the default VPC security group does not allow it. diff --git a/registry/coder/templates/aws-nixos/README.md b/registry/coder/templates/aws-nixos/README.md new file mode 100644 index 000000000..5b68415bf --- /dev/null +++ b/registry/coder/templates/aws-nixos/README.md @@ -0,0 +1,321 @@ +--- +display_name: AWS EC2 (NixOS) +description: Provision NixOS EC2 VMs as Coder workspaces from a flake +icon: ../../../../.icons/nixos.svg +verified: false +tags: [vm, linux, aws, nixos, persistent-vm] +--- + +# Remote development on NixOS AWS EC2 VMs + +Provision NixOS EC2 instances as [Coder workspaces](https://coder.com/docs/workspaces), configured +declaratively from a flake in a Git repository. The Coder agent is declared as a NixOS systemd unit, +so it survives `nixos-rebuild` and the workspace stays reachable across configuration changes. The +reference configuration lives at [coder/nixos-example-flake](https://github.com/coder/nixos-example-flake); +point the template at your own fork to control the environment. + + + +## Prerequisites + +### Authentication + +This template authenticates to AWS using the provider's default [authentication methods](https://registry.terraform.io/providers/hashicorp/aws/latest/docs#authentication-and-configuration). + +The simplest way is to set `AWS_ACCESS_KEY_ID` and `AWS_SECRET_ACCESS_KEY` in the environment of the +Coder provisioner. See [PREREQUISITES.md](./PREREQUISITES.md) for the IAM policy, the default VPC +requirement, and the reasons credentials must not be passed as template variables. + +## How it works + +The template does not build anything itself. It launches an official NixOS AMI and hands the +instance three things: the agent token, the agent init script, and a flake reference. The instance +then applies that flake. + +```text +Terraform ──user-data──▶ amazon-init ──▶ nixos-rebuild switch ──▶ coder-agent.service + (every boot) (from your flake) (started once it finishes) +``` + +The NixOS AMI does not run cloud-init. It runs `amazon-init.service`, which reads +`/etc/ec2-metadata/user-data` and execs it as a shell script when it begins with `#!` — after +`multi-user.target`, on every boot. The template's script writes the agent handoff, generates the +per-workspace Nix values, and runs one `nixos-rebuild switch`. + +User-data itself is a short self-extracting wrapper: the boot script, its two libraries and the +agent init script come to roughly 19 KiB against EC2's 16 KiB limit, so it ships compressed and +unpacks to `/run/coder/bootstrap.sh` — which is also where to look when debugging a boot. + +**The agent only starts once the rebuild has finished**, so a build that has work to do shows no +agent while it runs — minutes on a first create, usually seconds afterwards. Progress is streamed +to a workspace log source named "NixOS" while that happens, and `connection_timeout` is raised to +20 minutes so a normal first build does not look like a failure. + +That delay is deliberate. `coder-agent.service` is declared without `wantedBy`, so systemd never +starts it on its own and the boot script starts it explicitly after the switch. Left to systemd the +agent would come up at `multi-user.target`, which on every boot after the first is _before_ +`amazon-init` has fetched the new commit and rebuilt: the workspace would be reported ready and its +startup scripts would install into a generation that is about to be replaced. A workspace that is +late is better than a workspace that is wrong. + +If the rebuild fails, the boot script starts the agent anyway — a workspace whose flake does not +build is exactly the one you need a terminal on. + +## Choosing which configuration is applied + +Two variables control this: + +| Variable | Default | Meaning | +| ------------ | ----------------------------------------------------------- | ---------------------------------------------- | +| `flake_ref` | `git+https://github.com/coder/nixos-example-flake?ref=main` | where the flake lives | +| `flake_attr` | `workspace-$ARCH` | which `nixosConfigurations` attribute to apply | + +`flake_attr` is the part after `#` in a flake reference, so this is exactly the selection you would +make by hand: + +```console +nixos-rebuild switch --flake 'github:your-org/config#workspace-x86_64' +``` + +`$ARCH` in `flake_attr` is replaced with `x86_64` or `aarch64` to match the chosen instance type. +That keeps the AMI architecture, `coder_agent.arch` and the flake attribute in agreement — they are +all derived from one map in `main.tf`, so a Graviton instance type cannot accidentally boot an +x86 configuration. If you keep a single configuration instead, set `flake_attr` to a fixed name and +only offer instance types of the matching architecture. + +Any reference `nixos-rebuild --flake` understands works, including `github:owner/repo`, +`git+ssh://` for private repositories, and `?dir=subdir` for a flake in a subdirectory. The +configuration must be committed: a Git flake reference only ever sees committed files. + +## Values passed into the flake + +None. The flake is evaluated exactly as written, with no `--override-input`, no `--impure` and no +injected inputs — which is what makes the command the template runs reproducible by hand. + +Per-workspace facts are published as a runtime file instead: + +```json +// /run/coder/workspace.json, mode 0644 +{ + "workspace": "my-workspace", + "owner": "jane", + "owner_name": "Jane Doe", + "owner_email": "jane@example.com", + "access_url": "https://coder.example.com", + "hostname": "my-workspace" +} +``` + +It cannot be an evaluation input: a pure flake may not read an absolute path outside itself, so +consuming it at eval time would need `--impure` and would stop `nixos-rebuild switch` from +reproducing what the template applied. A configuration that wants these values reads the file from +a service at runtime. + +Git identity is not in the flake's hands either — it comes from the +[git-config](https://registry.coder.com/modules/coder/git-config) module, which configures the +workspace user's `~/.gitconfig` after the rebuild. + +> [!IMPORTANT] +> Never pass a secret into Nix — not as an input, `--argstr` or `builtins.getEnv`. It is copied +> into `/nix/store`, which is world-readable to every process on the workspace and persists across +> generations and past rotation. The agent token is deliberately handed over through `/run/coder` +> at runtime instead, at mode 0600 on a tmpfs, so that it never reaches Nix. + +## Rebuilding by hand + +The configuration is a git checkout at `/etc/nixos`, owned by the workspace +user, and that is what the template builds. So the command is the ordinary one: + +```console +sudo nixos-rebuild switch --flake /etc/nixos#workspace-x86_64 +``` + +No overrides, no `--impure`, no injected inputs — what you get by hand is +exactly what the template applies. The attribute is shown in the workspace's +`Flake URI` metadata and in the boot log. + +> [!NOTE] +> A bare `sudo nixos-rebuild switch` only works if your flake exposes a +> configuration named after the machine's hostname, which is what +> `nixos-rebuild` defaults to. The example flake uses per-architecture names +> instead, so pass `--flake /etc/nixos#`. If your workspaces have stable +> names, naming the configuration after the hostname makes the bare form work. + +On every boot the template syncs the checkout: it clones if missing, +fast-forwards a clean checkout on its tracking branch, and **leaves a dirty +tree or local commits alone** and builds those instead. So edits survive a +restart, and the machine tracks upstream until you change something. + +A flake built from a git checkout ignores untracked files — `git add` a new +`.nix` file or the rebuild will not see it. + +## Keeping workspaces up to date + +The `update_process` parameter decides how a periodic rebuild is applied: + +- **`boot` (default)** — builds the new configuration and makes it the boot default without + activating it. Nothing restarts while you are working; the change lands on your next workspace + restart. The "NixOS" metric in the workspace header shows `(restart to apply update)` when a + generation is staged. +- **`switch`** — activates immediately, restarting any service whose definition changed. + +The schedule comes from the `update_schedule` variable, default `0 0 4 * * *` (04:00 daily). + +> [!NOTE] +> `update_schedule` is a **six** field cron expression with seconds first, evaluated in the +> workspace's own timezone. A five field expression is silently misinterpreted rather than +> rejected, and descriptors like `@daily` pass validation but then fail on the agent. Set +> `time.timeZone` in your configuration so the schedule means what you intend. + +Set `update_schedule = ""` to disable periodic rebuilds entirely. + +To rebuild immediately: + +```console +sudo nixos-rebuild switch --flake 'git+https://github.com/coder/nixos-example-flake?ref=main#workspace-x86_64' \ + --override-input coder-vars path:/etc/coder/vars --no-write-lock-file --refresh +``` + +## Where the logs are + +The workspace UI streams `nixos-rebuild`'s own output under a log source named **NixOS** — the same +lines you would see in a terminal. Three kinds of noise are dropped: the enumerated store paths +under `these N derivations will be built:`, per-derivation compiler output, and the expected +`not writing modified lock file` notice. The complete transcript is on the instance: + +```console +/var/log/coder-nixos/rebuild-latest.log # symlink to the most recent run +/var/log/coder-nixos/coder-script.log # the periodic rebuild script +``` + +Keeping compiler output out of the UI is not cosmetic. Coder caps agent logs at **1 MiB per +agent**, shared across every log source, and exceeding it does not truncate — the log is marked +overflowed and all later logs for that agent are dropped permanently. The template budgets itself +to half the cap and goes quiet with a pointer to the transcript if it ever gets there. + +## Persistence + +The instance is stopped and started rather than destroyed and recreated, so the root volume — and +with it `/home` and the Nix store — persists across workspace restarts. + +Two lifecycle settings make that safe, and both matter: + +- `ignore_changes = [ami]` — the official NixOS AMIs are republished weekly and garbage-collected + after 90 days. Without this, a new AMI id would replace every live workspace and destroy its root + volume. The side effect is that **changing `nixos_release` only affects newly created + workspaces**; existing ones keep their AMI and get their packages from your flake's nixpkgs pin + anyway. +- `user_data_replace_on_change = false` — the agent token is inside user-data and rotates on every + start, so user-data changes on every start. With replacement enabled, every restart would destroy + the volume. + +`root_volume_size` is mutable: the AMI enables `boot.growPartition` and `autoResize`, so a larger +volume is picked up on the next restart. + +## Architecture support + +Both `x86_64` and `arm64` (Graviton) instance types are offered. The AMI filter, +`coder_agent.arch` and the flake attribute are all derived from the instance type, so they cannot +disagree — but your flake must expose a configuration for the architecture you select. The +reference flake ships `workspace-x86_64` and `workspace-aarch64`. + +The smallest instance type offered is `t3.medium` on purpose: the NixOS AMI configures no swap and +the Nix store shares the root volume, so a rebuild that has to compile anything will exhaust a +1–2 GiB instance. + +## Troubleshooting + +### The workspace has been building for a long time + +Expected whenever there is a rebuild to do, and always on first create: the agent is not started +until `nixos-rebuild switch` finishes. Watch the "NixOS" log source. A cold closure on a small +instance can take ten minutes or more. A restart with nothing to rebuild skips straight to starting +the agent. + +### The agent never connects + +The boot script writes its handoff to `/run/coder` before doing anything else, so the usual cause is +a failed rebuild — or, if there are no logs at all, an instance with no route to the internet (a +NixOS workspace fetches its own configuration on boot, so it needs egress before it can report +anything). The workspace metadata shows the instance id; the AMI logs to the serial console, which +needs no SSH: + +```console +aws ec2 get-console-output --instance-id i-0123456789abcdef0 --output text +``` + +On the instance: + +```console +systemctl status amazon-init coder-agent +journalctl -u amazon-init -b +cat /var/log/coder-nixos/rebuild-latest.log +``` + +### A configuration change broke the workspace + +`nixos-rebuild switch` builds before it activates, so a configuration that fails to _build_ never +touches the running system — the previous generation keeps running and the failure appears in the +logs. If a configuration builds but misbehaves, roll back: + +```console +sudo nixos-rebuild switch --rollback +``` + +There is deliberately no automatic rollback: on a fresh instance the previous generation is the bare +AMI, which has no Coder agent at all, so rolling back automatically would trade a visible failure +for an unreachable workspace. + +### Recovering an unreachable instance + +`aws ec2 get-console-output` above needs no access to the instance at all and is usually enough. + +For a shell, the AMI enables OpenSSH and `amazon-ssm-agent`. Neither is reachable out of the box — +the template attaches no key pair and the default security group allows no inbound traffic — so +attach what you need for the session with the AWS CLI and detach it afterwards: + +```console +aws ec2 create-security-group --group-name coder-debug --description "temporary SSH" --vpc-id +aws ec2 authorize-security-group-ingress --group-id --protocol tcp --port 22 --cidr /32 +aws ec2 modify-instance-attribute --instance-id --groups +``` + +### Private flake repositories + +Nix fetches flakes as root via the daemon, so credentials must be readable by root rather than by +the workspace user. Prefer `nix.settings.netrc-file` pointing at a file the boot script writes at +mode 0600, or an `!include` of a private file from `/etc/nix/nix.conf`. Do **not** put a token in +`nix.settings.access-tokens` directly: that renders it into `/etc/nix/nix.conf` by way of the Nix +store, where every process on the workspace can read it. + +## Extending the template + +Three registry modules are included: +[code-server](https://registry.coder.com/modules/coder/code-server), +[jetbrains-gateway](https://registry.coder.com/modules/coder/jetbrains-gateway) and +[git-config](https://registry.coder.com/modules/coder/git-config). + +The first two push a dynamically linked binary into the workspace and exec it, so they work only +because the reference flake sets `programs.nix-ld.enable = true` — remove that and both fail with a +misleading "No such file or directory". Gateway is also told which architecture to fetch, from the +same instance-type map that picks the AMI, and is restricted to the IDEs JetBrains publishes an +`aarch64` backend for. + +> [!NOTE] +> A registry module whose script starts with `#!/bin/bash` cannot run here. NixOS puts nothing in +> `/bin` but `sh`, so the kernel fails the exec before anything runs and the agent reports exit +> 255 with an empty log. `#!/usr/bin/env bash` works. Check the module before adding it. + +The `coder` CLI itself is put on `PATH` by the flake's Coder module, which the `coder stat` +metadata scripts depend on. The agent prepends its own directory to the `PATH` it hands scripts, +but on NixOS those run through a login shell and `/etc/profile` rebuilds `PATH` from the system +environment, dropping it — so without that wrapper every CPU/memory/disk metric reads +`coder: command not found`. + +For anything heavier, prefer declaring the tool in your flake and exposing it with a +[`coder_app`](https://registry.terraform.io/providers/coder/coder/latest/docs/resources/app): +it is reproducible and avoids the dynamic-linking problem entirely. + +The Nix-specific parts of the boot and rebuild paths live in +[`modules/nix/`](./modules/nix/README.md), kept separate so they can become a standalone Coder +module that manages a flake lifecycle on any Linux host, not just NixOS. diff --git a/registry/coder/templates/aws-nixos/main.tf b/registry/coder/templates/aws-nixos/main.tf new file mode 100644 index 000000000..fb81bf699 --- /dev/null +++ b/registry/coder/templates/aws-nixos/main.tf @@ -0,0 +1,421 @@ +terraform { + required_providers { + coder = { + source = "coder/coder" + version = "~> 2.0" + } + aws = { + source = "hashicorp/aws" + } + } +} + +module "aws_region" { + source = "registry.coder.com/coder/aws-region/coder" + version = "~> 1.0" + default = "eu-west-3" +} + +provider "aws" { + region = module.aws_region.value +} + +variable "flake_ref" { + description = <<-EOT + Git remote holding the NixOS configuration. Cloned to /etc/nixos on the + workspace, which is what makes a bare `sudo nixos-rebuild switch` work. + + Anything `git clone` accepts, so private repositories need git + credentials on the instance rather than Nix's netrc. + EOT + type = string + default = "https://github.com/coder/nixos-example-flake" +} + +variable "flake_branch" { + description = "Branch to track in flake_ref." + type = string + default = "main" +} + +variable "flake_attr" { + description = <<-EOT + Which `nixosConfigurations` attribute to apply, i.e. the part after `#` + in the flake reference. `$ARCH` is replaced with `x86_64` or `aarch64` to + match the chosen instance type. + EOT + type = string + default = "workspace-$ARCH" +} + +variable "nixos_release" { + description = <<-EOT + NixOS release series used to select the AMI, matched as + `nixos/*`. Only affects newly created workspaces; packages and + the kernel come from the flake's own nixpkgs pin. + EOT + type = string + default = "26.05" +} + +variable "update_schedule" { + description = <<-EOT + Cron schedule for the periodic `nixos-rebuild`, or empty to disable. + + SIX fields with seconds first, in the workspace's timezone. A five field + expression is silently misinterpreted rather than rejected, and + descriptors like `@daily` pass validation then fail on the agent. + EOT + type = string + default = "0 0 4 * * *" + + validation { + condition = var.update_schedule == "" || length(split(" ", trimspace(var.update_schedule))) == 6 + error_message = "update_schedule must be a 6-field cron expression (seconds first), or empty." + } +} + +data "coder_parameter" "instance_type" { + name = "instance_type" + display_name = "Instance type" + description = <<-EOT + The smallest option is t3.medium on purpose: the NixOS AMI configures no + swap and the Nix store shares the root volume, so a rebuild that has to + compile anything will exhaust a 1-2 GiB instance. + EOT + default = "t3.medium" + mutable = false + + option { + name = "2 vCPU, 4 GiB RAM" + value = "t3.medium" + } + option { + name = "2 vCPU, 8 GiB RAM" + value = "t3.large" + } + option { + name = "4 vCPU, 16 GiB RAM" + value = "t3.xlarge" + } + option { + name = "8 vCPU, 32 GiB RAM" + value = "t3.2xlarge" + } + option { + name = "2 vCPU, 4 GiB RAM (Graviton)" + value = "t4g.medium" + } + option { + name = "2 vCPU, 8 GiB RAM (Graviton)" + value = "m7g.large" + } + option { + name = "4 vCPU, 16 GiB RAM (Graviton)" + value = "m7g.xlarge" + } +} + +data "coder_parameter" "root_volume_size" { + name = "root_volume_size" + display_name = "Root volume size (GiB)" + description = "Holds the Nix store as well as /home. Can be increased later." + type = "number" + default = 80 + mutable = true + + validation { + min = 40 + max = 2000 + } +} + +data "coder_parameter" "update_process" { + name = "update_process" + display_name = "Configuration updates" + description = <<-EOT + How a periodic `nixos-rebuild` is applied while the workspace is running. + `boot` stages the new configuration without activating it, so nothing + restarts under you; `switch` activates it immediately. + EOT + type = "string" + default = "boot" + mutable = true + + option { + name = "Apply on next restart (boot)" + value = "boot" + } + option { + name = "Apply immediately (switch)" + value = "switch" + } +} + +data "coder_workspace" "me" {} +data "coder_workspace_owner" "me" {} + +data "aws_ami" "nixos" { + most_recent = true + filter { + name = "name" + values = ["nixos/${var.nixos_release}*"] + } + filter { + name = "architecture" + values = [local.arch.ami] + } + # Required: aws_ami matches on a name pattern, and AMI names are not unique + # across accounts. Without an owner filter, most_recent would happily pick + # a stranger's image named nixos/... and boot it with the agent token. + owners = ["427812963091"] # NixOS +} + +resource "coder_agent" "main" { + count = data.coder_workspace.me.start_count + arch = local.arch.agent + os = "linux" + # Token rather than instance identity: the boot script needs a bearer token + # for the agent log API anyway. + auth = "token" + # The first boot completes a nixos-rebuild switch before the agent exists, + # so the default 120s looks like a failed workspace. + connection_timeout = 1200 + + metadata { + key = "cpu" + display_name = "CPU Usage" + interval = 5 + timeout = 5 + script = "coder stat cpu" + } + metadata { + key = "memory" + display_name = "Memory Usage" + interval = 5 + timeout = 5 + script = "coder stat mem" + } + metadata { + key = "disk" + display_name = "Disk Usage" + interval = 600 + timeout = 30 + script = "coder stat disk --path $HOME" + } + # Makes `update_process = boot` visible; a staged generation is otherwise + # invisible and looks like updates being ignored. + metadata { + key = "nixos" + display_name = "NixOS version" + interval = 60 + timeout = 10 + # /run/current-system is the activated system; /run/booted-system is what + # the kernel booted and still points at the previous generation after a + # switch, which would mark every new workspace as needing a restart. + script = <<-EOT + version=$(nixos-version 2>/dev/null || echo unknown) + if [ "$(readlink -f /run/current-system)" = "$(readlink -f /nix/var/nix/profiles/system)" ]; then + echo "$version" + else + echo "$version (restart to apply update)" + fi + EOT + } +} + +# See https://registry.coder.com/modules/coder/code-server +module "code-server" { + count = data.coder_workspace.me.start_count + source = "registry.coder.com/coder/code-server/coder" + version = "~> 1.0" + agent_id = coder_agent.main[0].id + order = 1 +} + +# See https://registry.coder.com/modules/coder/jetbrains-gateway +# +# The IDE backend is a dynamically linked download that Gateway unpacks into +# the workspace and execs, which on NixOS needs `programs.nix-ld`. The +# reference flake enables it. +module "jetbrains_gateway" { + count = data.coder_workspace.me.start_count + source = "registry.coder.com/coder/jetbrains-gateway/coder" + version = "~> 1.2" + agent_id = coder_agent.main[0].id + agent_name = "main" + arch = local.arch.agent + # The flake names the workspace user; `coder` is the reference flake's + # default and what the rest of this template assumes. + folder = "/home/coder" + # Restricted to the IDEs JetBrains publishes an aarch64 backend for, since + # half the instance types here are Graviton. + jetbrains_ides = ["IU", "PY", "GO", "WS"] + default = "IU" + # Without this the module hands Gateway its pinned 2024.3 build numbers. + # The cost is that a workspace build now asks data.services.jetbrains.com + # for the current release. + latest = true + order = 2 +} + +# Git authorship, which the NixOS configuration deliberately does not set: it +# is per-workspace state a flake has no pure way to learn. +# +# See https://registry.coder.com/modules/coder/git-config +module "git-config" { + count = data.coder_workspace.me.start_count + source = "registry.coder.com/coder/git-config/coder" + version = "~> 1.0" + agent_id = coder_agent.main[0].id +} + +resource "coder_script" "nixos_rebuild" { + count = var.update_schedule == "" ? 0 : data.coder_workspace.me.start_count + agent_id = coder_agent.main[0].id + display_name = "NixOS rebuild" + cron = var.update_schedule + # The boot script has already switched by the time the agent exists. + run_on_start = false + start_blocks_login = false + timeout = 3600 + log_path = "${local.log_dir}/coder-script.log" + + script = templatefile("${path.module}/scripts/rebuild.sh.tftpl", { + LOG_SH = local.log_sh + LIFECYCLE_SH = local.lifecycle_sh + ARG_FLAKE_REF = local.flake_url + ARG_FLAKE_BRANCH = var.flake_branch + ARG_FLAKE_ATTR = local.flake_attr + ARG_UPDATE_PROCESS = data.coder_parameter.update_process.value + ARG_ACCESS_URL = data.coder_workspace.me.access_url + ARG_LOG_SOURCE_ID = local.log_source_id + }) +} + +locals { + # One map so the AMI architecture, coder_agent.arch and the flake attribute + # cannot disagree. + arch_map = { + "t3.medium" = { agent = "amd64", ami = "x86_64", attr = "x86_64" } + "t3.large" = { agent = "amd64", ami = "x86_64", attr = "x86_64" } + "t3.xlarge" = { agent = "amd64", ami = "x86_64", attr = "x86_64" } + "t3.2xlarge" = { agent = "amd64", ami = "x86_64", attr = "x86_64" } + "t4g.medium" = { agent = "arm64", ami = "arm64", attr = "aarch64" } + "m7g.large" = { agent = "arm64", ami = "arm64", attr = "aarch64" } + "m7g.xlarge" = { agent = "arm64", ami = "arm64", attr = "aarch64" } + } + arch = local.arch_map[data.coder_parameter.instance_type.value] + flake_attr = replace(var.flake_attr, "$ARCH", local.arch.attr) + log_dir = "/var/log/coder-nixos" + + # `git clone` is what runs on the instance, so accept a Nix-style flake + # reference too and reduce it to a plain remote: strip a `git+` scheme + # prefix and any query string. `flake_branch` carries the ref instead. + flake_url = replace(replace(var.flake_ref, "/^git\\+/", ""), "/\\?.*$/", "") + + # Constant, not uuid(): log sources are scoped to an agent, Coder treats a + # repeat POST with the same id as a no-op, and a generated value would churn + # the plan every run. + log_source_id = "6e1f4a2c-9b3d-4c8e-8a71-5f0d2b6c4e93" + + # Sourced verbatim into the two entrypoints below. Plain shell rather than + # templates so they stay readable and get covered by the repo's shellcheck. + log_sh = file("${path.module}/scripts/log.sh") + lifecycle_sh = file("${path.module}/modules/nix/lifecycle.sh") + + # EC2 caps user-data at 16 KiB, and the boot script plus its two libraries + # plus the agent init script come to roughly 19 KiB. So user-data is a + # six-line self-extracting wrapper around a compressed copy. + # + # This is transparent to the NixOS AMI: amazon-init only inspects the first + # two bytes for `#!` before exec'ing the blob, and it has no decompression + # step of its own. Extracting to a fixed path also means the real script is + # on disk when something needs debugging. + user_data = <<-SH + #!/usr/bin/env bash + set -eu + install -d -m 0700 /run/coder + base64 -d <<'CODER_PAYLOAD' | gzip -dc >/run/coder/bootstrap.sh + ${base64gzip(local.bootstrap)} + CODER_PAYLOAD + exec bash /run/coder/bootstrap.sh + SH + + bootstrap = templatefile("${path.module}/scripts/bootstrap.sh.tftpl", { + LOG_SH = local.log_sh + LIFECYCLE_SH = local.lifecycle_sh + ARG_FLAKE_REF = local.flake_url + ARG_FLAKE_BRANCH = var.flake_branch + ARG_FLAKE_ATTR = local.flake_attr + ARG_ACCESS_URL = data.coder_workspace.me.access_url + ARG_AGENT_TOKEN = try(coder_agent.main[0].token, "") + ARG_INIT_SCRIPT_B64 = base64encode(try(coder_agent.main[0].init_script, "")) + ARG_LOG_SOURCE_ID = local.log_source_id + ARG_HOSTNAME = lower(data.coder_workspace.me.name) + ARG_WORKSPACE_NAME = data.coder_workspace.me.name + ARG_OWNER = data.coder_workspace_owner.me.name + # base64 because a full name may contain quotes and is interpolated into + # both a shell string and a Nix string. + ARG_OWNER_NAME_B64 = base64encode(coalesce(data.coder_workspace_owner.me.full_name, data.coder_workspace_owner.me.name)) + ARG_OWNER_EMAIL = data.coder_workspace_owner.me.email + }) +} + +resource "aws_instance" "dev" { + ami = data.aws_ami.nixos.id + availability_zone = "${module.aws_region.value}a" + instance_type = data.coder_parameter.instance_type.value + user_data = local.user_data + + # The agent token is inside user-data and rotates on every workspace start, + # so user-data changes on every start. With replacement enabled, every + # restart would destroy the root volume and with it /home and the Nix store. + user_data_replace_on_change = false + + root_block_device { + volume_size = data.coder_parameter.root_volume_size.value + volume_type = "gp3" + encrypted = true + } + + tags = { + Name = "coder-${data.coder_workspace_owner.me.name}-${data.coder_workspace.me.name}" + # Required if you are using our example policy, see template README + Coder_Provisioned = "true" + } + + lifecycle { + # NixOS AMIs are republished weekly and garbage-collected after 90 days. + # Without this, a new AMI id replaces every live workspace. + ignore_changes = [ami] + + precondition { + # nonsensitive because user-data contains the token, so its length is + # sensitive by propagation and Terraform would suppress the message. + condition = nonsensitive(length(local.user_data)) < 16384 + error_message = "Rendered user-data is ${nonsensitive(length(local.user_data))} bytes; EC2 allows at most 16384." + } + } +} + +resource "coder_metadata" "workspace_info" { + resource_id = aws_instance.dev.id + item { + key = "AMI" + value = data.aws_ami.nixos.name + } + item { + key = "Flake URI" + value = "${local.flake_url}?ref=${var.flake_branch}#${local.flake_attr}" + } + item { + key = "Build logs location" + value = local.log_dir + } +} + +resource "aws_ec2_instance_state" "dev" { + instance_id = aws_instance.dev.id + state = data.coder_workspace.me.transition == "start" ? "running" : "stopped" +} diff --git a/registry/coder/templates/aws-nixos/modules/nix/README.md b/registry/coder/templates/aws-nixos/modules/nix/README.md new file mode 100644 index 000000000..8cd664fc4 --- /dev/null +++ b/registry/coder/templates/aws-nixos/modules/nix/README.md @@ -0,0 +1,62 @@ +# Nix flake lifecycle + +`lifecycle.sh.tftpl` is the only part of this template that knows about Nix. +It is kept separate so it can be lifted into a standalone Coder registry +module without untangling it from EC2 and Coder specifics first. + +## Contract + +Inputs are environment variables, set by the caller: + +| Variable | Meaning | +| ---------------- | ---------------------------------------------- | +| `NIX_FLAKE_DIR` | checkout to sync and build (e.g. `/etc/nixos`) | +| `NIX_FLAKE_ATTR` | `nixosConfigurations` attribute | +| `NIX_STATE_DIR` | revision marker and lock file | +| `NIX_LOG_DIR` | transcripts | + +Output goes through a `nix_log ` function that the caller +may define; it falls back to stdout. That hook is the only coupling to +Coder, and it is one function. + +| Function | Responsibility | +| --------------------------------------- | ----------------------------------------------- | +| `nix_sync_checkout ` | clone, fast-forward, or leave local work alone | +| `nix_needs_rebuild` | echo the revision, return 0 when work is needed | +| `nix_apply ` | build and apply | +| `nix_record_rev ` | record what was applied | +| `nix_pending_generation` | true when a generation is staged but not booted | +| `nix_filter_log` | drop noise from nix's output | +| `nix_lock [nowait]` | serialise concurrent callers | + +## Where this is going + +The eventual `registry/coder/modules/nix/` should manage a flake lifecycle on +any Linux host, not only NixOS: `nix develop`, `nix profile`, devshells, with +`nixos-rebuild` as one backend among several. + +That is why the split is **resolve → decide → apply**, and why `nix_apply` is +a single function. It is the only NixOS-specific piece, so a `nix develop` +backend becomes a sibling of it rather than a rewrite. `nix_flake_rev`, +`nix_needs_rebuild`, `nix_filter_log` and `nix_lock` are already +platform-agnostic. + +No Terraform module is published yet; this is structure and documentation. + +## Two invariants worth not breaking + +**Nothing is injected at evaluation time.** The commands this module runs are +exactly what a user can type by hand -- no `--override-input`, no `--impure`, +no `--no-write-lock-file`. That is deliberate: the template does not control +the flake, and a rebuild that only works with special flags is a rebuild the +user cannot reproduce. Facts about the workspace are exposed as a runtime +JSON file instead, and the agent token lives in a tmpfs file at mode 0600. +Do not "simplify" either into a flake input -- anything Nix sees is +world-readable in the store and persists across generations. + +**Local work is never discarded.** `nix_sync_checkout` fast-forwards only a +clean checkout on its tracking branch; a dirty tree or local commits are +built as they are. + +**Logging is best-effort.** A failure to report progress must never abort a +rebuild. `nix_log` failures are swallowed by the caller. diff --git a/registry/coder/templates/aws-nixos/modules/nix/lifecycle.sh b/registry/coder/templates/aws-nixos/modules/nix/lifecycle.sh new file mode 100644 index 000000000..f03a6151f --- /dev/null +++ b/registry/coder/templates/aws-nixos/modules/nix/lifecycle.sh @@ -0,0 +1,239 @@ +# shellcheck shell=bash +# +# Nix flake lifecycle: sync a checkout, decide, apply. No EC2 and no Coder API +# calls. Inputs and contract: see README.md in this directory. + +export NIX_CONFIG="experimental-features = nix-command flakes" + +# Callers run as root from user-data and as the workspace user from a +# coder_script, so privilege handling belongs here rather than in each. +NIX_SUDO="" +[ "$(id -u)" -eq 0 ] || NIX_SUDO="sudo" + +command -v nix_log > /dev/null 2>&1 || nix_log() { + shift + printf '%s\n' "$*" +} + +nix_strip_ansi() { + sed -e 's/\x1b\[[0-9;]*[a-zA-Z]//g' -e 's/\r$//' +} + +# --------------------------------------------------------------------------- +# the checkout +# --------------------------------------------------------------------------- +# +# The configuration is a real git checkout at NIX_FLAKE_DIR (/etc/nixos), and +# that is what gets built. It is what makes a bare `nixos-rebuild switch` +# work, since nixos-rebuild looks for /etc/nixos/flake.nix on its own, and it +# means the configuration running the machine is something the user can read +# and edit. +# +# Local work is never discarded: we fast-forward only a clean checkout sitting +# on its tracking branch. A workspace whose configuration silently reverted on +# restart would be worse than one that drifts. +nix_checkout_dirty() { + [ -n "$($NIX_SUDO git -C "$NIX_FLAKE_DIR" status --porcelain 2> /dev/null | head -1)" ] +} + +nix_checkout_rev() { + $NIX_SUDO git -C "$NIX_FLAKE_DIR" rev-parse HEAD 2> /dev/null || true +} + +# Keep the whole tree owned by whoever owns the directory. We clone and fetch +# as root, which otherwise leaves .git root-owned inside a user-owned +# directory -- the user can edit files but every git command fails on +# .git/index.lock. +nix_own_checkout() { + local owner + owner=$(stat -c %U "$NIX_FLAKE_DIR" 2> /dev/null || echo root) + $NIX_SUDO chown -R "$owner" "$NIX_FLAKE_DIR" 2> /dev/null || true +} + +nix_sync_checkout() { + local ref="$1" branch="$2" upstream_rev local_rev + + if [ ! -e "$NIX_FLAKE_DIR/flake.nix" ]; then + nix_log info "Cloning $ref into $NIX_FLAKE_DIR" + $NIX_SUDO install -d -m 0755 "$NIX_FLAKE_DIR" + # Clone into a temporary directory and move the contents, because the + # directory already exists (systemd-tmpfiles creates it) and git refuses + # to clone into a non-empty one. + local tmp + tmp="$($NIX_SUDO mktemp -d)" + if ! $NIX_SUDO git clone --branch "$branch" "$ref" "$tmp/repo" 2>&1 | nix_filter_log; then + nix_log error "Could not clone $ref" + $NIX_SUDO rm -rf "$tmp" + return 1 + fi + $NIX_SUDO sh -c "cd '$tmp/repo' && tar cf - ." | $NIX_SUDO tar xf - -C "$NIX_FLAKE_DIR" + $NIX_SUDO rm -rf "$tmp" + nix_own_checkout + return 0 + fi + + if nix_checkout_dirty; then + nix_log info "$NIX_FLAKE_DIR has local changes; building those instead of $branch" + return 0 + fi + + # Fetching as root writes into .git, so re-assert ownership afterwards. + $NIX_SUDO git -C "$NIX_FLAKE_DIR" fetch --quiet origin "$branch" 2>&1 | nix_filter_log || { + nix_log warn "Could not reach the remote; building the existing checkout" + return 0 + } + + local_rev="$(nix_checkout_rev)" + upstream_rev="$($NIX_SUDO git -C "$NIX_FLAKE_DIR" rev-parse FETCH_HEAD 2> /dev/null || true)" + [ -n "$upstream_rev" ] || return 0 + [ "$local_rev" != "$upstream_rev" ] || return 0 + + # Fast-forward only. A checkout carrying local commits is left alone. + if $NIX_SUDO git -C "$NIX_FLAKE_DIR" merge-base --is-ancestor "$local_rev" "$upstream_rev" 2> /dev/null; then + nix_log info "Updating $NIX_FLAKE_DIR to ${upstream_rev:0:12}" + $NIX_SUDO git -C "$NIX_FLAKE_DIR" reset --hard --quiet "$upstream_rev" + nix_own_checkout + else + nix_log info "$NIX_FLAKE_DIR has local commits; building those instead of $branch" + fi +} + +# --------------------------------------------------------------------------- +# deciding +# --------------------------------------------------------------------------- + +nix_recorded_rev() { + cat "$NIX_STATE_DIR/flake.rev" 2> /dev/null || true +} + +nix_record_rev() { + $NIX_SUDO install -d -m 0755 "$NIX_STATE_DIR" + printf '%s\n' "$1" | $NIX_SUDO tee "$NIX_STATE_DIR/flake.rev" > /dev/null +} + +# True when a generation is the boot default but is not the running system, +# i.e. `boot` was used. A revision check alone would call that up to date. +# +# Compares against /run/current-system, the *activated* system, and not +# /run/booted-system, which is whatever the kernel booted. After a switch the +# booted symlink still points at the previous generation -- so using it here +# reports every freshly created workspace as needing a restart. +nix_pending_generation() { + [ "$(readlink -f /run/current-system)" != "$(readlink -f /nix/var/nix/profiles/system)" ] +} + +# Echoes a revision (or "dirty") and returns 0 when a rebuild is needed. +nix_needs_rebuild() { + local rev + + if nix_checkout_dirty; then + # Uncommitted work has no revision to compare, so always rebuild. + printf 'dirty' + return 0 + fi + + rev="$(nix_checkout_rev)" + [ -n "$rev" ] || rev="unknown" + printf '%s' "$rev" + + [ "$rev" = "$(nix_recorded_rev)" ] || return 0 + nix_pending_generation +} + +# --------------------------------------------------------------------------- +# applying +# --------------------------------------------------------------------------- + +# Drop the things that are noise rather than progress; everything else reaches +# the caller as nix wrote it. Nix's plain output is already the readable +# output -- every --log-format is byte-identical off a TTY. +nix_filter_log() { + local line prefix + while IFS= read -r line || [ -n "$line" ]; do + case "$line" in + *$'\033'*) line=$(printf '%s' "$line" | nix_strip_ansi) ;; + esac + case "$line" in + "") continue ;; + # The enumerated store paths under "these N derivations will be built:" + " "*/nix/store/*) continue ;; + # Those list headers end in a colon promising the list we just dropped, + # which reads as truncated output. Restate them as complete sentences. + "these "*" will be "*: | "this "*" will be "*:) + case "$line" in + "these "*) line=${line#these } ;; + "this "*) line="1 ${line#this }" ;; + esac + printf '%s\n' "${line%:}" + continue + ;; + # git's fetch progress, including the indented ref-update line. + "remote: "* | "From "* | " "*".."*"->"*) continue ;; + # Per-derivation build output, "> text". The "building '...'" + # line already marks it and the text is in the transcript. Requiring a + # space-free prefix stops this eating lines that contain "> ". + *"> "*) + prefix=${line%%> *} + case "$prefix" in + "" | *[[:space:]]*) ;; + *) continue ;; + esac + ;; + esac + + # Nix writes its progress in lowercase ("building the system + # configuration..."), and in the workspace UI those lines sit among + # sentences from every other log source. Capitalise the first letter -- + # but only when the first word is a plain word, so that program names + # ("nixos-rebuild: ...") and paths stay exactly as they were written. + case "${line%% *}" in + [a-z]*[!a-zA-Z:]*) ;; + [a-z]*) line="${line^}" ;; + esac + + printf '%s\n' "$line" + done +} + +# `switch` and `boot` both build before activating, so a configuration that +# fails to build never touches the running system; a separate `nix build` +# gate would add nothing. +# +# No --override-input, no --refresh, no --no-write-lock-file: the flake is a +# local checkout that we fetched explicitly, so the command below is exactly +# what a user can type by hand. +nix_apply() { + local operation="$1" transcript="$2" rc + + $NIX_SUDO install -d -m 0755 "$NIX_LOG_DIR" + $NIX_SUDO install -m 0644 /dev/null "$transcript" + $NIX_SUDO ln -sfn "$transcript" "$NIX_LOG_DIR/rebuild-latest.log" + + # stdbuf so output streams instead of arriving in one burst at the end. + set +e + $NIX_SUDO nixos-rebuild "$operation" \ + --flake "$NIX_FLAKE_DIR#$NIX_FLAKE_ATTR" \ + --print-build-logs \ + 2>&1 | stdbuf -oL $NIX_SUDO tee -a "$transcript" | nix_filter_log \ + | while IFS= read -r line; do nix_log info "$line"; done + rc=${PIPESTATUS[0]} + set -e + + return "$rc" +} + +# coder_script does not prevent overlapping cron runs, and two concurrent +# nixos-rebuild processes are a bad time. +nix_lock() { + local lock="$NIX_STATE_DIR/rebuild.lock" + $NIX_SUDO install -d -m 0755 "$NIX_STATE_DIR" + # Mode 0666 so the workspace user can take the same advisory lock as root. + # It carries no data, only the lock. + [ -e "$lock" ] || $NIX_SUDO install -m 0666 /dev/null "$lock" + exec 9> "$lock" + if [ "${1:-wait}" = "nowait" ]; then + flock -n 9 + else + flock 9 + fi +} diff --git a/registry/coder/templates/aws-nixos/scripts/bootstrap.sh.tftpl b/registry/coder/templates/aws-nixos/scripts/bootstrap.sh.tftpl new file mode 100644 index 000000000..d0897bea2 --- /dev/null +++ b/registry/coder/templates/aws-nixos/scripts/bootstrap.sh.tftpl @@ -0,0 +1,191 @@ +#!/usr/bin/env bash +# EC2 user-data. The NixOS AMI does not run cloud-init; it runs +# amazon-init.service, which execs this as a shell script when it starts with +# `#!` -- after multi-user.target, on every boot. So it must be idempotent. + +set -euo pipefail +export PATH="/run/current-system/sw/bin:$PATH" +export HOME=/root + +FLAKE_REF='${ARG_FLAKE_REF}' +FLAKE_BRANCH='${ARG_FLAKE_BRANCH}' +FLAKE_ATTR='${ARG_FLAKE_ATTR}' +ACCESS_URL='${ARG_ACCESS_URL}' +AGENT_TOKEN='${ARG_AGENT_TOKEN}' +INIT_SCRIPT_B64='${ARG_INIT_SCRIPT_B64}' +LOG_SOURCE_ID='${ARG_LOG_SOURCE_ID}' +HOSTNAME_='${ARG_HOSTNAME}' +WORKSPACE_NAME='${ARG_WORKSPACE_NAME}' +OWNER='${ARG_OWNER}' +OWNER_NAME_B64='${ARG_OWNER_NAME_B64}' +OWNER_EMAIL='${ARG_OWNER_EMAIL}' + +RUNTIME_DIR=/run/coder +STATE_DIR=/var/lib/coder-nixos +LOG_DIR=/var/log/coder-nixos +FLAKE_DIR=/etc/nixos + +# --- 1. Hand the agent its token ------------------------------------------- +# +# Done first, before anything that can fail, so that the agent can be started +# from any later exit path -- including a failed rebuild, where a reachable +# workspace is the only way to debug the flake that broke it. +# +# The token stays out of Nix on purpose. As a flake input it would be baked +# into a derivation, land world-readable in /nix/store and persist across +# generations and past rotation. RUNTIME_DIR is a tmpfs, so nothing here +# survives a reboot -- which is the point, since the token rotates on every +# workspace start. + +install -d -m 0700 "$RUNTIME_DIR" +install -d -m 0755 "$STATE_DIR" "$LOG_DIR" +rm -f "$RUNTIME_DIR/ready" + +umask 077 +printf 'CODER_AGENT_TOKEN=%s\nCODER_AGENT_URL=%s\n' "$AGENT_TOKEN" "$ACCESS_URL" \ + >"$RUNTIME_DIR/agent.env" +umask 022 +printf '%s' "$INIT_SCRIPT_B64" | base64 -d >"$RUNTIME_DIR/init.sh" +chmod 0755 "$RUNTIME_DIR/init.sh" + +# The unit runs as the workspace user, so it has to be able to traverse the +# directory and read the token. Mode stays 0600/0700; only the owner changes. +# +# The owner is whoever the flake says the agent runs as -- asked of the unit +# rather than guessed by the template, so a configuration we do not control +# cannot disagree with us. +# +# On the very first boot the unit does not exist yet, so this is a no-op and +# is retried after the switch, before the agent is started. +chown_handoff() { + local target + target=$(systemctl show coder-agent -p User --value 2>/dev/null || true) + [ -n "$target" ] || return 0 + chown -R "$target" "$RUNTIME_DIR" 2>/dev/null || true +} +chown_handoff +touch "$RUNTIME_DIR/ready" +chown_handoff + +[ -z "$HOSTNAME_" ] || hostnamectl set-hostname "$HOSTNAME_" || true + +# --- 2. Logging ------------------------------------------------------------- + +CODER_ACCESS_URL="$ACCESS_URL" +CODER_AGENT_TOKEN="$AGENT_TOKEN" +CODER_LOG_SOURCE_ID="$LOG_SOURCE_ID" +CODER_LOG_STATE_DIR="$RUNTIME_DIR" +# Half of Coder's 1 MiB per-agent cap, leaving room for coder_script output. +# Overflowing the cap is permanent and silently drops all later logs. +CODER_LOG_BUDGET=524288 + +${LOG_SH} + +TRANSCRIPT="$LOG_DIR/rebuild-$(date -u +%Y%m%dT%H%M%SZ).log" +coder_log_init "NixOS" "/icon/nix.svg" || true + +nix_log() { coder_log "$@" || true; } + +# The agent is not `wantedBy = multi-user.target` in the flake; nothing starts +# it but this. That is what keeps the workspace from being handed to the user +# on the generation we are about to replace -- startup scripts would install +# into a system that is seconds from being swapped out. +# +# So every exit path from here on has to go through this, including the +# failure ones: a workspace whose flake does not build is exactly the +# workspace someone needs a terminal on. +start_agent() { + if systemctl is-active --quiet coder-agent; then + return 0 + fi + if ! systemctl cat coder-agent >/dev/null 2>&1; then + coder_log error "coder-agent.service does not exist: the flake has never been applied." || true + return 1 + fi + + chown_handoff + systemctl start coder-agent || true + + # Type=simple, so systemd reports the unit active as soon as it has forked. + # Give a start that is going to fail immediately the chance to do so. + sleep 3 + if systemctl is-active --quiet coder-agent; then + coder_log info "coder-agent is active." || true + return 0 + fi + + coder_log error "coder-agent failed to start." || true + journalctl -u coder-agent --no-pager --lines=30 2>/dev/null | coder_log_pipe error || true + return 1 +} + +on_error() { + local rc=$? + coder_log error "Boot script failed (exit $rc). Transcript: $TRANSCRIPT" || true + coder_log_tail "$TRANSCRIPT" 200 || true + start_agent || true + exit "$rc" +} +trap on_error ERR + +# --- 3. Facts about this workspace ----------------------------------------- +# +# A runtime interface, not an evaluation input: a flake cannot read an +# absolute path outside itself in pure evaluation mode, so consuming this at +# eval time would need --impure and would stop `nixos-rebuild switch` from +# reproducing what we apply. Configurations read it from a service instead. +# +# Identity only. The token lives in agent.env at 0600 and never appears here. + +cat >"$RUNTIME_DIR/workspace.json" < /dev/null 2>&1; then + CODER_CURL=$(command -v curl) + else + CODER_CURL="$(nix build --no-link --print-out-paths nixpkgs#curl 2> /dev/null)/bin/curl" + fi + [ -x "$CODER_CURL" ] || return 1 + export CODER_CURL + fi + "$CODER_CURL" "$@" +} + +# Coder caps agent logs at 1 MiB per AGENT, shared across every log source. +# Overflowing does not truncate: the batch is rejected and the agent is +# flagged overflowed permanently, silently dropping all later logs from every +# source. So track our own usage and go quiet before the server says no. +coder_log_budget_left() { + local used + used=$(cat "$CODER_LOG_STATE_DIR/log-budget" 2> /dev/null || echo 0) + echo $((CODER_LOG_BUDGET - used)) +} + +coder_log_budget_add() { + local used + used=$(cat "$CODER_LOG_STATE_DIR/log-budget" 2> /dev/null || echo 0) + echo $((used + $1)) > "$CODER_LOG_STATE_DIR/log-budget" 2> /dev/null || true +} + +# Idempotent: Coder swallows a duplicate id, so this can run on every boot +# with a fixed UUID. Retries on 401 because the agent record is only created +# when the provisioner job completes -- an instance can boot before then. +coder_log_init() { + local display_name="$1" icon="$2" attempt=0 code + mkdir -p "$CODER_LOG_STATE_DIR" 2> /dev/null || true + + while [ "$attempt" -lt 40 ]; do + code=$( + coder_curl -sS -o /dev/null -w '%{http_code}' -X POST \ + "$CODER_ACCESS_URL/api/v2/workspaceagents/me/log-source" \ + -H "Coder-Session-Token: $CODER_AGENT_TOKEN" \ + -H 'Content-Type: application/json' \ + --data-binary @- << JSON || echo 000 +{"id":"$CODER_LOG_SOURCE_ID","display_name":"$display_name","icon":"$icon"} +JSON + ) + case "$code" in + # The handler returns 201 on both create and already-exists. + 200 | 201) + CODER_LOG_READY=1 + return 0 + ;; + 401 | 403 | 000 | 5??) + attempt=$((attempt + 1)) + sleep 15 + ;; + *) return 1 ;; + esac + done + return 1 +} + +coder_log_json() { + local level="$1" line="$2" + [ "${#line}" -le "$CODER_LOG_MAX_LINE" ] || line="${line:0:$CODER_LOG_MAX_LINE}..." + line=$(printf '%s' "$line" | sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\r//g' -e 's/\t/ /g') + printf '{"created_at":"%s","level":"%s","output":"%s"}' \ + "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$level" "$line" +} + +coder_log_send() { + local payload size + payload=$(paste -sd, -) + [ -n "$payload" ] || return 0 + + size=${#payload} + [ "$(coder_log_budget_left)" -gt "$size" ] || return 0 + coder_log_budget_add "$size" + + coder_curl -sS -o /dev/null -X PATCH \ + "$CODER_ACCESS_URL/api/v2/workspaceagents/me/logs" \ + -H "Coder-Session-Token: $CODER_AGENT_TOKEN" \ + -H 'Content-Type: application/json' \ + --data-binary @- << JSON || true +{"log_source_id":"$CODER_LOG_SOURCE_ID","logs":[$payload]} +JSON +} + +coder_log() { + local level="$1" + shift + [ "$CODER_LOG_READY" = 1 ] || return 0 + coder_log_json "$level" "$*" | coder_log_send +} + +# Reads plain lines on stdin and ships them in batches. +coder_log_pipe() { + local level="${1:-info}" line batch="" n=0 bytes=0 obj + [ "$CODER_LOG_READY" = 1 ] || { + cat > /dev/null + return 0 + } + + while IFS= read -r line || [ -n "$line" ]; do + obj=$(coder_log_json "$level" "$line") + batch="${batch:+$batch +}$obj" + n=$((n + 1)) + bytes=$((bytes + ${#obj})) + if [ "$n" -ge 50 ] || [ "$bytes" -ge 32768 ]; then + printf '%s\n' "$batch" | coder_log_send + batch="" + n=0 + bytes=0 + fi + done + + [ -z "$batch" ] || printf '%s\n' "$batch" | coder_log_send + return 0 +} + +# Tail of a failed transcript, so the UI shows the actual error. +coder_log_tail() { + local file="$1" lines="${2:-200}" + [ -f "$file" ] || return 0 + coder_log error "--- last $lines lines of $file ---" + tail -n "$lines" "$file" | coder_log_pipe error +} diff --git a/registry/coder/templates/aws-nixos/scripts/rebuild.sh.tftpl b/registry/coder/templates/aws-nixos/scripts/rebuild.sh.tftpl new file mode 100644 index 000000000..cf0c170fa --- /dev/null +++ b/registry/coder/templates/aws-nixos/scripts/rebuild.sh.tftpl @@ -0,0 +1,76 @@ +#!/usr/bin/env bash +# Periodic nixos-rebuild, run by a coder_script on a cron schedule. +# +# coder_script bodies run as the workspace user through its login shell and +# there is no run_as argument, so everything privileged goes through sudo. + +set -euo pipefail + +FLAKE_REF='${ARG_FLAKE_REF}' +FLAKE_BRANCH='${ARG_FLAKE_BRANCH}' +FLAKE_ATTR='${ARG_FLAKE_ATTR}' +OPERATION='${ARG_UPDATE_PROCESS}' +ACCESS_URL='${ARG_ACCESS_URL}' +LOG_SOURCE_ID='${ARG_LOG_SOURCE_ID}' + +STATE_DIR=/var/lib/coder-nixos +LOG_DIR=/var/log/coder-nixos +FLAKE_DIR=/etc/nixos + +CODER_ACCESS_URL="$ACCESS_URL" +CODER_LOG_SOURCE_ID="$LOG_SOURCE_ID" +CODER_LOG_STATE_DIR="$STATE_DIR" +# Always present: the agent sets it for the scripts it runs. Logging simply +# stays disabled if it is not. +CODER_AGENT_TOKEN="$${CODER_AGENT_TOKEN:-}" +CODER_LOG_BUDGET=524288 + +${LOG_SH} + +NIX_FLAKE_DIR="$FLAKE_DIR" +NIX_FLAKE_ATTR="$FLAKE_ATTR" +NIX_STATE_DIR="$STATE_DIR" +NIX_LOG_DIR="$LOG_DIR" + +${LIFECYCLE_SH} + +# Don't queue: if the boot script or a previous tick still holds the lock, +# doing nothing is correct. coder_script does not prevent overlapping cron +# runs, so this is the only thing that does. +if ! nix_lock nowait; then + echo "Another nixos-rebuild is in progress; skipping this run." + exit 0 +fi + +TRANSCRIPT="$LOG_DIR/rebuild-$(date -u +%Y%m%dT%H%M%SZ).log" + +nix_sync_checkout "$FLAKE_REF" "$FLAKE_BRANCH" + +REV=$(nix_needs_rebuild) && NEEDS_REBUILD=1 || NEEDS_REBUILD=0 +if [ "$NEEDS_REBUILD" -eq 0 ]; then + echo "Already up to date ($${REV:0:12})." + exit 0 +fi + +echo "Running nixos-rebuild $OPERATION for $FLAKE_DIR#$FLAKE_ATTR ($${REV:0:12})" +echo "Transcript: $TRANSCRIPT" + +if ! nix_apply "$OPERATION" "$TRANSCRIPT"; then + echo "nixos-rebuild $OPERATION failed; the running system is untouched." >&2 + echo "See $TRANSCRIPT." >&2 + exit 1 +fi +nix_record_rev "$REV" + +case "$OPERATION" in + boot) + if nix_pending_generation; then + echo "A new system generation is staged and applies on the next restart." + fi + ;; + switch) + # The agent survives because coder-agent.service is declared with + # restartIfChanged = false. + echo "New system generation is active." + ;; +esac From 580f1412905e22c37eed8791bb4ba86b90f64613 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 21 Sep 2026 12:27:59 +0000 Subject: [PATCH 02/24] docs(AGENTS.md): require /usr/bin/env bash in module scripts NixOS workspaces have no /bin/bash, and the failure mode is an agent script that exits 255 with an empty log. --- AGENTS.md | 1 + 1 file changed, 1 insertion(+) diff --git a/AGENTS.md b/AGENTS.md index 22007f657..4796156dc 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -109,6 +109,7 @@ output "scripts" { - Use `tf` (not `hcl`) for code blocks in README; use relative icon paths (e.g., `../../../../.icons/`) - **Never include parameter listings or input/output variable tables in module or template READMEs.** This includes workspace parameters declared with `coder_parameter`. The registry automatically parses the Terraform source and displays parameters in a dedicated tab on `registry.coder.com`; input/output documentation is also generated from the source. Duplicating these listings in the README is redundant and creates maintenance drift. - Usage examples (e.g., a `module "..." { }` block) and explanations of parameter behavior are encouraged, but not tables or lists enumerating parameters, inputs, or outputs. +- Script shebangs must be `#!/usr/bin/env bash`, never `#!/bin/bash`. NixOS workspaces have only `/bin/sh`, so the kernel fails the exec before anything runs and the agent reports exit 255 with an empty log — which looks like a broken module, not a missing interpreter. ### Variable and output conventions From 95dd235a6e235719c7437f38ff618c71aae1df8c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= <57866459+phorcys420@users.noreply.github.com> Date: Mon, 21 Sep 2026 14:52:06 +0200 Subject: [PATCH 03/24] refactor(aws-nixos): move the amazon-init logic into its own module (#1134) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Stacked on `phorcys/nixos-tests` (the `aws-nixos` template itself), so the diff here is only the split. ## Why The template mixed two concerns that change for different reasons: - **how a NixOS machine is configured from a flake** — already isolated in `modules/nix/` - **how EC2 gets a script onto a NixOS AMI at all** — spread across `main.tf` locals and `scripts/bootstrap.sh.tftpl` The second is the part that is not portable. It exists entirely because the NixOS AMI runs `amazon-init.service` instead of cloud-init: one hook, no `runcmd`, no `write_files`, no once-vs-per-boot distinction, and nothing can be ordered `After` it without deadlocking the first boot. Another cloud replaces that half wholesale and leaves the Nix half untouched. ## What moved `registry/coder/templates/aws-nixos/modules/amazon-init/` now owns: - `scripts/bootstrap.sh.tftpl` (moved, unchanged) - the `templatefile()` call and its arguments - the self-extracting user-data wrapper and the reason it exists (the script plus its two shell libraries plus the agent init script come to ~19 KiB against EC2's 16 KiB limit; compressed it is ~9 KiB) It exports `user_data` (sensitive — the agent token is inside it), `user_data_bytes` unwrapped so the caller can still assert the limit in a `precondition`, and `bootstrap_path`. The template now passes in the workspace identity and the flake to build, and no longer knows how any of it is delivered. The two shell libraries stay in the template and are passed as strings, so `scripts/rebuild.sh.tftpl` keeps sharing one copy of each. ## Verification Behaviour-preserving: same script, same wrapper, same rendered user-data. - `terraform validate`, `readmevalidation`, `shellcheck --severity=warning`, `bun run fmt` all clean - Built a real workspace from the refactored template on a test deployment: clone, rebuild, `Switch complete; starting the agent`, then the startup scripts — identical ordering to before Generated with [Xum](https://mux.coder.com/) using Claude. --- registry/coder/templates/aws-nixos/README.md | 5 +- registry/coder/templates/aws-nixos/main.tf | 66 +++----- .../aws-nixos/modules/amazon-init/README.md | 59 ++++++++ .../aws-nixos/modules/amazon-init/main.tf | 143 ++++++++++++++++++ .../amazon-init}/scripts/bootstrap.sh.tftpl | 0 5 files changed, 231 insertions(+), 42 deletions(-) create mode 100644 registry/coder/templates/aws-nixos/modules/amazon-init/README.md create mode 100644 registry/coder/templates/aws-nixos/modules/amazon-init/main.tf rename registry/coder/templates/aws-nixos/{ => modules/amazon-init}/scripts/bootstrap.sh.tftpl (100%) diff --git a/registry/coder/templates/aws-nixos/README.md b/registry/coder/templates/aws-nixos/README.md index 5b68415bf..4d0e8851e 100644 --- a/registry/coder/templates/aws-nixos/README.md +++ b/registry/coder/templates/aws-nixos/README.md @@ -318,4 +318,7 @@ it is reproducible and avoids the dynamic-linking problem entirely. The Nix-specific parts of the boot and rebuild paths live in [`modules/nix/`](./modules/nix/README.md), kept separate so they can become a standalone Coder -module that manages a flake lifecycle on any Linux host, not just NixOS. +module that manages a flake lifecycle on any Linux host, not just NixOS. The EC2 side — the boot +script and the user-data wrapper that `amazon-init` execs — is +[`modules/amazon-init/`](./modules/amazon-init/README.md), so the two halves can move +independently: another cloud replaces the second without touching the first. diff --git a/registry/coder/templates/aws-nixos/main.tf b/registry/coder/templates/aws-nixos/main.tf index fb81bf699..c0409ea64 100644 --- a/registry/coder/templates/aws-nixos/main.tf +++ b/registry/coder/templates/aws-nixos/main.tf @@ -324,49 +324,35 @@ locals { log_sh = file("${path.module}/scripts/log.sh") lifecycle_sh = file("${path.module}/modules/nix/lifecycle.sh") - # EC2 caps user-data at 16 KiB, and the boot script plus its two libraries - # plus the agent init script come to roughly 19 KiB. So user-data is a - # six-line self-extracting wrapper around a compressed copy. - # - # This is transparent to the NixOS AMI: amazon-init only inspects the first - # two bytes for `#!` before exec'ing the blob, and it has no decompression - # step of its own. Extracting to a fixed path also means the real script is - # on disk when something needs debugging. - user_data = <<-SH - #!/usr/bin/env bash - set -eu - install -d -m 0700 /run/coder - base64 -d <<'CODER_PAYLOAD' | gzip -dc >/run/coder/bootstrap.sh - ${base64gzip(local.bootstrap)} - CODER_PAYLOAD - exec bash /run/coder/bootstrap.sh - SH - - bootstrap = templatefile("${path.module}/scripts/bootstrap.sh.tftpl", { - LOG_SH = local.log_sh - LIFECYCLE_SH = local.lifecycle_sh - ARG_FLAKE_REF = local.flake_url - ARG_FLAKE_BRANCH = var.flake_branch - ARG_FLAKE_ATTR = local.flake_attr - ARG_ACCESS_URL = data.coder_workspace.me.access_url - ARG_AGENT_TOKEN = try(coder_agent.main[0].token, "") - ARG_INIT_SCRIPT_B64 = base64encode(try(coder_agent.main[0].init_script, "")) - ARG_LOG_SOURCE_ID = local.log_source_id - ARG_HOSTNAME = lower(data.coder_workspace.me.name) - ARG_WORKSPACE_NAME = data.coder_workspace.me.name - ARG_OWNER = data.coder_workspace_owner.me.name - # base64 because a full name may contain quotes and is interpolated into - # both a shell string and a Nix string. - ARG_OWNER_NAME_B64 = base64encode(coalesce(data.coder_workspace_owner.me.full_name, data.coder_workspace_owner.me.name)) - ARG_OWNER_EMAIL = data.coder_workspace_owner.me.email - }) +} + +# Everything about getting Coder onto a NixOS AMI through amazon-init: the +# agent handoff, the workspace facts and the first rebuild, wrapped for EC2's +# 16 KiB user-data limit. See ./modules/amazon-init/README.md. +module "amazon_init" { + source = "./modules/amazon-init" + + flake_ref = local.flake_url + flake_branch = var.flake_branch + flake_attr = local.flake_attr + access_url = data.coder_workspace.me.access_url + agent_token = try(coder_agent.main[0].token, "") + agent_init_script = try(coder_agent.main[0].init_script, "") + log_source_id = local.log_source_id + workspace_name = data.coder_workspace.me.name + hostname = lower(data.coder_workspace.me.name) + owner = data.coder_workspace_owner.me.name + owner_name = coalesce(data.coder_workspace_owner.me.full_name, data.coder_workspace_owner.me.name) + owner_email = data.coder_workspace_owner.me.email + log_library = local.log_sh + lifecycle_library = local.lifecycle_sh } resource "aws_instance" "dev" { ami = data.aws_ami.nixos.id availability_zone = "${module.aws_region.value}a" instance_type = data.coder_parameter.instance_type.value - user_data = local.user_data + user_data = module.amazon_init.user_data # The agent token is inside user-data and rotates on every workspace start, # so user-data changes on every start. With replacement enabled, every @@ -391,10 +377,8 @@ resource "aws_instance" "dev" { ignore_changes = [ami] precondition { - # nonsensitive because user-data contains the token, so its length is - # sensitive by propagation and Terraform would suppress the message. - condition = nonsensitive(length(local.user_data)) < 16384 - error_message = "Rendered user-data is ${nonsensitive(length(local.user_data))} bytes; EC2 allows at most 16384." + condition = module.amazon_init.user_data_bytes < 16384 + error_message = "Rendered user-data is ${module.amazon_init.user_data_bytes} bytes; EC2 allows at most 16384." } } } diff --git a/registry/coder/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder/templates/aws-nixos/modules/amazon-init/README.md new file mode 100644 index 000000000..e0d124fd4 --- /dev/null +++ b/registry/coder/templates/aws-nixos/modules/amazon-init/README.md @@ -0,0 +1,59 @@ +# amazon-init + +The platform half of this template: everything that is true because the +workspace is a NixOS AMI on EC2, and nothing that is true because it is Nix +(that is [`../nix`](../nix/README.md)) or because it is Coder. + +It renders one thing — EC2 user-data — and is the only part of the template +that knows how a NixOS instance is bootstrapped. + +## Why this is not cloud-init + +The NixOS AMI does not run cloud-init. It runs `amazon-init.service`, which +reads `/etc/ec2-metadata/user-data` and execs it as a shell script when it +begins with `#!` — after `multi-user.target`, **on every boot**. There is no +`runcmd`, no `write_files`, no per-boot/once distinction, and no ordering +hooks. Three things follow, and they shape the whole script: + +- It must be idempotent, because it runs again on every restart. +- It is the only hook available, so the agent handoff, the workspace facts and + the rebuild all have to live in it. +- Nothing can be ordered `After` it. `nixos-rebuild switch` starts new units + synchronously and `amazon-init` cannot become active until its script exits, + so a unit that waits for it deadlocks the first boot. This is why the Coder + agent unit is started _by_ this script rather than wanted by a target. + +## What the script does + +1. Publishes the agent handoff to `/run/coder` — `agent.env` (0600), `init.sh`, + then `ready` last — before anything that can fail, so the agent can still be + started from a failure path. +2. Writes the per-workspace facts to `/run/coder/workspace.json`. +3. Syncs the flake checkout and rebuilds, streaming progress to the workspace + UI. +4. Starts `coder-agent.service`, and does so on every exit path, including a + failed rebuild. + +## The user-data wrapper + +The boot script, its two shell libraries and the agent init script come to +roughly 19 KiB, against EC2's 16 KiB limit. So the user-data this module +outputs is a six-line self-extracting wrapper around a gzipped copy, which +lands at about 9 KiB. + +That is transparent to the AMI: `amazon-init` only checks the first two bytes +for `#!` before exec'ing the blob. The wrapper extracts to a fixed path, +`/run/coder/bootstrap.sh`, so the real script is on disk when a boot needs +debugging. + +`user_data_bytes` is exported unwrapped so the caller can assert the limit in a +`precondition` — the output itself is sensitive, because the agent token is +inside it, and Terraform suppresses error messages derived from sensitive +values. + +## Contract + +Inputs are the workspace's identity and the flake to build; the two shell +libraries are passed in as strings rather than read here, so the template's +other entrypoints share one copy. The token is written to a tmpfs at 0600 and +never passed into Nix. diff --git a/registry/coder/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder/templates/aws-nixos/modules/amazon-init/main.tf new file mode 100644 index 000000000..b4534cfd1 --- /dev/null +++ b/registry/coder/templates/aws-nixos/modules/amazon-init/main.tf @@ -0,0 +1,143 @@ +# Everything that has to happen because this is a NixOS AMI on EC2, and +# nothing that has to happen because it is Nix or because it is Coder. +# +# The NixOS AMI does not run cloud-init. It runs amazon-init.service, which +# reads /etc/ec2-metadata/user-data and execs it as a shell script when it +# begins with `#!` -- after multi-user.target, on every boot. That is the only +# hook this platform offers, so it has to carry the agent handoff, the +# workspace facts and the rebuild, and it has to be idempotent. + +terraform { + required_version = ">= 1.0" +} + +variable "flake_ref" { + description = "Git remote to clone into the checkout, as `git clone` understands it." + type = string +} + +variable "flake_branch" { + description = "Branch to track in that remote." + type = string +} + +variable "flake_attr" { + description = "`nixosConfigurations` attribute to build, architecture already resolved." + type = string +} + +variable "access_url" { + description = "Deployment access URL the agent and the log API are reached on." + type = string +} + +variable "agent_token" { + description = "Agent token. Written to /run/coder/agent.env at 0600 on a tmpfs and never passed into Nix." + type = string + sensitive = true +} + +variable "agent_init_script" { + description = "`coder_agent.init_script`, run verbatim once the rebuild has finished." + type = string + sensitive = true +} + +variable "log_source_id" { + description = "Log source the rebuild streams to. Must be stable across builds." + type = string +} + +variable "workspace_name" { + description = "Workspace name, published in /run/coder/workspace.json." + type = string +} + +variable "hostname" { + description = "Hostname to set on the instance." + type = string +} + +variable "owner" { + description = "Workspace owner's username." + type = string +} + +variable "owner_name" { + description = "Workspace owner's full name." + type = string +} + +variable "owner_email" { + description = "Workspace owner's email address." + type = string +} + +variable "log_library" { + description = <<-EOT + Contents of the shell library providing `coder_log`, sourced into the boot + script. Passed in rather than read here so the same copy is shared with + the template's other entrypoints. + EOT + type = string +} + +variable "lifecycle_library" { + description = "Contents of the Nix lifecycle shell library, sourced into the boot script." + type = string +} + +locals { + bootstrap = templatefile("${path.module}/scripts/bootstrap.sh.tftpl", { + LOG_SH = var.log_library + LIFECYCLE_SH = var.lifecycle_library + ARG_FLAKE_REF = var.flake_ref + ARG_FLAKE_BRANCH = var.flake_branch + ARG_FLAKE_ATTR = var.flake_attr + ARG_ACCESS_URL = var.access_url + ARG_AGENT_TOKEN = var.agent_token + ARG_INIT_SCRIPT_B64 = base64encode(var.agent_init_script) + ARG_LOG_SOURCE_ID = var.log_source_id + ARG_HOSTNAME = var.hostname + ARG_WORKSPACE_NAME = var.workspace_name + ARG_OWNER = var.owner + # base64 because a full name may contain quotes and is interpolated into + # a shell string. + ARG_OWNER_NAME_B64 = base64encode(var.owner_name) + ARG_OWNER_EMAIL = var.owner_email + }) + + # EC2 caps user-data at 16 KiB, and the boot script plus its two libraries + # plus the agent init script come to roughly 19 KiB. So user-data is a + # six-line self-extracting wrapper around a compressed copy. + # + # This is transparent to the NixOS AMI: amazon-init only inspects the first + # two bytes for `#!` before exec'ing the blob, and it has no decompression + # step of its own. Extracting to a fixed path also means the real script is + # on disk when something needs debugging. + user_data = <<-SH + #!/usr/bin/env bash + set -eu + install -d -m 0700 /run/coder + base64 -d <<'CODER_PAYLOAD' | gzip -dc >/run/coder/bootstrap.sh + ${base64gzip(local.bootstrap)} + CODER_PAYLOAD + exec bash /run/coder/bootstrap.sh + SH +} + +output "user_data" { + description = "Rendered EC2 user-data. Sensitive: it carries the agent token." + value = local.user_data + sensitive = true +} + +output "user_data_bytes" { + description = "Size of the rendered user-data, for the caller's 16 KiB precondition." + value = nonsensitive(length(local.user_data)) +} + +output "bootstrap_path" { + description = "Where the wrapper extracts the real boot script on the instance." + value = "/run/coder/bootstrap.sh" +} diff --git a/registry/coder/templates/aws-nixos/scripts/bootstrap.sh.tftpl b/registry/coder/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl similarity index 100% rename from registry/coder/templates/aws-nixos/scripts/bootstrap.sh.tftpl rename to registry/coder/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl From 4ea3ce113851dc51c3f11b985f7739b2957aea42 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 21 Sep 2026 13:36:29 +0000 Subject: [PATCH 04/24] refactor(aws-nixos): make amazon-init know nothing about Nix The module still had the flake wired through it: flake_ref, flake_branch and flake_attr as variables, the rebuild inside its boot script, and the caller handing it a "lifecycle library" it never used itself. Splitting the files was not the same as splitting the concern. Now the contract is one opaque string. `boot_script` is run as a child process, as root, after the handoff and workspace facts are published and before the agent is started, and nothing in the module looks inside it. Everything Nix moved to scripts/boot.sh.tftpl in the template, which is where the flake, the checkout and the state directories are named. The module also stopped taking the workspace's identity as arguments and reads coder_workspace and coder_workspace_owner itself, and it now owns the logging library outright: it writes it to the runtime directory, registers the log source, exports the environment the boot script needs, and exposes the library to callers that want coder_log in a coder_script of their own. The `nix build nixpkgs#curl` fallback became curl_resolve_command, since only the caller knows how to get a binary on an image the module has never seen. Two things this shook out: - Payloads are interpolated as text, not base64. Base64 inflates by a third and leaves gzip nothing to compress, which took user-data from ~9 KiB to 18431 bytes against EC2's 16384 limit. As text it is ~10.5 KiB. - CODER_LOG_READY is exported and inherited. It was per-process, so once the boot script became a child it logged into a void -- the rebuild ran correctly and reported nothing. Verified by creating workspaces from the refactored template: clone, rebuild with progressive output, `Switch complete.`, then the agent, then the startup scripts. workspace.json is populated from the module's own data sources. --- registry/coder/templates/aws-nixos/README.md | 18 +- registry/coder/templates/aws-nixos/main.tf | 55 +++-- .../aws-nixos/modules/amazon-init/README.md | 159 ++++++++---- .../aws-nixos/modules/amazon-init/main.tf | 230 ++++++++++++------ .../amazon-init/scripts/bootstrap.sh.tftpl | 168 ++++++------- .../{ => modules/amazon-init}/scripts/log.sh | 23 +- .../templates/aws-nixos/scripts/boot.sh.tftpl | 68 ++++++ 7 files changed, 474 insertions(+), 247 deletions(-) rename registry/coder/templates/aws-nixos/{ => modules/amazon-init}/scripts/log.sh (82%) create mode 100644 registry/coder/templates/aws-nixos/scripts/boot.sh.tftpl diff --git a/registry/coder/templates/aws-nixos/README.md b/registry/coder/templates/aws-nixos/README.md index 4d0e8851e..124e43a0a 100644 --- a/registry/coder/templates/aws-nixos/README.md +++ b/registry/coder/templates/aws-nixos/README.md @@ -316,9 +316,15 @@ For anything heavier, prefer declaring the tool in your flake and exposing it wi [`coder_app`](https://registry.terraform.io/providers/coder/coder/latest/docs/resources/app): it is reproducible and avoids the dynamic-linking problem entirely. -The Nix-specific parts of the boot and rebuild paths live in -[`modules/nix/`](./modules/nix/README.md), kept separate so they can become a standalone Coder -module that manages a flake lifecycle on any Linux host, not just NixOS. The EC2 side — the boot -script and the user-data wrapper that `amazon-init` execs — is -[`modules/amazon-init/`](./modules/amazon-init/README.md), so the two halves can move -independently: another cloud replaces the second without touching the first. +The template is two halves that do not know about each other: + +- [`modules/amazon-init/`](./modules/amazon-init/README.md) gets Coder onto an EC2 instance whose + AMI runs `amazon-init` instead of cloud-init. It publishes the agent handoff and the workspace + identity, streams logs before an agent exists, runs one script, and starts the agent last. + Nothing in it mentions Nix — the script it runs is an opaque string. +- [`modules/nix/`](./modules/nix/README.md) is the flake lifecycle: sync a checkout, decide whether + a rebuild is needed, apply it, filter the output. Nothing in it mentions EC2 or user-data. + +`scripts/boot.sh.tftpl` is the seam: it is handed to the first as `boot_script` and uses the +second. Either half can be lifted into a standalone registry module without untangling it from the +other. diff --git a/registry/coder/templates/aws-nixos/main.tf b/registry/coder/templates/aws-nixos/main.tf index c0409ea64..adcf16bb6 100644 --- a/registry/coder/templates/aws-nixos/main.tf +++ b/registry/coder/templates/aws-nixos/main.tf @@ -282,14 +282,15 @@ resource "coder_script" "nixos_rebuild" { log_path = "${local.log_dir}/coder-script.log" script = templatefile("${path.module}/scripts/rebuild.sh.tftpl", { - LOG_SH = local.log_sh + # The same library the boot path uses, taken from the module that owns it. + LOG_SH = module.amazon_init.log_library LIFECYCLE_SH = local.lifecycle_sh ARG_FLAKE_REF = local.flake_url ARG_FLAKE_BRANCH = var.flake_branch ARG_FLAKE_ATTR = local.flake_attr ARG_UPDATE_PROCESS = data.coder_parameter.update_process.value ARG_ACCESS_URL = data.coder_workspace.me.access_url - ARG_LOG_SOURCE_ID = local.log_source_id + ARG_LOG_SOURCE_ID = module.amazon_init.log_source_id }) } @@ -314,38 +315,42 @@ locals { # prefix and any query string. `flake_branch` carries the ref instead. flake_url = replace(replace(var.flake_ref, "/^git\\+/", ""), "/\\?.*$/", "") - # Constant, not uuid(): log sources are scoped to an agent, Coder treats a - # repeat POST with the same id as a no-op, and a generated value would churn - # the plan every run. - log_source_id = "6e1f4a2c-9b3d-4c8e-8a71-5f0d2b6c4e93" - - # Sourced verbatim into the two entrypoints below. Plain shell rather than - # templates so they stay readable and get covered by the repo's shellcheck. - log_sh = file("${path.module}/scripts/log.sh") + # Sourced verbatim into both entrypoints. Plain shell rather than a template + # so it stays readable and gets covered by the repo's shellcheck. lifecycle_sh = file("${path.module}/modules/nix/lifecycle.sh") + state_dir = "/var/lib/coder-nixos" + flake_dir = "/etc/nixos" + + # Handed to the amazon-init module, which runs it as root before starting + # the agent and otherwise does not look inside it. + boot_script = templatefile("${path.module}/scripts/boot.sh.tftpl", { + LIFECYCLE_SH = local.lifecycle_sh + ARG_FLAKE_REF = local.flake_url + ARG_FLAKE_BRANCH = var.flake_branch + ARG_FLAKE_ATTR = local.flake_attr + ARG_STATE_DIR = local.state_dir + ARG_LOG_DIR = local.log_dir + ARG_FLAKE_DIR = local.flake_dir + }) } -# Everything about getting Coder onto a NixOS AMI through amazon-init: the -# agent handoff, the workspace facts and the first rebuild, wrapped for EC2's -# 16 KiB user-data limit. See ./modules/amazon-init/README.md. +# Gets Coder onto the instance and runs one script on every boot. It knows +# nothing about Nix: `boot_script` is an opaque string to it, and the flake is +# applied entirely inside that string. See ./modules/amazon-init/README.md. module "amazon_init" { source = "./modules/amazon-init" - flake_ref = local.flake_url - flake_branch = var.flake_branch - flake_attr = local.flake_attr - access_url = data.coder_workspace.me.access_url agent_token = try(coder_agent.main[0].token, "") agent_init_script = try(coder_agent.main[0].init_script, "") - log_source_id = local.log_source_id - workspace_name = data.coder_workspace.me.name - hostname = lower(data.coder_workspace.me.name) - owner = data.coder_workspace_owner.me.name - owner_name = coalesce(data.coder_workspace_owner.me.full_name, data.coder_workspace_owner.me.name) - owner_email = data.coder_workspace_owner.me.email - log_library = local.log_sh - lifecycle_library = local.lifecycle_sh + boot_script = local.boot_script + + log_display_name = "NixOS" + log_icon = "/icon/nix.svg" + + # The AMI may not have curl before the first switch, and on NixOS the way to + # get one is to build it. + curl_resolve_command = "printf '%s' \"$(nix build --no-link --print-out-paths nixpkgs#curl)/bin/curl\"" } resource "aws_instance" "dev" { diff --git a/registry/coder/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder/templates/aws-nixos/modules/amazon-init/README.md index e0d124fd4..954c412c1 100644 --- a/registry/coder/templates/aws-nixos/modules/amazon-init/README.md +++ b/registry/coder/templates/aws-nixos/modules/amazon-init/README.md @@ -1,59 +1,128 @@ # amazon-init -The platform half of this template: everything that is true because the -workspace is a NixOS AMI on EC2, and nothing that is true because it is Nix -(that is [`../nix`](../nix/README.md)) or because it is Coder. +Coder on an EC2 AMI that runs `amazon-init` instead of cloud-init. -It renders one thing — EC2 user-data — and is the only part of the template -that knows how a NixOS instance is bootstrapped. +It renders user-data that publishes the agent handoff, publishes the +workspace's identity, runs **one script you supply**, and then starts the +agent. It does not know or care what that script does — `boot_script` is an +opaque string, and nothing in this module reads it. + +```tf +module "amazon_init" { + source = "./modules/amazon-init" + + agent_token = coder_agent.main.token + agent_init_script = coder_agent.main.init_script + boot_script = local.my_boot_script +} + +resource "aws_instance" "dev" { + user_data = module.amazon_init.user_data + + lifecycle { + precondition { + condition = module.amazon_init.user_data_bytes < 16384 + error_message = "Rendered user-data is ${module.amazon_init.user_data_bytes} bytes; EC2 allows at most 16384." + } + } +} +``` + +Workspace and owner identity are read from `coder_workspace` and +`coder_workspace_owner` here, so the caller passes neither. ## Why this is not cloud-init -The NixOS AMI does not run cloud-init. It runs `amazon-init.service`, which -reads `/etc/ec2-metadata/user-data` and execs it as a shell script when it -begins with `#!` — after `multi-user.target`, **on every boot**. There is no -`runcmd`, no `write_files`, no per-boot/once distinction, and no ordering -hooks. Three things follow, and they shape the whole script: +Some AMIs — the official NixOS images among them — do not ship cloud-init. +They run `amazon-init.service`, which reads `/etc/ec2-metadata/user-data` and +execs it as a shell script when it begins with `#!` — after +`multi-user.target`, **on every boot**. There is no `runcmd`, no +`write_files`, no per-boot/once distinction and no ordering hooks. Three +things follow: -- It must be idempotent, because it runs again on every restart. +- Everything must be idempotent, because it all runs again on every restart. - It is the only hook available, so the agent handoff, the workspace facts and - the rebuild all have to live in it. -- Nothing can be ordered `After` it. `nixos-rebuild switch` starts new units - synchronously and `amazon-init` cannot become active until its script exits, - so a unit that waits for it deadlocks the first boot. This is why the Coder - agent unit is started _by_ this script rather than wanted by a target. - -## What the script does - -1. Publishes the agent handoff to `/run/coder` — `agent.env` (0600), `init.sh`, - then `ready` last — before anything that can fail, so the agent can still be - started from a failure path. -2. Writes the per-workspace facts to `/run/coder/workspace.json`. -3. Syncs the flake checkout and rebuilds, streaming progress to the workspace - UI. -4. Starts `coder-agent.service`, and does so on every exit path, including a - failed rebuild. + whatever the image needs doing all have to live in it. +- Nothing can be ordered `After` it. A boot script that reconfigures the + machine may start units synchronously, and `amazon-init` cannot become + active until it exits — so a unit that waits for `amazon-init` deadlocks the + first boot. -## The user-data wrapper +## Order of operations -The boot script, its two shell libraries and the agent init script come to -roughly 19 KiB, against EC2's 16 KiB limit. So the user-data this module -outputs is a six-line self-extracting wrapper around a gzipped copy, which -lands at about 9 KiB. +1. **Handoff** — `agent.env` (0600), `init.sh`, then `ready`, into + `runtime_dir` (a tmpfs). Written before anything that can fail, so the + agent can still be started from a failure path. +2. **Hostname**, from `hostname` or the workspace name. +3. **Logging** — the shell library is written to `$${runtime_dir}/log.sh` and + sourced, and the log source is registered. +4. **Identity** — `$${runtime_dir}/workspace.json`, mode 0644, no secrets. +5. **Files** — anything in `files`. +6. **Your boot script**, as a child process. +7. **The agent**, always, whatever step 6 did. -That is transparent to the AMI: `amazon-init` only checks the first two bytes -for `#!` before exec'ing the blob. The wrapper extracts to a fixed path, -`/run/coder/bootstrap.sh`, so the real script is on disk when a boot needs -debugging. +## Starting the agent last is the point + +`coder-agent.service` must not be `wantedBy` anything; this module starts it, +and only once the boot script has finished. Left to systemd, the agent comes +up at `multi-user.target` — before the boot script has finished changing the +machine — reports the workspace ready, and runs its startup scripts against a +system that is about to be replaced under them. + +The mirror image of that rule is that a boot script failure must not leave an +unreachable workspace, so the agent is started even when the boot script +exits non-zero. Its status is reported and propagated, never suppressed. + +## What the boot script gets + +Root, a child process, and these: -`user_data_bytes` is exported unwrapped so the caller can assert the limit in a -`precondition` — the output itself is sensitive, because the agent token is -inside it, and Terraform suppresses error messages derived from sensitive -values. +| Variable | Meaning | +| ----------------------- | ------------------------------ | +| `CODER_LOG_LIBRARY` | path to source for `coder_log` | +| `CODER_RUNTIME_DIR` | the tmpfs this module owns | +| `CODER_WORKSPACE_FACTS` | path to `workspace.json` | +| `CODER_ACCESS_URL` | deployment URL | +| `CODER_AGENT_TOKEN` | agent token | +| `CODER_LOG_SOURCE_ID` | log source to write to | -## Contract +The log source is registered before the boot script starts, and `CODER_LOG_READY` +is exported, so sourcing the library is enough — `coder_log` works from the +first line and the boot script must not register the source again. + +## Logging before the agent exists + +The agent is the normal way to get output into the workspace UI, and on a +first boot it does not exist for as long as the boot script runs. So +`scripts/log.sh` calls the log API directly with the agent's own token: +`coder_log `, `coder_log_pipe ` for a stream, and +`coder_log_tail ` for the end of a transcript. + +It is exported as `log_library` for callers that want the same functions in a +`coder_script` later, and it is on the instance at `$${runtime_dir}/log.sh`. + +Two things it handles that are easy to get wrong: + +- **The 1 MiB cap.** Coder caps agent logs at 1 MiB per agent across every + source. Overflowing does not truncate — the agent is flagged overflowed and + every later log from every source is dropped permanently. The library tracks + its own usage and goes quiet at `log_budget_bytes`, which defaults to half + the cap. +- **No curl.** A minimal AMI may not have one. `curl_resolve_command` is a + command that prints a path to a binary, run once, only if `curl` is not + already on `PATH`. + +## The user-data wrapper + +EC2 caps user-data at 16 KiB, and the bootstrap script plus its payloads is +past that. So the output is a six-line self-extracting wrapper around a +gzipped copy. + +That is transparent to `amazon-init`: it only checks the first two bytes for +`#!` before exec'ing the blob. The wrapper extracts to +`$${runtime_dir}/bootstrap.sh`, so the real script is on disk when a boot needs +debugging. -Inputs are the workspace's identity and the flake to build; the two shell -libraries are passed in as strings rather than read here, so the template's -other entrypoints share one copy. The token is written to a tmpfs at 0600 and -never passed into Nix. +`user_data_bytes` is exported unwrapped so a caller can assert the limit — the +`user_data` output itself is sensitive, because the token is inside it, and +Terraform suppresses messages derived from sensitive values. diff --git a/registry/coder/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder/templates/aws-nixos/modules/amazon-init/main.tf index b4534cfd1..16a47f214 100644 --- a/registry/coder/templates/aws-nixos/modules/amazon-init/main.tf +++ b/registry/coder/templates/aws-nixos/modules/amazon-init/main.tf @@ -1,128 +1,196 @@ -# Everything that has to happen because this is a NixOS AMI on EC2, and -# nothing that has to happen because it is Nix or because it is Coder. +# Coder on an AMI that runs amazon-init instead of cloud-init. # -# The NixOS AMI does not run cloud-init. It runs amazon-init.service, which -# reads /etc/ec2-metadata/user-data and execs it as a shell script when it -# begins with `#!` -- after multi-user.target, on every boot. That is the only -# hook this platform offers, so it has to carry the agent handoff, the -# workspace facts and the rebuild, and it has to be idempotent. +# Renders EC2 user-data that publishes the agent handoff, publishes the +# workspace's identity, runs one boot script supplied by the caller, and then +# starts the agent. What that boot script does is none of this module's +# business: it is a string, and the module never looks inside it. terraform { required_version = ">= 1.0" -} -variable "flake_ref" { - description = "Git remote to clone into the checkout, as `git clone` understands it." - type = string -} - -variable "flake_branch" { - description = "Branch to track in that remote." - type = string + required_providers { + coder = { + source = "coder/coder" + version = ">= 2.5" + } + } } -variable "flake_attr" { - description = "`nixosConfigurations` attribute to build, architecture already resolved." - type = string -} +data "coder_workspace" "me" {} -variable "access_url" { - description = "Deployment access URL the agent and the log API are reached on." - type = string -} +data "coder_workspace_owner" "me" {} variable "agent_token" { - description = "Agent token. Written to /run/coder/agent.env at 0600 on a tmpfs and never passed into Nix." + description = "Agent token. Written to `agent.env` at 0600 on a tmpfs." type = string sensitive = true } variable "agent_init_script" { - description = "`coder_agent.init_script`, run verbatim once the rebuild has finished." + description = "`coder_agent.init_script`, run verbatim once the boot script has finished." type = string sensitive = true } -variable "log_source_id" { - description = "Log source the rebuild streams to. Must be stable across builds." +variable "boot_script" { + description = <<-EOT + Shell script run on every boot, after the handoff and workspace facts are + published and before the agent is started. + + It runs as root, as a child process, with these set: + + | Variable | Meaning | + | ------------------------ | ------------------------------------------ | + | `CODER_LOG_LIBRARY` | path to source for `coder_log` | + | `CODER_RUNTIME_DIR` | the tmpfs this module owns | + | `CODER_WORKSPACE_FACTS` | path to `workspace.json` | + | `CODER_ACCESS_URL` | deployment URL | + | `CODER_AGENT_TOKEN` | agent token | + | `CODER_LOG_SOURCE_ID` | log source to write to | + + Its exit status is reported and propagated, but never suppresses the agent + start: a workspace whose boot script failed still has to be reachable. + EOT type = string + default = "" } -variable "workspace_name" { - description = "Workspace name, published in /run/coder/workspace.json." +variable "files" { + description = <<-EOT + Extra files to write before the boot script runs, keyed by absolute path. + Content is carried base64-encoded, so any bytes are safe. + EOT + type = map(object({ + content = string + mode = optional(string, "0644") + })) + default = {} + + validation { + condition = alltrue([for path in keys(var.files) : startswith(path, "/")]) + error_message = "File paths must be absolute." + } +} + +variable "runtime_dir" { + description = "Directory for the agent handoff and this module's own state. Must be on a tmpfs: it holds the token." type = string + default = "/run/coder" } -variable "hostname" { - description = "Hostname to set on the instance." +variable "path" { + description = "Prepended to `PATH` for the boot script. amazon-init's own PATH is short." + type = string + default = "/run/current-system/sw/bin" +} + +variable "curl_resolve_command" { + description = <<-EOT + Shell command that prints a path to a `curl` binary, used only when the + image has none. Logging is the only thing that needs it, and it is the one + thing this module cannot work out for an image it does not know. + EOT type = string + default = "" } -variable "owner" { - description = "Workspace owner's username." +variable "log_source_id" { + description = "UUID of the log source boot output is streamed to. Must be constant across builds; Coder treats a repeat POST of the same id as a no-op." type = string + default = "6e1f4a2c-9b3d-4c8e-8a71-5f0d2b6c4e93" + + validation { + condition = can(regex("^[0-9a-f-]{36}$", var.log_source_id)) + error_message = "log_source_id must be a UUID." + } } -variable "owner_name" { - description = "Workspace owner's full name." +variable "log_display_name" { + description = "Name of that log source in the workspace UI." type = string + default = "Boot" } -variable "owner_email" { - description = "Workspace owner's email address." +variable "log_icon" { + description = "Icon for that log source." type = string + default = "/icon/widgets.svg" } -variable "log_library" { +variable "log_budget_bytes" { description = <<-EOT - Contents of the shell library providing `coder_log`, sourced into the boot - script. Passed in rather than read here so the same copy is shared with - the template's other entrypoints. + How many bytes of log this module will push before going quiet. + + Coder caps agent logs at 1 MiB per agent across every source, and + overflowing does not truncate: the agent is flagged overflowed and all + later logs are dropped permanently. The default leaves half the cap for + everything else. EOT - type = string + type = number + default = 524288 } -variable "lifecycle_library" { - description = "Contents of the Nix lifecycle shell library, sourced into the boot script." +variable "hostname" { + description = "Hostname to set on the instance. Defaults to the workspace name." type = string + default = "" } locals { + hostname = var.hostname != "" ? var.hostname : lower(data.coder_workspace.me.name) + + # Written by the bootstrap script before the boot script runs. Content is + # base64 so that a heredoc in the file cannot terminate the one writing it. + files_sh = join("\n", [ + for path, file in var.files : <<-SH + install -d -m 0755 "$(dirname ${path})" + printf '%s' '${base64encode(file.content)}' | base64 -d >'${path}' + chmod ${file.mode} '${path}' + SH + ]) + bootstrap = templatefile("${path.module}/scripts/bootstrap.sh.tftpl", { - LOG_SH = var.log_library - LIFECYCLE_SH = var.lifecycle_library - ARG_FLAKE_REF = var.flake_ref - ARG_FLAKE_BRANCH = var.flake_branch - ARG_FLAKE_ATTR = var.flake_attr - ARG_ACCESS_URL = var.access_url + LOG_SH = file("${path.module}/scripts/log.sh") + FILES_SH = local.files_sh + BOOT_SCRIPT = var.boot_script + + ARG_ACCESS_URL = data.coder_workspace.me.access_url ARG_AGENT_TOKEN = var.agent_token ARG_INIT_SCRIPT_B64 = base64encode(var.agent_init_script) - ARG_LOG_SOURCE_ID = var.log_source_id - ARG_HOSTNAME = var.hostname - ARG_WORKSPACE_NAME = var.workspace_name - ARG_OWNER = var.owner + ARG_RUNTIME_DIR = var.runtime_dir + ARG_PATH = var.path + + ARG_LOG_SOURCE_ID = var.log_source_id + ARG_LOG_DISPLAY_NAME_B64 = base64encode(var.log_display_name) + ARG_LOG_ICON = var.log_icon + ARG_LOG_BUDGET = var.log_budget_bytes + ARG_CURL_RESOLVE_B64 = base64encode(var.curl_resolve_command) + + ARG_HOSTNAME = local.hostname + ARG_WORKSPACE_NAME = data.coder_workspace.me.name + ARG_OWNER = data.coder_workspace_owner.me.name # base64 because a full name may contain quotes and is interpolated into - # a shell string. - ARG_OWNER_NAME_B64 = base64encode(var.owner_name) - ARG_OWNER_EMAIL = var.owner_email + # both a shell string and a JSON document. + ARG_OWNER_NAME_B64 = base64encode(coalesce(data.coder_workspace_owner.me.full_name, data.coder_workspace_owner.me.name)) + ARG_OWNER_EMAIL = data.coder_workspace_owner.me.email }) - # EC2 caps user-data at 16 KiB, and the boot script plus its two libraries - # plus the agent init script come to roughly 19 KiB. So user-data is a - # six-line self-extracting wrapper around a compressed copy. + # EC2 caps user-data at 16 KiB and the script above plus its payloads is + # comfortably past that, so user-data is a six-line self-extracting wrapper + # around a compressed copy. # - # This is transparent to the NixOS AMI: amazon-init only inspects the first - # two bytes for `#!` before exec'ing the blob, and it has no decompression - # step of its own. Extracting to a fixed path also means the real script is - # on disk when something needs debugging. + # This is transparent to amazon-init: it only inspects the first two bytes + # for `#!` before exec'ing the blob, and it has no decompression step of its + # own. Extracting to a fixed path also means the real script is on disk when + # something needs debugging. user_data = <<-SH #!/usr/bin/env bash set -eu - install -d -m 0700 /run/coder - base64 -d <<'CODER_PAYLOAD' | gzip -dc >/run/coder/bootstrap.sh + install -d -m 0700 ${var.runtime_dir} + base64 -d <<'CODER_PAYLOAD' | gzip -dc >${var.runtime_dir}/bootstrap.sh ${base64gzip(local.bootstrap)} CODER_PAYLOAD - exec bash /run/coder/bootstrap.sh + exec bash ${var.runtime_dir}/bootstrap.sh SH } @@ -133,11 +201,31 @@ output "user_data" { } output "user_data_bytes" { - description = "Size of the rendered user-data, for the caller's 16 KiB precondition." + description = "Size of the rendered user-data, for the caller's 16 KiB precondition. Unwrapped so the number can be shown in an error message." value = nonsensitive(length(local.user_data)) } +output "log_library" { + description = "The shell logging library, for callers that want `coder_log` in their own `coder_script`s. On the instance it is also at `$${runtime_dir}/log.sh`." + value = file("${path.module}/scripts/log.sh") +} + +output "log_source_id" { + description = "Log source the boot output is streamed to." + value = var.log_source_id +} + +output "runtime_dir" { + description = "Directory holding the agent handoff, the logging library and the workspace facts." + value = var.runtime_dir +} + +output "workspace_facts_path" { + description = "Path to the workspace identity file written on every boot." + value = "${var.runtime_dir}/workspace.json" +} + output "bootstrap_path" { - description = "Where the wrapper extracts the real boot script on the instance." - value = "/run/coder/bootstrap.sh" + description = "Where the user-data wrapper extracts the real boot script." + value = "${var.runtime_dir}/bootstrap.sh" } diff --git a/registry/coder/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl b/registry/coder/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl index d0897bea2..7c2e256d2 100644 --- a/registry/coder/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl +++ b/registry/coder/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl @@ -1,44 +1,42 @@ #!/usr/bin/env bash -# EC2 user-data. The NixOS AMI does not run cloud-init; it runs -# amazon-init.service, which execs this as a shell script when it starts with +# EC2 user-data for an AMI that runs amazon-init.service rather than +# cloud-init. amazon-init execs this as a shell script when it starts with # `#!` -- after multi-user.target, on every boot. So it must be idempotent. +# +# It knows nothing about what the instance is for. It publishes the agent +# handoff, publishes the workspace's identity, runs one caller-supplied boot +# script, and then starts the agent -- in that order, always. set -euo pipefail -export PATH="/run/current-system/sw/bin:$PATH" +export PATH="${ARG_PATH}:$PATH" export HOME=/root -FLAKE_REF='${ARG_FLAKE_REF}' -FLAKE_BRANCH='${ARG_FLAKE_BRANCH}' -FLAKE_ATTR='${ARG_FLAKE_ATTR}' ACCESS_URL='${ARG_ACCESS_URL}' AGENT_TOKEN='${ARG_AGENT_TOKEN}' INIT_SCRIPT_B64='${ARG_INIT_SCRIPT_B64}' LOG_SOURCE_ID='${ARG_LOG_SOURCE_ID}' +LOG_DISPLAY_NAME_B64='${ARG_LOG_DISPLAY_NAME_B64}' +LOG_ICON='${ARG_LOG_ICON}' +LOG_BUDGET='${ARG_LOG_BUDGET}' HOSTNAME_='${ARG_HOSTNAME}' WORKSPACE_NAME='${ARG_WORKSPACE_NAME}' OWNER='${ARG_OWNER}' OWNER_NAME_B64='${ARG_OWNER_NAME_B64}' OWNER_EMAIL='${ARG_OWNER_EMAIL}' +CURL_RESOLVE_B64='${ARG_CURL_RESOLVE_B64}' -RUNTIME_DIR=/run/coder -STATE_DIR=/var/lib/coder-nixos -LOG_DIR=/var/log/coder-nixos -FLAKE_DIR=/etc/nixos +RUNTIME_DIR='${ARG_RUNTIME_DIR}' # --- 1. Hand the agent its token ------------------------------------------- # -# Done first, before anything that can fail, so that the agent can be started -# from any later exit path -- including a failed rebuild, where a reachable -# workspace is the only way to debug the flake that broke it. +# Done first, before anything that can fail, so that the agent can still be +# started from a later failure path -- an instance whose boot script broke is +# exactly the instance someone needs a terminal on. # -# The token stays out of Nix on purpose. As a flake input it would be baked -# into a derivation, land world-readable in /nix/store and persist across -# generations and past rotation. RUNTIME_DIR is a tmpfs, so nothing here -# survives a reboot -- which is the point, since the token rotates on every -# workspace start. +# RUNTIME_DIR is a tmpfs, so nothing written here survives a reboot. That is +# the point: the token rotates on every workspace start. install -d -m 0700 "$RUNTIME_DIR" -install -d -m 0755 "$STATE_DIR" "$LOG_DIR" rm -f "$RUNTIME_DIR/ready" umask 077 @@ -48,15 +46,13 @@ umask 022 printf '%s' "$INIT_SCRIPT_B64" | base64 -d >"$RUNTIME_DIR/init.sh" chmod 0755 "$RUNTIME_DIR/init.sh" -# The unit runs as the workspace user, so it has to be able to traverse the -# directory and read the token. Mode stays 0600/0700; only the owner changes. -# -# The owner is whoever the flake says the agent runs as -- asked of the unit -# rather than guessed by the template, so a configuration we do not control -# cannot disagree with us. +# coder-agent.service may run as an unprivileged user, so it has to be able to +# traverse the directory and read the token. Modes stay 0600/0700; only the +# owner changes. # -# On the very first boot the unit does not exist yet, so this is a no-op and -# is retried after the switch, before the agent is started. +# The owner is asked of the unit rather than guessed, so an image we do not +# control cannot disagree with us. On the very first boot the unit does not +# exist yet, so this is a no-op and is retried before the agent is started. chown_handoff() { local target target=$(systemctl show coder-agent -p User --value 2>/dev/null || true) @@ -70,36 +66,47 @@ chown_handoff [ -z "$HOSTNAME_" ] || hostnamectl set-hostname "$HOSTNAME_" || true # --- 2. Logging ------------------------------------------------------------- - -CODER_ACCESS_URL="$ACCESS_URL" -CODER_AGENT_TOKEN="$AGENT_TOKEN" -CODER_LOG_SOURCE_ID="$LOG_SOURCE_ID" -CODER_LOG_STATE_DIR="$RUNTIME_DIR" -# Half of Coder's 1 MiB per-agent cap, leaving room for coder_script output. -# Overflowing the cap is permanent and silently drops all later logs. -CODER_LOG_BUDGET=524288 - +# +# The agent is the usual way to get output into the workspace UI, and it does +# not exist yet -- on a first boot it will not exist for as long as the boot +# script runs. So the log API is called directly, with the agent's own token. + +export CODER_ACCESS_URL="$ACCESS_URL" +export CODER_AGENT_TOKEN="$AGENT_TOKEN" +export CODER_LOG_SOURCE_ID="$LOG_SOURCE_ID" +export CODER_LOG_STATE_DIR="$RUNTIME_DIR" +export CODER_LOG_BUDGET="$LOG_BUDGET" +export CODER_CURL_RESOLVE +CODER_CURL_RESOLVE=$(printf '%s' "$CURL_RESOLVE_B64" | base64 -d) + +# Written to disk as well as sourced, so the boot script and anything the +# agent runs later can use the same functions without a second copy. +# +# Interpolated as text rather than base64: user-data is gzipped and has 16 KiB +# to fit in, and base64 both inflates by a third and destroys the redundancy +# gzip lives on. The delimiters are long enough not to occur in the payload. +cat >"$RUNTIME_DIR/log.sh" <<'CODER_AMAZON_INIT_LOG_LIBRARY' ${LOG_SH} - -TRANSCRIPT="$LOG_DIR/rebuild-$(date -u +%Y%m%dT%H%M%SZ).log" -coder_log_init "NixOS" "/icon/nix.svg" || true - -nix_log() { coder_log "$@" || true; } - -# The agent is not `wantedBy = multi-user.target` in the flake; nothing starts -# it but this. That is what keeps the workspace from being handed to the user -# on the generation we are about to replace -- startup scripts would install -# into a system that is seconds from being swapped out. +CODER_AMAZON_INIT_LOG_LIBRARY +chmod 0644 "$RUNTIME_DIR/log.sh" +export CODER_LOG_LIBRARY="$RUNTIME_DIR/log.sh" +# shellcheck source=/dev/null +. "$CODER_LOG_LIBRARY" + +coder_log_init "$(printf '%s' "$LOG_DISPLAY_NAME_B64" | base64 -d)" "$LOG_ICON" || true + +# Nothing starts coder-agent.service but this, and only once the boot script +# has finished. If systemd started it at multi-user.target instead, the agent +# would report the workspace ready and run its startup scripts while the boot +# script was still changing the machine underneath them. # -# So every exit path from here on has to go through this, including the -# failure ones: a workspace whose flake does not build is exactly the -# workspace someone needs a terminal on. +# So every exit path from here on has to go through this. start_agent() { if systemctl is-active --quiet coder-agent; then return 0 fi if ! systemctl cat coder-agent >/dev/null 2>&1; then - coder_log error "coder-agent.service does not exist: the flake has never been applied." || true + coder_log error "coder-agent.service does not exist; the boot script has never created it." || true return 1 fi @@ -121,8 +128,7 @@ start_agent() { on_error() { local rc=$? - coder_log error "Boot script failed (exit $rc). Transcript: $TRANSCRIPT" || true - coder_log_tail "$TRANSCRIPT" 200 || true + coder_log error "Boot failed before the boot script ran (exit $rc)." || true start_agent || true exit "$rc" } @@ -130,12 +136,8 @@ trap on_error ERR # --- 3. Facts about this workspace ----------------------------------------- # -# A runtime interface, not an evaluation input: a flake cannot read an -# absolute path outside itself in pure evaluation mode, so consuming this at -# eval time would need --impure and would stop `nixos-rebuild switch` from -# reproducing what we apply. Configurations read it from a service instead. -# -# Identity only. The token lives in agent.env at 0600 and never appears here. +# A runtime interface: identity only, mode 0644, no secrets. The token lives +# in agent.env at 0600 and never appears here. cat >"$RUNTIME_DIR/workspace.json" <"$RUNTIME_DIR/workspace.json" <"$RUNTIME_DIR/boot.sh" <<'CODER_AMAZON_INIT_BOOT_SCRIPT' +${BOOT_SCRIPT} +CODER_AMAZON_INIT_BOOT_SCRIPT +chmod 0700 "$RUNTIME_DIR/boot.sh" -nix_lock -nix_sync_checkout "$FLAKE_REF" "$FLAKE_BRANCH" +# Run as a child rather than sourced: the boot script is someone else's code, +# and it must not be able to exit this one before the agent is started. +set +e +bash "$RUNTIME_DIR/boot.sh" +boot_rc=$? +set -e -REV=$(nix_needs_rebuild) && NEEDS_REBUILD=1 || NEEDS_REBUILD=0 -if [ "$NEEDS_REBUILD" -eq 0 ]; then - coder_log info "Configuration unchanged ($${REV:0:12}); skipping rebuild." || true - start_agent - exit 0 +if [ "$boot_rc" -ne 0 ]; then + coder_log error "Boot script exited $boot_rc; starting the agent anyway." || true fi -coder_log info "Applying $FLAKE_DIR#$FLAKE_ATTR ($${REV:0:12})" || true -coder_log info "Transcript: $TRANSCRIPT" || true - -if ! nix_apply switch "$TRANSCRIPT"; then - coder_log error "nixos-rebuild switch failed; the previous generation is still active." || true - coder_log_tail "$TRANSCRIPT" 200 || true - # Only reachable on a boot that already has a generation of its own, since - # the first boot has no agent unit to fall back to. - start_agent || true - exit 1 -fi - -nix_record_rev "$REV" -# The workspace user exists now, so systemd-tmpfiles has set the flake -# directory's owner; make the tree match it. -nix_own_checkout - -# Activation has finished, so the agent's startup scripts see a settled -# system: the final generation's /etc, its PATH, its interpreters. -coder_log info "Switch complete; starting the agent." || true start_agent +exit "$boot_rc" diff --git a/registry/coder/templates/aws-nixos/scripts/log.sh b/registry/coder/templates/aws-nixos/modules/amazon-init/scripts/log.sh similarity index 82% rename from registry/coder/templates/aws-nixos/scripts/log.sh rename to registry/coder/templates/aws-nixos/modules/amazon-init/scripts/log.sh index 314bd9e73..bd137b2a2 100644 --- a/registry/coder/templates/aws-nixos/scripts/log.sh +++ b/registry/coder/templates/aws-nixos/modules/amazon-init/scripts/log.sh @@ -1,8 +1,8 @@ # shellcheck shell=bash # # Pushes lines to the Coder agent log API -- the only way to show anything in -# the workspace UI before the agent exists, which on first boot is the whole -# duration of the NixOS switch. +# the workspace UI before the agent exists, which on a first boot lasts for as +# long as the boot script runs. # # Callers set CODER_ACCESS_URL, CODER_AGENT_TOKEN, CODER_LOG_SOURCE_ID and # optionally CODER_LOG_BUDGET. Sets no shell options: failing to log must @@ -11,18 +11,23 @@ CODER_LOG_BUDGET="${CODER_LOG_BUDGET:-524288}" CODER_LOG_STATE_DIR="${CODER_LOG_STATE_DIR:-/run/coder}" CODER_LOG_MAX_LINE=2048 -CODER_LOG_READY=0 +# Inherited, so a child process that sources this library can log without +# registering the source again -- registration is per workspace build, not per +# process, and a child has no way to know whether it already happened. +CODER_LOG_READY="${CODER_LOG_READY:-0}" -# amazon-init's PATH has no curl. Resolve once; `nix shell` per call would add -# an evaluation to every batch. +# A minimal AMI may not ship curl at all, and amazon-init's PATH is short. +# When it is missing, CODER_CURL_RESOLVE -- a command supplied by whoever +# instantiated this module, because only they know how to obtain a binary on +# their image -- is run once and expected to print a path to one. coder_curl() { if [ -z "${CODER_CURL:-}" ]; then if command -v curl > /dev/null 2>&1; then CODER_CURL=$(command -v curl) - else - CODER_CURL="$(nix build --no-link --print-out-paths nixpkgs#curl 2> /dev/null)/bin/curl" + elif [ -n "${CODER_CURL_RESOLVE:-}" ]; then + CODER_CURL=$(eval "$CODER_CURL_RESOLVE" 2> /dev/null || true) fi - [ -x "$CODER_CURL" ] || return 1 + [ -n "${CODER_CURL:-}" ] && [ -x "$CODER_CURL" ] || return 1 export CODER_CURL fi "$CODER_CURL" "$@" @@ -64,7 +69,7 @@ JSON case "$code" in # The handler returns 201 on both create and already-exists. 200 | 201) - CODER_LOG_READY=1 + export CODER_LOG_READY=1 return 0 ;; 401 | 403 | 000 | 5??) diff --git a/registry/coder/templates/aws-nixos/scripts/boot.sh.tftpl b/registry/coder/templates/aws-nixos/scripts/boot.sh.tftpl new file mode 100644 index 000000000..e495ff3b6 --- /dev/null +++ b/registry/coder/templates/aws-nixos/scripts/boot.sh.tftpl @@ -0,0 +1,68 @@ +#!/usr/bin/env bash +# Applies the flake on every boot. Handed to the amazon-init module as an +# opaque string and run by it as root, after the agent handoff exists and +# before the agent is started -- so everything here happens while the +# workspace is still building, and the agent never sees a half-applied system. +# +# Runs on every boot, so it must be idempotent. + +set -euo pipefail + +FLAKE_REF='${ARG_FLAKE_REF}' +FLAKE_BRANCH='${ARG_FLAKE_BRANCH}' +FLAKE_ATTR='${ARG_FLAKE_ATTR}' + +STATE_DIR='${ARG_STATE_DIR}' +LOG_DIR='${ARG_LOG_DIR}' +FLAKE_DIR='${ARG_FLAKE_DIR}' + +install -d -m 0755 "$STATE_DIR" "$LOG_DIR" + +# Provided by the amazon-init module, along with the CODER_* variables it +# needs. Logging must never be fatal, hence the fallbacks. +# shellcheck source=/dev/null +. "$${CODER_LOG_LIBRARY:?the amazon-init module sets this}" + +nix_log() { coder_log "$@" || true; } + +NIX_FLAKE_DIR="$FLAKE_DIR" +NIX_FLAKE_ATTR="$FLAKE_ATTR" +NIX_STATE_DIR="$STATE_DIR" +NIX_LOG_DIR="$LOG_DIR" + +${LIFECYCLE_SH} + +TRANSCRIPT="$LOG_DIR/rebuild-$(date -u +%Y%m%dT%H%M%SZ).log" + +on_error() { + local rc=$? + coder_log error "Boot script failed (exit $rc). Transcript: $TRANSCRIPT" || true + coder_log_tail "$TRANSCRIPT" 200 || true + exit "$rc" +} +trap on_error ERR + +nix_lock +nix_sync_checkout "$FLAKE_REF" "$FLAKE_BRANCH" + +REV=$(nix_needs_rebuild) && NEEDS_REBUILD=1 || NEEDS_REBUILD=0 +if [ "$NEEDS_REBUILD" -eq 0 ]; then + coder_log info "Configuration unchanged ($${REV:0:12}); skipping rebuild." || true + exit 0 +fi + +coder_log info "Applying $FLAKE_DIR#$FLAKE_ATTR ($${REV:0:12})" || true +coder_log info "Transcript: $TRANSCRIPT" || true + +if ! nix_apply switch "$TRANSCRIPT"; then + coder_log error "nixos-rebuild switch failed; the previous generation is still active." || true + coder_log_tail "$TRANSCRIPT" 200 || true + exit 1 +fi + +nix_record_rev "$REV" +# The workspace user exists now, so systemd-tmpfiles has set the flake +# directory's owner; make the tree match it. +nix_own_checkout + +coder_log info "Switch complete." || true From b47eaec75fbcca05cc58b92b48f9d1f015f31d39 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 21 Sep 2026 16:18:07 +0000 Subject: [PATCH 05/24] refactor(aws-nixos): split the template into a nix module and a platform module `modules/nix/` was a shell library with a README calling itself a module. It is now an actual Terraform module that owns the whole flake lifecycle: both entrypoints (`boot.sh.tftpl`, `rebuild.sh.tftpl`), the `coder_script` that runs the periodic rebuild, the `$ARCH` substitution, the reference parsing and the three directory paths -- which the two scripts had been declaring separately, in one case by hardcoding them again. It exposes `boot_script`, `flake_uri`, `flake_attr`, `flake_dir`, `log_dir` and `version_command`, the last of these because agent metadata must be declared inline on the agent but deciding what "up to date" means for a NixOS machine should not be a template's job. Nothing in it mentions EC2 or user-data; `main.tf` keeps only the AMI, the instance, the agent and the IDE modules. `flake_branch` is gone. One reference carries the branch the way nix writes it -- `git+https://host/org/repo?ref=dev` -- and an absent `?ref=` means the remote's default branch, resolved on the instance because Terraform cannot know it without talking to the remote. Two failure-hiding bugs in `nix_sync_checkout` went with it: the clone and fetch tested the log filter's exit status rather than git's, so a failed clone read as success, and asking for a branch the checkout is not on reported "local commits" while building something else entirely. amazon-init: - `files` is now a plain path-to-text map, carried gzipped. - The logging library is fully internal; there is no `log_library` output, since the rebuild script never used one -- it set `CODER_LOG_*`, embedded 4.5 KiB of shell and then only ever called `echo`. - The log source id is a `random_uuid` in state instead of a constant. - The 16 KiB user-data limit is asserted on the `user_data` output, so the plan fails in the module that decides what goes into user-data rather than in the caller. Verified on real workspaces: `?ref=main`, a bare URL resolving the default branch, a restart fast-forwarding c2a4363 -> 1a08e3d and rebuilding, and the periodic script run by hand reporting "Already up to date". Finally, the template moves to the coder-labs namespace. --- .../templates/aws-nixos/PREREQUISITES.md | 0 .../templates/aws-nixos/README.md | 23 ++- .../templates/aws-nixos/main.tf | 117 +++-------- .../aws-nixos/modules/amazon-init/README.md | 29 +-- .../aws-nixos/modules/amazon-init/main.tf | 75 +++---- .../amazon-init/scripts/bootstrap.sh.tftpl | 0 .../modules/amazon-init/scripts/log.sh | 0 .../templates/aws-nixos/modules/nix/README.md | 85 ++++++++ .../templates/aws-nixos/modules/nix/main.tf | 189 ++++++++++++++++++ .../modules/nix}/scripts/boot.sh.tftpl | 36 +++- .../modules/nix/scripts}/lifecycle.sh | 43 +++- .../modules/nix}/scripts/rebuild.sh.tftpl | 24 +-- .../templates/aws-nixos/modules/nix/README.md | 62 ------ 13 files changed, 444 insertions(+), 239 deletions(-) rename registry/{coder => coder-labs}/templates/aws-nixos/PREREQUISITES.md (100%) rename registry/{coder => coder-labs}/templates/aws-nixos/README.md (93%) rename registry/{coder => coder-labs}/templates/aws-nixos/main.tf (71%) rename registry/{coder => coder-labs}/templates/aws-nixos/modules/amazon-init/README.md (83%) rename registry/{coder => coder-labs}/templates/aws-nixos/modules/amazon-init/main.tf (77%) rename registry/{coder => coder-labs}/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl (100%) rename registry/{coder => coder-labs}/templates/aws-nixos/modules/amazon-init/scripts/log.sh (100%) create mode 100644 registry/coder-labs/templates/aws-nixos/modules/nix/README.md create mode 100644 registry/coder-labs/templates/aws-nixos/modules/nix/main.tf rename registry/{coder/templates/aws-nixos => coder-labs/templates/aws-nixos/modules/nix}/scripts/boot.sh.tftpl (63%) rename registry/{coder/templates/aws-nixos/modules/nix => coder-labs/templates/aws-nixos/modules/nix/scripts}/lifecycle.sh (83%) rename registry/{coder/templates/aws-nixos => coder-labs/templates/aws-nixos/modules/nix}/scripts/rebuild.sh.tftpl (76%) delete mode 100644 registry/coder/templates/aws-nixos/modules/nix/README.md diff --git a/registry/coder/templates/aws-nixos/PREREQUISITES.md b/registry/coder-labs/templates/aws-nixos/PREREQUISITES.md similarity index 100% rename from registry/coder/templates/aws-nixos/PREREQUISITES.md rename to registry/coder-labs/templates/aws-nixos/PREREQUISITES.md diff --git a/registry/coder/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md similarity index 93% rename from registry/coder/templates/aws-nixos/README.md rename to registry/coder-labs/templates/aws-nixos/README.md index 124e43a0a..53900d170 100644 --- a/registry/coder/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -63,15 +63,17 @@ build is exactly the one you need a terminal on. ## Choosing which configuration is applied -Two variables control this: +Two Terraform variables control this: `flake_ref`, the Git reference to the configuration, and +`flake_attr`, the `nixosConfigurations` attribute to apply. -| Variable | Default | Meaning | -| ------------ | ----------------------------------------------------------- | ---------------------------------------------- | -| `flake_ref` | `git+https://github.com/coder/nixos-example-flake?ref=main` | where the flake lives | -| `flake_attr` | `workspace-$ARCH` | which `nixosConfigurations` attribute to apply | +`flake_ref` takes the reference in the form `nix` itself accepts — a `git+` prefix is optional and +`?ref=` selects a branch, so `https://host/org/repo`, +`git+https://host/org/repo?ref=dev` and `git+ssh://git@host/org/repo?ref=dev` all work. Without +`?ref=` the remote's default branch is used, resolved on the instance. The configuration must be +committed: a Git flake reference only ever sees committed files. -`flake_attr` is the part after `#` in a flake reference, so this is exactly the selection you would -make by hand: +`flake_attr` is the part after `#` in a flake reference, so between them this is exactly the +selection you would make by hand: ```console nixos-rebuild switch --flake 'github:your-org/config#workspace-x86_64' @@ -325,6 +327,7 @@ The template is two halves that do not know about each other: - [`modules/nix/`](./modules/nix/README.md) is the flake lifecycle: sync a checkout, decide whether a rebuild is needed, apply it, filter the output. Nothing in it mentions EC2 or user-data. -`scripts/boot.sh.tftpl` is the seam: it is handed to the first as `boot_script` and uses the -second. Either half can be lifted into a standalone registry module without untangling it from the -other. +The seam is one string: the nix module renders a boot script and exports it, and the amazon-init +module runs it without looking inside. Either half can be lifted into a standalone registry module +without untangling it from the other, and `main.tf` is left with the things that are genuinely +about this template — the AMI, the instance, the agent and the IDE modules. diff --git a/registry/coder/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf similarity index 71% rename from registry/coder/templates/aws-nixos/main.tf rename to registry/coder-labs/templates/aws-nixos/main.tf index adcf16bb6..e9773eebd 100644 --- a/registry/coder/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -22,28 +22,19 @@ provider "aws" { variable "flake_ref" { description = <<-EOT - Git remote holding the NixOS configuration. Cloned to /etc/nixos on the - workspace, which is what makes a bare `sudo nixos-rebuild switch` work. + Git reference to the NixOS configuration, in the form `nix` itself + accepts: `https://host/org/repo`, optionally with a `git+` prefix and a + `?ref=` branch. Without `?ref=` the remote's default branch is used. - Anything `git clone` accepts, so private repositories need git - credentials on the instance rather than Nix's netrc. + The configuration must be committed -- a Git flake reference only ever + sees committed files. EOT type = string default = "https://github.com/coder/nixos-example-flake" } -variable "flake_branch" { - description = "Branch to track in flake_ref." - type = string - default = "main" -} - variable "flake_attr" { - description = <<-EOT - Which `nixosConfigurations` attribute to apply, i.e. the part after `#` - in the flake reference. `$ARCH` is replaced with `x86_64` or `aarch64` to - match the chosen instance type. - EOT + description = "`nixosConfigurations` attribute to build. `$ARCH` is replaced with `x86_64` or `aarch64` to match the instance type." type = string default = "workspace-$ARCH" } @@ -204,23 +195,15 @@ resource "coder_agent" "main" { script = "coder stat disk --path $HOME" } # Makes `update_process = boot` visible; a staged generation is otherwise - # invisible and looks like updates being ignored. + # invisible and looks like updates being ignored. The command comes from the + # nix module -- metadata has to be declared on the agent, but what it means + # to be up to date is not this file's business. metadata { key = "nixos" display_name = "NixOS version" interval = 60 timeout = 10 - # /run/current-system is the activated system; /run/booted-system is what - # the kernel booted and still points at the previous generation after a - # switch, which would mark every new workspace as needing a restart. - script = <<-EOT - version=$(nixos-version 2>/dev/null || echo unknown) - if [ "$(readlink -f /run/current-system)" = "$(readlink -f /nix/var/nix/profiles/system)" ]; then - echo "$version" - else - echo "$version (restart to apply update)" - fi - EOT + script = module.nix.version_command } } @@ -270,30 +253,6 @@ module "git-config" { agent_id = coder_agent.main[0].id } -resource "coder_script" "nixos_rebuild" { - count = var.update_schedule == "" ? 0 : data.coder_workspace.me.start_count - agent_id = coder_agent.main[0].id - display_name = "NixOS rebuild" - cron = var.update_schedule - # The boot script has already switched by the time the agent exists. - run_on_start = false - start_blocks_login = false - timeout = 3600 - log_path = "${local.log_dir}/coder-script.log" - - script = templatefile("${path.module}/scripts/rebuild.sh.tftpl", { - # The same library the boot path uses, taken from the module that owns it. - LOG_SH = module.amazon_init.log_library - LIFECYCLE_SH = local.lifecycle_sh - ARG_FLAKE_REF = local.flake_url - ARG_FLAKE_BRANCH = var.flake_branch - ARG_FLAKE_ATTR = local.flake_attr - ARG_UPDATE_PROCESS = data.coder_parameter.update_process.value - ARG_ACCESS_URL = data.coder_workspace.me.access_url - ARG_LOG_SOURCE_ID = module.amazon_init.log_source_id - }) -} - locals { # One map so the AMI architecture, coder_agent.arch and the flake attribute # cannot disagree. @@ -306,33 +265,24 @@ locals { "m7g.large" = { agent = "arm64", ami = "arm64", attr = "aarch64" } "m7g.xlarge" = { agent = "arm64", ami = "arm64", attr = "aarch64" } } - arch = local.arch_map[data.coder_parameter.instance_type.value] - flake_attr = replace(var.flake_attr, "$ARCH", local.arch.attr) - log_dir = "/var/log/coder-nixos" - - # `git clone` is what runs on the instance, so accept a Nix-style flake - # reference too and reduce it to a plain remote: strip a `git+` scheme - # prefix and any query string. `flake_branch` carries the ref instead. - flake_url = replace(replace(var.flake_ref, "/^git\\+/", ""), "/\\?.*$/", "") - - # Sourced verbatim into both entrypoints. Plain shell rather than a template - # so it stays readable and gets covered by the repo's shellcheck. - lifecycle_sh = file("${path.module}/modules/nix/lifecycle.sh") - - state_dir = "/var/lib/coder-nixos" - flake_dir = "/etc/nixos" - - # Handed to the amazon-init module, which runs it as root before starting - # the agent and otherwise does not look inside it. - boot_script = templatefile("${path.module}/scripts/boot.sh.tftpl", { - LIFECYCLE_SH = local.lifecycle_sh - ARG_FLAKE_REF = local.flake_url - ARG_FLAKE_BRANCH = var.flake_branch - ARG_FLAKE_ATTR = local.flake_attr - ARG_STATE_DIR = local.state_dir - ARG_LOG_DIR = local.log_dir - ARG_FLAKE_DIR = local.flake_dir - }) + arch = local.arch_map[data.coder_parameter.instance_type.value] +} + +# Everything about the flake: the checkout, the boot-time rebuild and the +# periodic one. It knows nothing about EC2 -- `boot_script` is a string for +# whoever runs scripts on the machine. See ./modules/nix/README.md. +module "nix" { + source = "./modules/nix" + + # Empty while the workspace is stopped, when there is no agent to attach the + # periodic rebuild to. The module skips the script in that case. + agent_id = try(coder_agent.main[0].id, "") + + flake_ref = var.flake_ref + flake_attr = var.flake_attr + arch = local.arch.attr + update_schedule = var.update_schedule + update_process = data.coder_parameter.update_process.value } # Gets Coder onto the instance and runs one script on every boot. It knows @@ -343,7 +293,7 @@ module "amazon_init" { agent_token = try(coder_agent.main[0].token, "") agent_init_script = try(coder_agent.main[0].init_script, "") - boot_script = local.boot_script + boot_script = module.nix.boot_script log_display_name = "NixOS" log_icon = "/icon/nix.svg" @@ -380,11 +330,6 @@ resource "aws_instance" "dev" { # NixOS AMIs are republished weekly and garbage-collected after 90 days. # Without this, a new AMI id replaces every live workspace. ignore_changes = [ami] - - precondition { - condition = module.amazon_init.user_data_bytes < 16384 - error_message = "Rendered user-data is ${module.amazon_init.user_data_bytes} bytes; EC2 allows at most 16384." - } } } @@ -396,11 +341,11 @@ resource "coder_metadata" "workspace_info" { } item { key = "Flake URI" - value = "${local.flake_url}?ref=${var.flake_branch}#${local.flake_attr}" + value = module.nix.flake_uri } item { key = "Build logs location" - value = local.log_dir + value = module.nix.log_dir } } diff --git a/registry/coder/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md similarity index 83% rename from registry/coder/templates/aws-nixos/modules/amazon-init/README.md rename to registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md index 954c412c1..cb15dde90 100644 --- a/registry/coder/templates/aws-nixos/modules/amazon-init/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md @@ -18,13 +18,6 @@ module "amazon_init" { resource "aws_instance" "dev" { user_data = module.amazon_init.user_data - - lifecycle { - precondition { - condition = module.amazon_init.user_data_bytes < 16384 - error_message = "Rendered user-data is ${module.amazon_init.user_data_bytes} bytes; EC2 allows at most 16384." - } - } } ``` @@ -57,7 +50,8 @@ things follow: 3. **Logging** — the shell library is written to `$${runtime_dir}/log.sh` and sourced, and the log source is registered. 4. **Identity** — `$${runtime_dir}/workspace.json`, mode 0644, no secrets. -5. **Files** — anything in `files`. +5. **Files** — anything in `files`, a map of absolute path to text, written + mode 0644 with parent directories created. 6. **Your boot script**, as a child process. 7. **The agent**, always, whatever step 6 did. @@ -98,8 +92,14 @@ first boot it does not exist for as long as the boot script runs. So `coder_log `, `coder_log_pipe ` for a stream, and `coder_log_tail ` for the end of a transcript. -It is exported as `log_library` for callers that want the same functions in a -`coder_script` later, and it is on the instance at `$${runtime_dir}/log.sh`. +The library is internal to this module — it is not an output. On the instance +it is at `$${runtime_dir}/log.sh`, which is where anything the agent runs later +should source it from, and `CODER_LOG_READY` is exported so a process that +sources it does not register the source a second time. + +The source id is a `random_uuid` held in Terraform state: stable across +stop/start, new only when the workspace is recreated, by which point the agent +and its logs are new anyway. Two things it handles that are easy to get wrong: @@ -123,6 +123,9 @@ That is transparent to `amazon-init`: it only checks the first two bytes for `$${runtime_dir}/bootstrap.sh`, so the real script is on disk when a boot needs debugging. -`user_data_bytes` is exported unwrapped so a caller can assert the limit — the -`user_data` output itself is sensitive, because the token is inside it, and -Terraform suppresses messages derived from sensitive values. +The 16 KiB limit is asserted here, as a `precondition` on the `user_data` +output, so the plan fails in the module that decides what goes into user-data +rather than at apply time with an EC2 error that names no cause. Anything +passed through `files` counts against it — and note that those contents are +gzipped before user-data is gzipped again, which buys nothing, so `files` is +for small text. diff --git a/registry/coder/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf similarity index 77% rename from registry/coder/templates/aws-nixos/modules/amazon-init/main.tf rename to registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf index 16a47f214..6f8e93888 100644 --- a/registry/coder/templates/aws-nixos/modules/amazon-init/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf @@ -13,9 +13,21 @@ terraform { source = "coder/coder" version = ">= 2.5" } + random = { + source = "hashicorp/random" + version = ">= 3.0" + } } } +# The log source has to keep the same id for the life of the workspace -- +# Coder treats a repeat POST of a known id as a no-op, and a value that +# changed every plan would churn user-data on every start. Terraform state is +# exactly the right place for that: stable across stop/start, new only when +# the workspace is recreated, by which point the agent and its logs are new +# too. +resource "random_uuid" "log_source" {} + data "coder_workspace" "me" {} data "coder_workspace_owner" "me" {} @@ -57,14 +69,16 @@ variable "boot_script" { variable "files" { description = <<-EOT - Extra files to write before the boot script runs, keyed by absolute path. - Content is carried base64-encoded, so any bytes are safe. + Files to write before the boot script runs: absolute path to contents. + Written mode 0644, parent directories created. + + Contents are carried gzipped and base64-encoded, so any text is safe -- + but note that user-data is itself compressed, and compressing twice buys + nothing. This is for small files; the size precondition on `user_data` is + what stops it being abused. EOT - type = map(object({ - content = string - mode = optional(string, "0644") - })) - default = {} + type = map(string) + default = {} validation { condition = alltrue([for path in keys(var.files) : startswith(path, "/")]) @@ -94,17 +108,6 @@ variable "curl_resolve_command" { default = "" } -variable "log_source_id" { - description = "UUID of the log source boot output is streamed to. Must be constant across builds; Coder treats a repeat POST of the same id as a no-op." - type = string - default = "6e1f4a2c-9b3d-4c8e-8a71-5f0d2b6c4e93" - - validation { - condition = can(regex("^[0-9a-f-]{36}$", var.log_source_id)) - error_message = "log_source_id must be a UUID." - } -} - variable "log_display_name" { description = "Name of that log source in the workspace UI." type = string @@ -139,13 +142,14 @@ variable "hostname" { locals { hostname = var.hostname != "" ? var.hostname : lower(data.coder_workspace.me.name) - # Written by the bootstrap script before the boot script runs. Content is - # base64 so that a heredoc in the file cannot terminate the one writing it. + # Written by the bootstrap script before the boot script runs. Carried + # gzipped and base64-encoded so that no content can terminate the heredoc + # that writes it. files_sh = join("\n", [ - for path, file in var.files : <<-SH - install -d -m 0755 "$(dirname ${path})" - printf '%s' '${base64encode(file.content)}' | base64 -d >'${path}' - chmod ${file.mode} '${path}' + for path, content in var.files : <<-SH + install -d -m 0755 "$(dirname '${path}')" + printf '%s' '${base64gzip(content)}' | base64 -d | gzip -dc >'${path}' + chmod 0644 '${path}' SH ]) @@ -160,7 +164,7 @@ locals { ARG_RUNTIME_DIR = var.runtime_dir ARG_PATH = var.path - ARG_LOG_SOURCE_ID = var.log_source_id + ARG_LOG_SOURCE_ID = random_uuid.log_source.result ARG_LOG_DISPLAY_NAME_B64 = base64encode(var.log_display_name) ARG_LOG_ICON = var.log_icon ARG_LOG_BUDGET = var.log_budget_bytes @@ -198,21 +202,22 @@ output "user_data" { description = "Rendered EC2 user-data. Sensitive: it carries the agent token." value = local.user_data sensitive = true -} -output "user_data_bytes" { - description = "Size of the rendered user-data, for the caller's 16 KiB precondition. Unwrapped so the number can be shown in an error message." - value = nonsensitive(length(local.user_data)) -} - -output "log_library" { - description = "The shell logging library, for callers that want `coder_log` in their own `coder_script`s. On the instance it is also at `$${runtime_dir}/log.sh`." - value = file("${path.module}/scripts/log.sh") + # EC2 rejects user-data over 16 KiB, and it does so at apply time with an + # error that says nothing about which part grew. Checking here fails the + # plan instead, in the module that decides what goes in. + # + # nonsensitive because the length is sensitive by propagation, and Terraform + # suppresses error messages derived from sensitive values. + precondition { + condition = nonsensitive(length(local.user_data)) < 16384 + error_message = "Rendered user-data is ${nonsensitive(length(local.user_data))} bytes; EC2 allows at most 16384." + } } output "log_source_id" { description = "Log source the boot output is streamed to." - value = var.log_source_id + value = random_uuid.log_source.result } output "runtime_dir" { diff --git a/registry/coder/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl similarity index 100% rename from registry/coder/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl rename to registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl diff --git a/registry/coder/templates/aws-nixos/modules/amazon-init/scripts/log.sh b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh similarity index 100% rename from registry/coder/templates/aws-nixos/modules/amazon-init/scripts/log.sh rename to registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/README.md b/registry/coder-labs/templates/aws-nixos/modules/nix/README.md new file mode 100644 index 000000000..dfaaadc59 --- /dev/null +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/README.md @@ -0,0 +1,85 @@ +# nix + +The flake lifecycle: keep a checkout in sync with a Git remote, decide whether +the running system is out of date, and rebuild it. + +Nothing here knows about EC2, user-data, or how the instance came to exist. +The boot path is a **string** the caller hands to whatever runs scripts on the +machine, and the periodic rebuild is an ordinary `coder_script`. On AWS the +caller is [`../amazon-init`](../amazon-init/README.md), but nothing depends on +that. + +```tf +module "nix" { + source = "./modules/nix" + + agent_id = try(coder_agent.main[0].id, "") + flake_ref = "git+https://github.com/coder/nixos-example-flake?ref=main" + arch = "x86_64" +} +``` + +## The flake reference + +One string, in the form `nix` itself accepts. `?ref=` carries the branch, and +without it the remote's default branch is used — resolved on the instance, +since Terraform cannot know it without talking to the remote. + +| `flake_ref` | builds | +| ------------------------------------- | --------------- | +| `https://host/org/repo` | default branch | +| `git+https://host/org/repo?ref=dev` | `dev` | +| `git+ssh://git@host/org/repo?ref=dev` | `dev`, over SSH | + +`$ARCH` in `flake_attr` is replaced with `arch`, so one template can offer both +architectures without the attribute and the machine disagreeing. + +## What runs where + +`boot_script` runs as root on every boot, before the agent is started. It syncs +the checkout, and rebuilds only if the configuration changed: + +- a **clean** checkout is fast-forwarded to the remote +- a **dirty** one, or one carrying local commits, is left alone and built as it + is — someone is working on it +- a checkout on a different branch than requested says so rather than silently + building the wrong thing + +The `coder_script` does the same later, on `update_schedule`, with +`update_process = boot` (stage for the next restart) or `switch` (apply now). +A `flock` in `state_dir` keeps the two from overlapping; the scheduled run +skips rather than queues. + +## Logging + +`scripts/lifecycle.sh` sends output through a single `nix_log ` +hook. Define it and progress goes wherever you want; leave it undefined and it +prints to stdout. + +`boot_script` sets that hook up from `CODER_LOG_LIBRARY` when the bootstrapper +provides one — that is how output reaches the workspace UI before an agent +exists — and falls back to plain `echo` when it does not. + +Raw `nix` output is filtered before it is logged: store-path lists, +per-derivation build output and lock-file noise are dropped, and list headers +that promised a list are rewritten as sentences. Transcripts in `log_dir` keep +everything, unfiltered. + +## Inputs and outputs + +`flake_dir`, `state_dir` and `log_dir` default to `/etc/nixos`, +`/var/lib/coder-nixos` and `/var/log/coder-nixos`, and both scripts take them +from here — the paths are defined once. + +Outputs exist for the things a caller genuinely cannot do itself: +`boot_script`, `flake_uri` and `flake_attr` for display, `log_dir` and +`flake_dir` for pointing people at, and `version_command` for a `coder_agent` +metadata block — which has to be declared inline on the agent, though what it +means for a NixOS machine to be up to date does not belong in a template. + +## Moving this to the registry + +It is a self-contained Terraform module already. Publishing it as +`registry.coder.com/coder/nix` needs `.tftest.hcl` coverage and a decision +about `nixos-rebuild` on non-NixOS hosts (`nix profile` would be the +equivalent), not untangling. diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf new file mode 100644 index 000000000..9f93773a7 --- /dev/null +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf @@ -0,0 +1,189 @@ +# The flake lifecycle: keep a checkout in sync, decide whether the running +# system is out of date, and rebuild it. +# +# Nothing here knows about EC2, user-data or how the instance was started. The +# boot path is exposed as a string for whatever puts scripts on the machine -- +# on AWS that is ../amazon-init -- and the periodic rebuild is an ordinary +# coder_script. + +terraform { + required_version = ">= 1.3" + + required_providers { + coder = { + source = "coder/coder" + version = ">= 2.5" + } + } +} + +data "coder_workspace" "me" {} + +variable "agent_id" { + description = "Agent that runs the periodic rebuild. May be empty while the workspace is stopped." + type = string +} + +variable "flake_ref" { + description = <<-EOT + Git reference to the flake, in the form `nix` itself accepts: + `https://host/org/repo`, optionally with a `git+` prefix and a `?ref=` + branch. Without `?ref=` the remote's default branch is used. + + The configuration must be committed -- a Git flake reference only ever + sees committed files. + EOT + type = string + + validation { + condition = can(regex("^(git\\+)?(https?|ssh)://", var.flake_ref)) + error_message = "flake_ref must be an http(s) or ssh Git URL, optionally prefixed with git+." + } +} + +variable "flake_attr" { + description = "`nixosConfigurations` attribute to build. `$ARCH` is replaced with `arch`." + type = string + default = "workspace-$ARCH" +} + +variable "arch" { + description = "Nix architecture name substituted into `flake_attr`." + type = string + default = "x86_64" + + validation { + condition = contains(["x86_64", "aarch64"], var.arch) + error_message = "arch must be x86_64 or aarch64." + } +} + +variable "update_schedule" { + description = <<-EOT + Cron schedule for the periodic rebuild, or empty to disable it. + + SIX fields with seconds first, in the workspace's timezone. A five field + expression is silently misinterpreted rather than rejected, and + descriptors like `@daily` pass validation then fail on the agent. + EOT + type = string + default = "0 0 4 * * *" + + validation { + condition = var.update_schedule == "" || length(split(" ", trimspace(var.update_schedule))) == 6 + error_message = "update_schedule must be a 6-field cron expression (seconds first), or empty." + } +} + +variable "update_process" { + description = "`switch` applies the new generation immediately; `boot` stages it for the next restart." + type = string + default = "boot" + + validation { + condition = contains(["boot", "switch"], var.update_process) + error_message = "update_process must be boot or switch." + } +} + +variable "flake_dir" { + description = "Checkout to build. Owned by the workspace user so the configuration can be edited in place." + type = string + default = "/etc/nixos" +} + +variable "state_dir" { + description = "Revision marker and rebuild lock." + type = string + default = "/var/lib/coder-nixos" +} + +variable "log_dir" { + description = "Rebuild transcripts." + type = string + default = "/var/log/coder-nixos" +} + +locals { + # `git clone` is what runs on the instance, so reduce the reference to a + # plain remote: strip a `git+` scheme prefix and any query string. + flake_url = replace(replace(var.flake_ref, "/^git\\+/", ""), "/\\?.*$/", "") + + # An absent `?ref=` means the remote's default branch, resolved on the + # instance -- Terraform cannot know it without talking to the remote. + flake_branch = try(regex("[?&]ref=([^&#]+)", var.flake_ref)[0], "") + + flake_attr = replace(var.flake_attr, "$ARCH", var.arch) + + lifecycle_sh = file("${path.module}/scripts/lifecycle.sh") + + script_args = { + LIFECYCLE_SH = local.lifecycle_sh + ARG_FLAKE_URL = local.flake_url + ARG_FLAKE_BRANCH = local.flake_branch + ARG_FLAKE_ATTR = local.flake_attr + ARG_FLAKE_DIR = var.flake_dir + ARG_STATE_DIR = var.state_dir + ARG_LOG_DIR = var.log_dir + } +} + +resource "coder_script" "nixos_rebuild" { + count = var.update_schedule == "" ? 0 : data.coder_workspace.me.start_count + agent_id = var.agent_id + display_name = "NixOS rebuild" + cron = var.update_schedule + # The boot path has already rebuilt by the time the agent exists. + run_on_start = false + start_blocks_login = false + timeout = 3600 + log_path = "${var.log_dir}/coder-script.log" + + script = templatefile("${path.module}/scripts/rebuild.sh.tftpl", merge(local.script_args, { + ARG_UPDATE_PROCESS = var.update_process + })) +} + +output "boot_script" { + description = "Applies the flake. Hand this to whatever runs a script on every boot; it expects to run as root." + value = templatefile("${path.module}/scripts/boot.sh.tftpl", local.script_args) +} + +output "flake_uri" { + description = "The reference actually built, normalised for display." + value = "${local.flake_url}${local.flake_branch == "" ? "" : "?ref=${local.flake_branch}"}#${local.flake_attr}" +} + +output "flake_attr" { + description = "`nixosConfigurations` attribute after `$ARCH` substitution." + value = local.flake_attr +} + +output "flake_dir" { + description = "Checkout on the instance." + value = var.flake_dir +} + +output "log_dir" { + description = "Where rebuild transcripts are written." + value = var.log_dir +} + +output "version_command" { + description = <<-EOT + Shell that reports the running NixOS version, and whether a generation is + staged but not yet booted. For a `coder_agent` metadata block, which has + to be declared inline on the agent. + EOT + # /run/current-system is the activated system; /run/booted-system is what + # the kernel booted and still points at the previous generation after a + # switch, which would mark every new workspace as needing a restart. + value = <<-EOT + version=$(nixos-version 2>/dev/null || echo unknown) + if [ "$(readlink -f /run/current-system)" = "$(readlink -f /nix/var/nix/profiles/system)" ]; then + echo "$version" + else + echo "$version (restart to apply update)" + fi + EOT +} diff --git a/registry/coder/templates/aws-nixos/scripts/boot.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl similarity index 63% rename from registry/coder/templates/aws-nixos/scripts/boot.sh.tftpl rename to registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl index e495ff3b6..25e143ddd 100644 --- a/registry/coder/templates/aws-nixos/scripts/boot.sh.tftpl +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl @@ -1,27 +1,43 @@ #!/usr/bin/env bash -# Applies the flake on every boot. Handed to the amazon-init module as an -# opaque string and run by it as root, after the agent handoff exists and -# before the agent is started -- so everything here happens while the +# Applies the flake on every boot, as root. +# +# Handed to whatever bootstraps the instance as an opaque string. On AWS that +# is the amazon-init module, which runs this after the agent handoff exists +# and before the agent is started -- so everything here happens while the # workspace is still building, and the agent never sees a half-applied system. # # Runs on every boot, so it must be idempotent. set -euo pipefail -FLAKE_REF='${ARG_FLAKE_REF}' +FLAKE_URL='${ARG_FLAKE_URL}' FLAKE_BRANCH='${ARG_FLAKE_BRANCH}' FLAKE_ATTR='${ARG_FLAKE_ATTR}' +FLAKE_DIR='${ARG_FLAKE_DIR}' STATE_DIR='${ARG_STATE_DIR}' LOG_DIR='${ARG_LOG_DIR}' -FLAKE_DIR='${ARG_FLAKE_DIR}' install -d -m 0755 "$STATE_DIR" "$LOG_DIR" -# Provided by the amazon-init module, along with the CODER_* variables it -# needs. Logging must never be fatal, hence the fallbacks. -# shellcheck source=/dev/null -. "$${CODER_LOG_LIBRARY:?the amazon-init module sets this}" +# Optional. A bootstrapper that can stream to the workspace UI before the +# agent exists points CODER_LOG_LIBRARY at a library providing these; without +# one, output still reaches whatever captured this script's stdout. +if [ -r "$${CODER_LOG_LIBRARY:-}" ]; then + # shellcheck source=/dev/null + . "$CODER_LOG_LIBRARY" +else + coder_log() { + local level="$1" + shift + printf '%s: %s\n' "$level" "$*" + } + coder_log_pipe() { cat; } + coder_log_tail() { + [ -f "$1" ] || return 0 + tail -n "$${2:-200}" "$1" + } +fi nix_log() { coder_log "$@" || true; } @@ -43,7 +59,7 @@ on_error() { trap on_error ERR nix_lock -nix_sync_checkout "$FLAKE_REF" "$FLAKE_BRANCH" +nix_sync_checkout "$FLAKE_URL" "$FLAKE_BRANCH" REV=$(nix_needs_rebuild) && NEEDS_REBUILD=1 || NEEDS_REBUILD=0 if [ "$NEEDS_REBUILD" -eq 0 ]; then diff --git a/registry/coder/templates/aws-nixos/modules/nix/lifecycle.sh b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh similarity index 83% rename from registry/coder/templates/aws-nixos/modules/nix/lifecycle.sh rename to registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh index f03a6151f..0af2634f3 100644 --- a/registry/coder/templates/aws-nixos/modules/nix/lifecycle.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh @@ -50,19 +50,35 @@ nix_own_checkout() { $NIX_SUDO chown -R "$owner" "$NIX_FLAKE_DIR" 2> /dev/null || true } +# The branch the checkout is actually on. With no `?ref=` in the flake +# reference there is nothing to ask but the checkout itself, and after a clone +# that is the remote's default branch. +nix_checkout_branch() { + $NIX_SUDO git -C "$NIX_FLAKE_DIR" rev-parse --abbrev-ref HEAD 2> /dev/null || true +} + nix_sync_checkout() { - local ref="$1" branch="$2" upstream_rev local_rev + local url="$1" branch="${2:-}" upstream_rev local_rev current_branch rc if [ ! -e "$NIX_FLAKE_DIR/flake.nix" ]; then - nix_log info "Cloning $ref into $NIX_FLAKE_DIR" + nix_log info "Cloning $url into $NIX_FLAKE_DIR" $NIX_SUDO install -d -m 0755 "$NIX_FLAKE_DIR" # Clone into a temporary directory and move the contents, because the # directory already exists (systemd-tmpfiles creates it) and git refuses # to clone into a non-empty one. local tmp tmp="$($NIX_SUDO mktemp -d)" - if ! $NIX_SUDO git clone --branch "$branch" "$ref" "$tmp/repo" 2>&1 | nix_filter_log; then - nix_log error "Could not clone $ref" + # PIPESTATUS, not the pipeline's status: that is the filter's, and the + # filter always succeeds. Without this a failed clone reads as a success + # and the next step fails somewhere much less obvious. + if [ -n "$branch" ]; then + $NIX_SUDO git clone --branch "$branch" "$url" "$tmp/repo" 2>&1 | nix_filter_log + else + $NIX_SUDO git clone "$url" "$tmp/repo" 2>&1 | nix_filter_log + fi + rc=${PIPESTATUS[0]} + if [ "$rc" -ne 0 ]; then + nix_log error "Could not clone $url${branch:+ (branch $branch)}" $NIX_SUDO rm -rf "$tmp" return 1 fi @@ -72,16 +88,31 @@ nix_sync_checkout() { return 0 fi + current_branch="$(nix_checkout_branch)" + # An empty branch means "whatever this checkout tracks", which after the + # clone above is the remote's default. + [ -n "$branch" ] || branch="$current_branch" + if nix_checkout_dirty; then nix_log info "$NIX_FLAKE_DIR has local changes; building those instead of $branch" return 0 fi + # Asking for a different branch than the checkout is on is not something to + # resolve silently: the fast-forward below would fail its ancestry test and + # report local commits, which is not what happened. + if [ -n "$current_branch" ] && [ "$current_branch" != "$branch" ]; then + nix_log warn "$NIX_FLAKE_DIR is on $current_branch, not $branch; building $current_branch" + nix_log warn "Check out $branch there, or delete $NIX_FLAKE_DIR to start from the remote" + return 0 + fi + # Fetching as root writes into .git, so re-assert ownership afterwards. - $NIX_SUDO git -C "$NIX_FLAKE_DIR" fetch --quiet origin "$branch" 2>&1 | nix_filter_log || { + $NIX_SUDO git -C "$NIX_FLAKE_DIR" fetch --quiet origin "$branch" 2>&1 | nix_filter_log + if [ "${PIPESTATUS[0]}" -ne 0 ]; then nix_log warn "Could not reach the remote; building the existing checkout" return 0 - } + fi local_rev="$(nix_checkout_rev)" upstream_rev="$($NIX_SUDO git -C "$NIX_FLAKE_DIR" rev-parse FETCH_HEAD 2> /dev/null || true)" diff --git a/registry/coder/templates/aws-nixos/scripts/rebuild.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/rebuild.sh.tftpl similarity index 76% rename from registry/coder/templates/aws-nixos/scripts/rebuild.sh.tftpl rename to registry/coder-labs/templates/aws-nixos/modules/nix/scripts/rebuild.sh.tftpl index cf0c170fa..c31ea8634 100644 --- a/registry/coder/templates/aws-nixos/scripts/rebuild.sh.tftpl +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/rebuild.sh.tftpl @@ -3,29 +3,19 @@ # # coder_script bodies run as the workspace user through its login shell and # there is no run_as argument, so everything privileged goes through sudo. +# Output is captured by the agent and shown under this script's own name, so +# it goes to stdout rather than the boot log source. set -euo pipefail -FLAKE_REF='${ARG_FLAKE_REF}' +FLAKE_URL='${ARG_FLAKE_URL}' FLAKE_BRANCH='${ARG_FLAKE_BRANCH}' FLAKE_ATTR='${ARG_FLAKE_ATTR}' OPERATION='${ARG_UPDATE_PROCESS}' -ACCESS_URL='${ARG_ACCESS_URL}' -LOG_SOURCE_ID='${ARG_LOG_SOURCE_ID}' -STATE_DIR=/var/lib/coder-nixos -LOG_DIR=/var/log/coder-nixos -FLAKE_DIR=/etc/nixos - -CODER_ACCESS_URL="$ACCESS_URL" -CODER_LOG_SOURCE_ID="$LOG_SOURCE_ID" -CODER_LOG_STATE_DIR="$STATE_DIR" -# Always present: the agent sets it for the scripts it runs. Logging simply -# stays disabled if it is not. -CODER_AGENT_TOKEN="$${CODER_AGENT_TOKEN:-}" -CODER_LOG_BUDGET=524288 - -${LOG_SH} +FLAKE_DIR='${ARG_FLAKE_DIR}' +STATE_DIR='${ARG_STATE_DIR}' +LOG_DIR='${ARG_LOG_DIR}' NIX_FLAKE_DIR="$FLAKE_DIR" NIX_FLAKE_ATTR="$FLAKE_ATTR" @@ -44,7 +34,7 @@ fi TRANSCRIPT="$LOG_DIR/rebuild-$(date -u +%Y%m%dT%H%M%SZ).log" -nix_sync_checkout "$FLAKE_REF" "$FLAKE_BRANCH" +nix_sync_checkout "$FLAKE_URL" "$FLAKE_BRANCH" REV=$(nix_needs_rebuild) && NEEDS_REBUILD=1 || NEEDS_REBUILD=0 if [ "$NEEDS_REBUILD" -eq 0 ]; then diff --git a/registry/coder/templates/aws-nixos/modules/nix/README.md b/registry/coder/templates/aws-nixos/modules/nix/README.md deleted file mode 100644 index 8cd664fc4..000000000 --- a/registry/coder/templates/aws-nixos/modules/nix/README.md +++ /dev/null @@ -1,62 +0,0 @@ -# Nix flake lifecycle - -`lifecycle.sh.tftpl` is the only part of this template that knows about Nix. -It is kept separate so it can be lifted into a standalone Coder registry -module without untangling it from EC2 and Coder specifics first. - -## Contract - -Inputs are environment variables, set by the caller: - -| Variable | Meaning | -| ---------------- | ---------------------------------------------- | -| `NIX_FLAKE_DIR` | checkout to sync and build (e.g. `/etc/nixos`) | -| `NIX_FLAKE_ATTR` | `nixosConfigurations` attribute | -| `NIX_STATE_DIR` | revision marker and lock file | -| `NIX_LOG_DIR` | transcripts | - -Output goes through a `nix_log ` function that the caller -may define; it falls back to stdout. That hook is the only coupling to -Coder, and it is one function. - -| Function | Responsibility | -| --------------------------------------- | ----------------------------------------------- | -| `nix_sync_checkout ` | clone, fast-forward, or leave local work alone | -| `nix_needs_rebuild` | echo the revision, return 0 when work is needed | -| `nix_apply ` | build and apply | -| `nix_record_rev ` | record what was applied | -| `nix_pending_generation` | true when a generation is staged but not booted | -| `nix_filter_log` | drop noise from nix's output | -| `nix_lock [nowait]` | serialise concurrent callers | - -## Where this is going - -The eventual `registry/coder/modules/nix/` should manage a flake lifecycle on -any Linux host, not only NixOS: `nix develop`, `nix profile`, devshells, with -`nixos-rebuild` as one backend among several. - -That is why the split is **resolve → decide → apply**, and why `nix_apply` is -a single function. It is the only NixOS-specific piece, so a `nix develop` -backend becomes a sibling of it rather than a rewrite. `nix_flake_rev`, -`nix_needs_rebuild`, `nix_filter_log` and `nix_lock` are already -platform-agnostic. - -No Terraform module is published yet; this is structure and documentation. - -## Two invariants worth not breaking - -**Nothing is injected at evaluation time.** The commands this module runs are -exactly what a user can type by hand -- no `--override-input`, no `--impure`, -no `--no-write-lock-file`. That is deliberate: the template does not control -the flake, and a rebuild that only works with special flags is a rebuild the -user cannot reproduce. Facts about the workspace are exposed as a runtime -JSON file instead, and the agent token lives in a tmpfs file at mode 0600. -Do not "simplify" either into a flake input -- anything Nix sees is -world-readable in the store and persists across generations. - -**Local work is never discarded.** `nix_sync_checkout` fast-forwards only a -clean checkout on its tracking branch; a dirty tree or local commits are -built as they are. - -**Logging is best-effort.** A failure to report progress must never abort a -rebuild. `nix_log` failures are swallowed by the caller. From 080b3919d6546e69bab64dfc84963999f82a5c90 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 21 Sep 2026 16:59:23 +0000 Subject: [PATCH 06/24] refactor(aws-nixos): drop the curl fallback, name the attributes coder-workspace-* `curl_resolve_command` existed for an AMI that might not ship curl, and the NixOS one does: generation 1, the image's own system before anything has been rebuilt, already has it on PATH. The hook could never have helped anyway -- the log source is registered before the boot script runs, so a PATH the boot script sets comes too late. Without it the library also stops `eval`-ing a string handed to it from Terraform. The flake's configurations are now `coder-workspace-`; an unqualified `workspace-x86_64` says nothing about who consumes it in a configuration that may hold others. Flake side is coder/nixos-example-flake@b7930fa. --- registry/coder-labs/templates/aws-nixos/README.md | 8 ++++---- registry/coder-labs/templates/aws-nixos/main.tf | 6 +----- .../aws-nixos/modules/amazon-init/README.md | 7 ++++--- .../aws-nixos/modules/amazon-init/main.tf | 11 ----------- .../amazon-init/scripts/bootstrap.sh.tftpl | 3 --- .../aws-nixos/modules/amazon-init/scripts/log.sh | 15 +++++---------- .../templates/aws-nixos/modules/nix/main.tf | 2 +- 7 files changed, 15 insertions(+), 37 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 53900d170..53af35ac0 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -76,7 +76,7 @@ committed: a Git flake reference only ever sees committed files. selection you would make by hand: ```console -nixos-rebuild switch --flake 'github:your-org/config#workspace-x86_64' +nixos-rebuild switch --flake 'github:your-org/config#coder-workspace-x86_64' ``` `$ARCH` in `flake_attr` is replaced with `x86_64` or `aarch64` to match the chosen instance type. @@ -129,7 +129,7 @@ The configuration is a git checkout at `/etc/nixos`, owned by the workspace user, and that is what the template builds. So the command is the ordinary one: ```console -sudo nixos-rebuild switch --flake /etc/nixos#workspace-x86_64 +sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-x86_64 ``` No overrides, no `--impure`, no injected inputs — what you get by hand is @@ -174,7 +174,7 @@ Set `update_schedule = ""` to disable periodic rebuilds entirely. To rebuild immediately: ```console -sudo nixos-rebuild switch --flake 'git+https://github.com/coder/nixos-example-flake?ref=main#workspace-x86_64' \ +sudo nixos-rebuild switch --flake 'git+https://github.com/coder/nixos-example-flake?ref=main#coder-workspace-x86_64' \ --override-input coder-vars path:/etc/coder/vars --no-write-lock-file --refresh ``` @@ -219,7 +219,7 @@ volume is picked up on the next restart. Both `x86_64` and `arm64` (Graviton) instance types are offered. The AMI filter, `coder_agent.arch` and the flake attribute are all derived from the instance type, so they cannot disagree — but your flake must expose a configuration for the architecture you select. The -reference flake ships `workspace-x86_64` and `workspace-aarch64`. +reference flake ships `coder-workspace-x86_64` and `coder-workspace-aarch64`. The smallest instance type offered is `t3.medium` on purpose: the NixOS AMI configures no swap and the Nix store shares the root volume, so a rebuild that has to compile anything will exhaust a diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index e9773eebd..84ce20ace 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -36,7 +36,7 @@ variable "flake_ref" { variable "flake_attr" { description = "`nixosConfigurations` attribute to build. `$ARCH` is replaced with `x86_64` or `aarch64` to match the instance type." type = string - default = "workspace-$ARCH" + default = "coder-workspace-$ARCH" } variable "nixos_release" { @@ -297,10 +297,6 @@ module "amazon_init" { log_display_name = "NixOS" log_icon = "/icon/nix.svg" - - # The AMI may not have curl before the first switch, and on NixOS the way to - # get one is to build it. - curl_resolve_command = "printf '%s' \"$(nix build --no-link --print-out-paths nixpkgs#curl)/bin/curl\"" } resource "aws_instance" "dev" { diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md index cb15dde90..b4c689f67 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md @@ -108,9 +108,10 @@ Two things it handles that are easy to get wrong: every later log from every source is dropped permanently. The library tracks its own usage and goes quiet at `log_budget_bytes`, which defaults to half the cap. -- **No curl.** A minimal AMI may not have one. `curl_resolve_command` is a - command that prints a path to a binary, run once, only if `curl` is not - already on `PATH`. +- **No curl.** The image must have one on `PATH` — the NixOS AMI does, in its + own first generation, before anything has been rebuilt. An image without one + simply gets no logs: every function here fails closed, because logging must + never be the reason a boot fails. ## The user-data wrapper diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf index 6f8e93888..80982b4cf 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf @@ -98,16 +98,6 @@ variable "path" { default = "/run/current-system/sw/bin" } -variable "curl_resolve_command" { - description = <<-EOT - Shell command that prints a path to a `curl` binary, used only when the - image has none. Logging is the only thing that needs it, and it is the one - thing this module cannot work out for an image it does not know. - EOT - type = string - default = "" -} - variable "log_display_name" { description = "Name of that log source in the workspace UI." type = string @@ -168,7 +158,6 @@ locals { ARG_LOG_DISPLAY_NAME_B64 = base64encode(var.log_display_name) ARG_LOG_ICON = var.log_icon ARG_LOG_BUDGET = var.log_budget_bytes - ARG_CURL_RESOLVE_B64 = base64encode(var.curl_resolve_command) ARG_HOSTNAME = local.hostname ARG_WORKSPACE_NAME = data.coder_workspace.me.name diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl index 7c2e256d2..bfb4affb7 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl @@ -23,7 +23,6 @@ WORKSPACE_NAME='${ARG_WORKSPACE_NAME}' OWNER='${ARG_OWNER}' OWNER_NAME_B64='${ARG_OWNER_NAME_B64}' OWNER_EMAIL='${ARG_OWNER_EMAIL}' -CURL_RESOLVE_B64='${ARG_CURL_RESOLVE_B64}' RUNTIME_DIR='${ARG_RUNTIME_DIR}' @@ -76,8 +75,6 @@ export CODER_AGENT_TOKEN="$AGENT_TOKEN" export CODER_LOG_SOURCE_ID="$LOG_SOURCE_ID" export CODER_LOG_STATE_DIR="$RUNTIME_DIR" export CODER_LOG_BUDGET="$LOG_BUDGET" -export CODER_CURL_RESOLVE -CODER_CURL_RESOLVE=$(printf '%s' "$CURL_RESOLVE_B64" | base64 -d) # Written to disk as well as sourced, so the boot script and anything the # agent runs later can use the same functions without a second copy. diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh index bd137b2a2..44259df45 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh @@ -16,18 +16,13 @@ CODER_LOG_MAX_LINE=2048 # process, and a child has no way to know whether it already happened. CODER_LOG_READY="${CODER_LOG_READY:-0}" -# A minimal AMI may not ship curl at all, and amazon-init's PATH is short. -# When it is missing, CODER_CURL_RESOLVE -- a command supplied by whoever -# instantiated this module, because only they know how to obtain a binary on -# their image -- is run once and expected to print a path to one. +# Resolved once. An image without curl gets no logs: every function here then +# fails closed, which is the right trade -- logging must never be the reason a +# boot fails. coder_curl() { if [ -z "${CODER_CURL:-}" ]; then - if command -v curl > /dev/null 2>&1; then - CODER_CURL=$(command -v curl) - elif [ -n "${CODER_CURL_RESOLVE:-}" ]; then - CODER_CURL=$(eval "$CODER_CURL_RESOLVE" 2> /dev/null || true) - fi - [ -n "${CODER_CURL:-}" ] && [ -x "$CODER_CURL" ] || return 1 + command -v curl > /dev/null 2>&1 || return 1 + CODER_CURL=$(command -v curl) export CODER_CURL fi "$CODER_CURL" "$@" diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf index 9f93773a7..1ead34ad1 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf @@ -44,7 +44,7 @@ variable "flake_ref" { variable "flake_attr" { description = "`nixosConfigurations` attribute to build. `$ARCH` is replaced with `arch`." type = string - default = "workspace-$ARCH" + default = "coder-workspace-$ARCH" } variable "arch" { From a4e281afb9e549ad07d4339cc25a238ede4d0553 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 21 Sep 2026 18:07:00 +0000 Subject: [PATCH 07/24] refactor(aws-nixos): hand the periodic rebuild to the configuration The template ran `nixos-rebuild` on a cron through a `coder_script`, duplicating most of the boot path to do it. Keeping a machine current is the machine's business, so the flake does it now with NixOS's own `system.autoUpgrade` (coder/nixos-example-flake@cd46473), on a systemd timer its owner can read and change in /etc/nixos. So `update_schedule`, the `update_process` parameter, the `coder_script` and `rebuild.sh.tftpl` are all gone. The nix module keeps the boot path, the reference parsing and its outputs. The cadence stops being a template knob -- a timer is decided at evaluation time and this template passes nothing into the flake at evaluation time, deliberately. To let a service on the instance log where the user will see it, `workspace.json` grows one key, `log_source_id`, which is the only thing of the sort a configuration cannot work out for itself. `values` is the general form of that: a map merged into the same file, so the next fact something needs costs one entry rather than a new file and a new variable. The nix module passes one through for the caller's convenience; it contributes nothing of its own, because the configuration already knows its checkout, attribute and directories. The file is now rendered by Terraform rather than assembled in shell, which is both how `values` gets in and how a full name containing a quote stops being able to break the document. Also in the logging library, found while watching an upgrade produce nothing: batches now flush on a timer as well as on size, so a slow producer appears live instead of sitting in a buffer until 50 lines have accumulated -- and, more to the point, so that what is buffered is not lost when the process is killed. --- .../coder-labs/templates/aws-nixos/README.md | 42 ++++++---- .../coder-labs/templates/aws-nixos/main.tf | 61 +++------------ .../aws-nixos/modules/amazon-init/README.md | 6 +- .../aws-nixos/modules/amazon-init/main.tf | 43 ++++++++-- .../amazon-init/scripts/bootstrap.sh.tftpl | 37 +++++---- .../modules/amazon-init/scripts/log.sh | 43 ++++++++-- .../templates/aws-nixos/modules/nix/README.md | 21 +++-- .../templates/aws-nixos/modules/nix/main.tf | 78 +++++-------------- .../modules/nix/scripts/rebuild.sh.tftpl | 66 ---------------- 9 files changed, 165 insertions(+), 232 deletions(-) delete mode 100644 registry/coder-labs/templates/aws-nixos/modules/nix/scripts/rebuild.sh.tftpl diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 53af35ac0..4cffa3376 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -153,29 +153,36 @@ A flake built from a git checkout ignores untracked files — `git add` a new ## Keeping workspaces up to date -The `update_process` parameter decides how a periodic rebuild is applied: +The schedule belongs to the machine, not to this template. The reference flake enables NixOS's own +[`system.autoUpgrade`](https://search.nixos.org/options?query=system.autoUpgrade), wrapped as +`coder.autoUpgrade` so the defaults make sense for a workspace: + +```nix +coder.autoUpgrade = { + dates = "04:40"; # systemd OnCalendar, not cron + operation = "boot"; # or "switch" +}; +``` - **`boot` (default)** — builds the new configuration and makes it the boot default without activating it. Nothing restarts while you are working; the change lands on your next workspace - restart. The "NixOS" metric in the workspace header shows `(restart to apply update)` when a - generation is staged. + restart, and the "NixOS version" metric shows `(restart to apply update)` until then. - **`switch`** — activates immediately, restarting any service whose definition changed. -The schedule comes from the `update_schedule` variable, default `0 0 4 * * *` (04:00 daily). - -> [!NOTE] -> `update_schedule` is a **six** field cron expression with seconds first, evaluated in the -> workspace's own timezone. A five field expression is silently misinterpreted rather than -> rejected, and descriptors like `@daily` pass validation but then fail on the agent. Set -> `time.timeZone` in your configuration so the schedule means what you intend. +Set `coder.autoUpgrade.enable = false` to turn it off, and edit `/etc/nixos` on the workspace to +change any of it — this is a normal NixOS timer, so `systemctl list-timers nixos-upgrade` and +`systemd-analyze calendar ''` tell you what will happen and when. -Set `update_schedule = ""` to disable periodic rebuilds entirely. +The timer is **not** `Persistent`: a schedule missed while the workspace was stopped is not made up +on the next boot, because booting already rebuilds from the checkout. Before each run the +configuration fast-forwards the checkout under the same lock the boot path uses, with the same +policy — a dirty tree or local commits are built as they are, never discarded. -To rebuild immediately: +To rebuild immediately, on the workspace: ```console -sudo nixos-rebuild switch --flake 'git+https://github.com/coder/nixos-example-flake?ref=main#coder-workspace-x86_64' \ - --override-input coder-vars path:/etc/coder/vars --no-write-lock-file --refresh +sudo systemctl start nixos-upgrade # sync, then rebuild, streamed to the UI +sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-x86_64 ``` ## Where the logs are @@ -186,10 +193,13 @@ under `these N derivations will be built:`, per-derivation compiler output, and `not writing modified lock file` notice. The complete transcript is on the instance: ```console -/var/log/coder-nixos/rebuild-latest.log # symlink to the most recent run -/var/log/coder-nixos/coder-script.log # the periodic rebuild script +/var/log/coder-nixos/rebuild-latest.log # symlink to the most recent boot rebuild ``` +Scheduled upgrades run as `nixos-upgrade.service` and write to the journal, which +`coder-stream-nixos-upgrade-logs.service` follows into the same **NixOS** log source while the +upgrade runs. `journalctl -u nixos-upgrade` has the unabridged copy. + Keeping compiler output out of the UI is not cosmetic. Coder caps agent logs at **1 MiB per agent**, shared across every log source, and exceeding it does not truncate — the log is marked overflowed and all later logs for that agent are dropped permanently. The template budgets itself diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index 84ce20ace..9388ed71a 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -49,23 +49,6 @@ variable "nixos_release" { default = "26.05" } -variable "update_schedule" { - description = <<-EOT - Cron schedule for the periodic `nixos-rebuild`, or empty to disable. - - SIX fields with seconds first, in the workspace's timezone. A five field - expression is silently misinterpreted rather than rejected, and - descriptors like `@daily` pass validation then fail on the agent. - EOT - type = string - default = "0 0 4 * * *" - - validation { - condition = var.update_schedule == "" || length(split(" ", trimspace(var.update_schedule))) == 6 - error_message = "update_schedule must be a 6-field cron expression (seconds first), or empty." - } -} - data "coder_parameter" "instance_type" { name = "instance_type" display_name = "Instance type" @@ -121,28 +104,6 @@ data "coder_parameter" "root_volume_size" { } } -data "coder_parameter" "update_process" { - name = "update_process" - display_name = "Configuration updates" - description = <<-EOT - How a periodic `nixos-rebuild` is applied while the workspace is running. - `boot` stages the new configuration without activating it, so nothing - restarts under you; `switch` activates it immediately. - EOT - type = "string" - default = "boot" - mutable = true - - option { - name = "Apply on next restart (boot)" - value = "boot" - } - option { - name = "Apply immediately (switch)" - value = "switch" - } -} - data "coder_workspace" "me" {} data "coder_workspace_owner" "me" {} @@ -194,10 +155,11 @@ resource "coder_agent" "main" { timeout = 30 script = "coder stat disk --path $HOME" } - # Makes `update_process = boot` visible; a staged generation is otherwise - # invisible and looks like updates being ignored. The command comes from the - # nix module -- metadata has to be declared on the agent, but what it means - # to be up to date is not this file's business. + # Makes a staged generation visible. `coder.autoUpgrade.operation = "boot"` + # in the flake stages rather than activates, which otherwise looks exactly + # like updates being ignored. The command comes from the nix module -- + # metadata has to be declared on the agent, but what it means to be up to + # date is not this file's business. metadata { key = "nixos" display_name = "NixOS version" @@ -274,15 +236,9 @@ locals { module "nix" { source = "./modules/nix" - # Empty while the workspace is stopped, when there is no agent to attach the - # periodic rebuild to. The module skips the script in that case. - agent_id = try(coder_agent.main[0].id, "") - - flake_ref = var.flake_ref - flake_attr = var.flake_attr - arch = local.arch.attr - update_schedule = var.update_schedule - update_process = data.coder_parameter.update_process.value + flake_ref = var.flake_ref + flake_attr = var.flake_attr + arch = local.arch.attr } # Gets Coder onto the instance and runs one script on every boot. It knows @@ -294,6 +250,7 @@ module "amazon_init" { agent_token = try(coder_agent.main[0].token, "") agent_init_script = try(coder_agent.main[0].init_script, "") boot_script = module.nix.boot_script + values = module.nix.values log_display_name = "NixOS" log_icon = "/icon/nix.svg" diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md index b4c689f67..f8503481f 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md @@ -49,7 +49,11 @@ things follow: 2. **Hostname**, from `hostname` or the workspace name. 3. **Logging** — the shell library is written to `$${runtime_dir}/log.sh` and sourced, and the log source is registered. -4. **Identity** — `$${runtime_dir}/workspace.json`, mode 0644, no secrets. +4. **Facts** — `$${runtime_dir}/workspace.json`, mode 0644, no secrets: the + workspace's identity, the deployment URL, the `log_source_id` a service on + the instance needs in order to log anywhere the user will see, and anything + the caller added through `values`. Rendered by Terraform, so a full name + with a quote in it cannot break the document. 5. **Files** — anything in `files`, a map of absolute path to text, written mode 0644 with parent directories created. 6. **Your boot script**, as a child process. diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf index 80982b4cf..b550742ac 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf @@ -86,6 +86,23 @@ variable "files" { } } +variable "values" { + description = <<-EOT + Extra facts to publish in `workspace.json`, merged with the ones this + module writes itself. + + This is the injection point for anything the machine needs to know at + runtime: one map entry rather than a new file and a new variable each + time. Keys this module writes (`workspace`, `owner`, `owner_name`, + `owner_email`, `access_url`, `hostname`, `log_source_id`) win on conflict. + + Not for secrets. The file is world-readable, by design -- an unprivileged + service reads it. + EOT + type = map(string) + default = {} +} + variable "runtime_dir" { description = "Directory for the agent handoff and this module's own state. Must be on a tmpfs: it holds the token." type = string @@ -143,7 +160,25 @@ locals { SH ]) + # Everything the machine is told about itself. Merged so that what this + # module knows wins: a caller cannot accidentally rewrite the workspace's + # own identity through `values`. + facts = merge(var.values, { + workspace = data.coder_workspace.me.name + owner = data.coder_workspace_owner.me.name + owner_name = coalesce(data.coder_workspace_owner.me.full_name, data.coder_workspace_owner.me.name) + owner_email = data.coder_workspace_owner.me.email + access_url = data.coder_workspace.me.access_url + hostname = local.hostname + + # The one thing a configuration on the instance cannot work out for + # itself, and needs in order to log anywhere the user will see. + log_source_id = random_uuid.log_source.result + }) + bootstrap = templatefile("${path.module}/scripts/bootstrap.sh.tftpl", { + FACTS_JSON = jsonencode(local.facts) + LOG_SH = file("${path.module}/scripts/log.sh") FILES_SH = local.files_sh BOOT_SCRIPT = var.boot_script @@ -159,13 +194,7 @@ locals { ARG_LOG_ICON = var.log_icon ARG_LOG_BUDGET = var.log_budget_bytes - ARG_HOSTNAME = local.hostname - ARG_WORKSPACE_NAME = data.coder_workspace.me.name - ARG_OWNER = data.coder_workspace_owner.me.name - # base64 because a full name may contain quotes and is interpolated into - # both a shell string and a JSON document. - ARG_OWNER_NAME_B64 = base64encode(coalesce(data.coder_workspace_owner.me.full_name, data.coder_workspace_owner.me.name)) - ARG_OWNER_EMAIL = data.coder_workspace_owner.me.email + ARG_HOSTNAME = local.hostname }) # EC2 caps user-data at 16 KiB and the script above plus its payloads is diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl index bfb4affb7..f110003e3 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl @@ -19,10 +19,6 @@ LOG_DISPLAY_NAME_B64='${ARG_LOG_DISPLAY_NAME_B64}' LOG_ICON='${ARG_LOG_ICON}' LOG_BUDGET='${ARG_LOG_BUDGET}' HOSTNAME_='${ARG_HOSTNAME}' -WORKSPACE_NAME='${ARG_WORKSPACE_NAME}' -OWNER='${ARG_OWNER}' -OWNER_NAME_B64='${ARG_OWNER_NAME_B64}' -OWNER_EMAIL='${ARG_OWNER_EMAIL}' RUNTIME_DIR='${ARG_RUNTIME_DIR}' @@ -125,7 +121,7 @@ start_agent() { on_error() { local rc=$? - coder_log error "Boot failed before the boot script ran (exit $rc)." || true + coder_log error "Bootstrap failed (exit $rc)." || true start_agent || true exit "$rc" } @@ -133,19 +129,18 @@ trap on_error ERR # --- 3. Facts about this workspace ----------------------------------------- # -# A runtime interface: identity only, mode 0644, no secrets. The token lives -# in agent.env at 0600 and never appears here. - -cat >"$RUNTIME_DIR/workspace.json" <"$RUNTIME_DIR/workspace.json" <<'CODER_AMAZON_INIT_FACTS' +${FACTS_JSON} +CODER_AMAZON_INIT_FACTS chmod 0644 "$RUNTIME_DIR/workspace.json" export CODER_WORKSPACE_FACTS="$RUNTIME_DIR/workspace.json" export CODER_RUNTIME_DIR="$RUNTIME_DIR" @@ -170,5 +165,9 @@ if [ "$boot_rc" -ne 0 ]; then coder_log error "Boot script exited $boot_rc; starting the agent anyway." || true fi -start_agent +# `|| true` so that failing to start the agent does not re-enter the ERR trap, +# which would report the bootstrap as having failed somewhere earlier. On a +# first boot whose configuration does not build there is no agent unit to +# start at all, and that is the honest outcome: nothing to report it with. +start_agent || true exit "$boot_rc" diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh index 44259df45..b0258457f 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh @@ -11,6 +11,7 @@ CODER_LOG_BUDGET="${CODER_LOG_BUDGET:-524288}" CODER_LOG_STATE_DIR="${CODER_LOG_STATE_DIR:-/run/coder}" CODER_LOG_MAX_LINE=2048 +CODER_LOG_FLUSH_SECS="${CODER_LOG_FLUSH_SECS:-5}" # Inherited, so a child process that sources this library can log without # registering the source again -- registration is per workspace build, not per # process, and a child has no way to know whether it already happened. @@ -111,24 +112,52 @@ coder_log() { } # Reads plain lines on stdin and ships them in batches. +# +# Batches are flushed by size, and also by time: a slow producer would +# otherwise sit in the buffer until 50 lines had accumulated, which reads as a +# hung workspace, and anything still buffered when the process is killed is +# simply lost. coder_log_pipe() { - local level="${1:-info}" line batch="" n=0 bytes=0 obj + local level="${1:-info}" line batch="" n=0 bytes=0 obj rc now last [ "$CODER_LOG_READY" = 1 ] || { cat > /dev/null return 0 } + last=$(date +%s) + + while :; do + line="" + # Not `if ! read`: `!` rewrites $? to 0, which turns every timeout into an + # end of input and ends the stream after the first quiet interval. + IFS= read -r -t "$CODER_LOG_FLUSH_SECS" line + rc=$? + + # read(1) returns >128 on timeout; any other failure is end of input, + # possibly with a last line that had no newline. + if [ "$rc" -ne 0 ] && [ "$rc" -le 128 ]; then + if [ -n "$line" ]; then + batch="${batch:+$batch +}$(coder_log_json "$level" "$line")" + fi + break + fi - while IFS= read -r line || [ -n "$line" ]; do - obj=$(coder_log_json "$level" "$line") - batch="${batch:+$batch + if [ -n "$line" ]; then + obj=$(coder_log_json "$level" "$line") + batch="${batch:+$batch }$obj" - n=$((n + 1)) - bytes=$((bytes + ${#obj})) - if [ "$n" -ge 50 ] || [ "$bytes" -ge 32768 ]; then + n=$((n + 1)) + bytes=$((bytes + ${#obj})) + fi + + now=$(date +%s) + if [ "$n" -ge 50 ] || [ "$bytes" -ge 32768 ] \ + || { [ "$n" -gt 0 ] && [ "$((now - last))" -ge "$CODER_LOG_FLUSH_SECS" ]; }; then printf '%s\n' "$batch" | coder_log_send batch="" n=0 bytes=0 + last=$now fi done diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/README.md b/registry/coder-labs/templates/aws-nixos/modules/nix/README.md index dfaaadc59..a6ad75a1a 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/README.md @@ -5,9 +5,12 @@ the running system is out of date, and rebuild it. Nothing here knows about EC2, user-data, or how the instance came to exist. The boot path is a **string** the caller hands to whatever runs scripts on the -machine, and the periodic rebuild is an ordinary `coder_script`. On AWS the -caller is [`../amazon-init`](../amazon-init/README.md), but nothing depends on -that. +machine. On AWS the caller is [`../amazon-init`](../amazon-init/README.md), but +nothing depends on that. + +Keeping the machine current _after_ boot is not this module's job either. That +belongs to the configuration, as `system.autoUpgrade` on a systemd timer the +machine's owner can read and change. ```tf module "nix" { @@ -45,10 +48,9 @@ the checkout, and rebuilds only if the configuration changed: - a checkout on a different branch than requested says so rather than silently building the wrong thing -The `coder_script` does the same later, on `update_schedule`, with -`update_process = boot` (stage for the next restart) or `switch` (apply now). -A `flock` in `state_dir` keeps the two from overlapping; the scheduled run -skips rather than queues. +A `flock` in `state_dir` is the lock everything rebuilding this machine should +take, including the configuration's own upgrade timer, so that two rebuilds +never race for the system profile. ## Logging @@ -77,6 +79,11 @@ Outputs exist for the things a caller genuinely cannot do itself: metadata block — which has to be declared inline on the agent, though what it means for a NixOS machine to be up to date does not belong in a template. +`values` passes straight through to `values` on the bootstrapper, so a caller +has one wire for runtime facts and a Nix-specific fact would have an obvious +home. There are none today: the configuration already knows its checkout, +attribute and directories, because it is what sets them. + ## Moving this to the registry It is a self-contained Terraform module already. Publishing it as diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf index 1ead34ad1..2efdd00f8 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf @@ -1,27 +1,16 @@ -# The flake lifecycle: keep a checkout in sync, decide whether the running -# system is out of date, and rebuild it. +# The flake lifecycle at boot: clone or fast-forward the checkout, decide +# whether the running system is out of date, and rebuild it. # -# Nothing here knows about EC2, user-data or how the instance was started. The -# boot path is exposed as a string for whatever puts scripts on the machine -- -# on AWS that is ../amazon-init -- and the periodic rebuild is an ordinary -# coder_script. +# Nothing here knows about EC2, user-data or how the instance was started -- +# the boot path is exposed as a string for whatever puts scripts on the +# machine, which on AWS is ../amazon-init. +# +# Keeping the machine current *afterwards* is not this module's job either. +# That is `system.autoUpgrade` in the configuration itself, on a systemd timer +# the machine's owner can read and change. terraform { required_version = ">= 1.3" - - required_providers { - coder = { - source = "coder/coder" - version = ">= 2.5" - } - } -} - -data "coder_workspace" "me" {} - -variable "agent_id" { - description = "Agent that runs the periodic rebuild. May be empty while the workspace is stopped." - type = string } variable "flake_ref" { @@ -58,32 +47,18 @@ variable "arch" { } } -variable "update_schedule" { +variable "values" { description = <<-EOT - Cron schedule for the periodic rebuild, or empty to disable it. + Extra facts for the bootstrapper to publish on the instance, passed + straight through to `values` on whatever writes them. - SIX fields with seconds first, in the workspace's timezone. A five field - expression is silently misinterpreted rather than rejected, and - descriptors like `@daily` pass validation then fail on the agent. + Routed through this module so the caller has one wire, and so a + Nix-specific runtime fact has an obvious home. There are none today: the + configuration already knows its own checkout, attribute and directories, + because it is the thing that sets them. EOT - type = string - default = "0 0 4 * * *" - - validation { - condition = var.update_schedule == "" || length(split(" ", trimspace(var.update_schedule))) == 6 - error_message = "update_schedule must be a 6-field cron expression (seconds first), or empty." - } -} - -variable "update_process" { - description = "`switch` applies the new generation immediately; `boot` stages it for the next restart." - type = string - default = "boot" - - validation { - condition = contains(["boot", "switch"], var.update_process) - error_message = "update_process must be boot or switch." - } + type = map(string) + default = {} } variable "flake_dir" { @@ -128,20 +103,9 @@ locals { } } -resource "coder_script" "nixos_rebuild" { - count = var.update_schedule == "" ? 0 : data.coder_workspace.me.start_count - agent_id = var.agent_id - display_name = "NixOS rebuild" - cron = var.update_schedule - # The boot path has already rebuilt by the time the agent exists. - run_on_start = false - start_blocks_login = false - timeout = 3600 - log_path = "${var.log_dir}/coder-script.log" - - script = templatefile("${path.module}/scripts/rebuild.sh.tftpl", merge(local.script_args, { - ARG_UPDATE_PROCESS = var.update_process - })) +output "values" { + description = "Facts to publish on the instance, for the bootstrapper's `values`." + value = var.values } output "boot_script" { diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/rebuild.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/rebuild.sh.tftpl deleted file mode 100644 index c31ea8634..000000000 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/rebuild.sh.tftpl +++ /dev/null @@ -1,66 +0,0 @@ -#!/usr/bin/env bash -# Periodic nixos-rebuild, run by a coder_script on a cron schedule. -# -# coder_script bodies run as the workspace user through its login shell and -# there is no run_as argument, so everything privileged goes through sudo. -# Output is captured by the agent and shown under this script's own name, so -# it goes to stdout rather than the boot log source. - -set -euo pipefail - -FLAKE_URL='${ARG_FLAKE_URL}' -FLAKE_BRANCH='${ARG_FLAKE_BRANCH}' -FLAKE_ATTR='${ARG_FLAKE_ATTR}' -OPERATION='${ARG_UPDATE_PROCESS}' - -FLAKE_DIR='${ARG_FLAKE_DIR}' -STATE_DIR='${ARG_STATE_DIR}' -LOG_DIR='${ARG_LOG_DIR}' - -NIX_FLAKE_DIR="$FLAKE_DIR" -NIX_FLAKE_ATTR="$FLAKE_ATTR" -NIX_STATE_DIR="$STATE_DIR" -NIX_LOG_DIR="$LOG_DIR" - -${LIFECYCLE_SH} - -# Don't queue: if the boot script or a previous tick still holds the lock, -# doing nothing is correct. coder_script does not prevent overlapping cron -# runs, so this is the only thing that does. -if ! nix_lock nowait; then - echo "Another nixos-rebuild is in progress; skipping this run." - exit 0 -fi - -TRANSCRIPT="$LOG_DIR/rebuild-$(date -u +%Y%m%dT%H%M%SZ).log" - -nix_sync_checkout "$FLAKE_URL" "$FLAKE_BRANCH" - -REV=$(nix_needs_rebuild) && NEEDS_REBUILD=1 || NEEDS_REBUILD=0 -if [ "$NEEDS_REBUILD" -eq 0 ]; then - echo "Already up to date ($${REV:0:12})." - exit 0 -fi - -echo "Running nixos-rebuild $OPERATION for $FLAKE_DIR#$FLAKE_ATTR ($${REV:0:12})" -echo "Transcript: $TRANSCRIPT" - -if ! nix_apply "$OPERATION" "$TRANSCRIPT"; then - echo "nixos-rebuild $OPERATION failed; the running system is untouched." >&2 - echo "See $TRANSCRIPT." >&2 - exit 1 -fi -nix_record_rev "$REV" - -case "$OPERATION" in - boot) - if nix_pending_generation; then - echo "A new system generation is staged and applies on the next restart." - fi - ;; - switch) - # The agent survives because coder-agent.service is declared with - # restartIfChanged = false. - echo "New system generation is active." - ;; -esac From cfd05d4b05c6c016c17d080e3c06915d8e863104 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Wed, 23 Sep 2026 20:29:38 +0000 Subject: [PATCH 08/24] fix(aws-nixos): make a boot that could not build the flake visible A first boot whose flake reference is wrong left nothing behind: the clone failed, no generation was ever built, no coder-agent.service existed, and the workspace sat in "starting" until the twenty-minute connection timeout with three lines in a log source nobody has a reason to open. There is no way to report that over the API. `report-lifecycle` was removed from coderd in v2.13, lifecycle is now dRPC-only over /me/rpc, and a deployment renders an agent's state only once the agent has connected -- so an instance with no agent has no state to show, whatever it posts. What a workspace does show is the exit status of a startup script, and that needs an agent. So: - When the boot script has not produced a coder-agent.service, the bootstrap runs coder_agent.init_script itself, as root, under a transient coder-agent-fallback.service. It is the AMI's environment and not the configuration that was asked for, which is the point: the workspace connects, the error is on screen, and there is a terminal to fix the flake from. `fallback_agent = false` turns it off. - A failed boot writes $${runtime_dir}/boot-failed with a sentence about what happened, removed at the start of every boot, and the agent's startup script fails on it. The agent then reports start_error and the workspace is unhealthy with "agent startup script exited with an error" -- true both on a first boot and on a later one that kept the previous generation. - The agent gets a troubleshooting_url, which is what the timeout tooltip links to when there is genuinely no agent (no route to the internet). Three bugs made the original report as empty as it was: - `git clone ... | nix_filter_log` under `set -e` ended the boot at the pipeline, before the PIPESTATUS check two lines down, so "Could not clone" was unreachable. The same shape in the fetch path meant an offline restart died instead of building the existing checkout. Both go through nix_run_logged now, which returns the command's own status -- and routes the command's output through nix_log, so git's own explanation reaches the workspace instead of the instance's journal. - Neither script set errtrace, so the ERR traps were not inherited by the functions where the work happens and never ran. - `set +e` around the boot script does not disable the ERR trap, which is not subject to errexit; the trap fired, reported "Bootstrap failed", and skipped the more specific message below it. It is a `|| boot_rc=$?` list now. user-data was at 17575 bytes of 16384 with the above, so the init script is written as a plain heredoc rather than base64: it is the largest thing in there, and base64 inflates by a third and destroys the redundancy the gzip wrapper lives on. Same reasoning as the log library. Verified on a live deployment, both paths: a bad flake reference now connects and reports start_error with git's own error in the log, and a good one still switches and starts the real agent. --- .../coder-labs/templates/aws-nixos/README.md | 22 ++++- .../coder-labs/templates/aws-nixos/main.tf | 36 +++++++ .../aws-nixos/modules/amazon-init/README.md | 26 +++++ .../aws-nixos/modules/amazon-init/main.tf | 45 +++++++-- .../amazon-init/scripts/bootstrap.sh.tftpl | 94 +++++++++++++++++-- .../modules/nix/scripts/boot.sh.tftpl | 6 +- .../modules/nix/scripts/lifecycle.sh | 35 +++++-- 7 files changed, 235 insertions(+), 29 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 4cffa3376..b4a31bdd3 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -244,12 +244,28 @@ until `nixos-rebuild switch` finishes. Watch the "NixOS" log source. A cold clos instance can take ten minutes or more. A restart with nothing to rebuild skips straight to starting the agent. +### The workspace says its startup script failed + +The configuration did not apply. The machine is up and reachable, but it is **not** running the +system it was built from — `nixos-rebuild switch` builds before it activates, so what is running is +whatever ran before: the previous generation, or on a first boot the bare AMI. + +The bootstrap records the reason in `/run/coder/boot-failed`, and the agent's startup script fails +on it, which is the only way a workspace will show an error for something that went wrong before +the agent existed. Read the "NixOS" log source for what actually happened, fix the flake, and +restart the workspace. + +On a first boot there is no agent in the AMI to start at all, so the bootstrap runs one itself as +root under `coder-agent-fallback.service`. That is why the workspace opens even though nothing was +built — the terminal is there so the flake can be fixed from inside. It has none of the packages, +users or services the configuration asks for. + ### The agent never connects The boot script writes its handoff to `/run/coder` before doing anything else, so the usual cause is -a failed rebuild — or, if there are no logs at all, an instance with no route to the internet (a -NixOS workspace fetches its own configuration on boot, so it needs egress before it can report -anything). The workspace metadata shows the instance id; the AMI logs to the serial console, which +an instance with no route to the internet (a NixOS workspace fetches its own configuration on boot, +so it needs egress before it can report anything) — a failed rebuild connects anyway and reports +the failure. The workspace metadata shows the instance id; the AMI logs to the serial console, which needs no SSH: ```console diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index 9388ed71a..a989061ad 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -49,6 +49,16 @@ variable "nixos_release" { default = "26.05" } +variable "troubleshooting_url" { + description = <<-EOT + Page the agent's "Troubleshoot" link points at. Shown when the agent has + not connected within `connection_timeout`, which on this template means + the instance never finished applying the flake. + EOT + type = string + default = "https://registry.coder.com/templates/coder-labs/aws-nixos" +} + data "coder_parameter" "instance_type" { name = "instance_type" display_name = "Instance type" @@ -133,6 +143,26 @@ resource "coder_agent" "main" { # The first boot completes a nixos-rebuild switch before the agent exists, # so the default 120s looks like a failed workspace. connection_timeout = 1200 + # Shown on the agent as a "Troubleshoot" link once it times out, which is + # the one moment the user has nothing else to go on. + troubleshooting_url = var.troubleshooting_url + + # The instance can come up perfectly while the configuration it was meant + # to run does not build, and a workspace has no way to say so: a deployment + # shows an agent's state only once it has connected, and nothing a script + # can call fails a build. What it does report is the startup script's exit + # status -- so the bootstrap leaves a file behind and this fails on it. + # See modules/amazon-init/README.md. + startup_script = <<-EOT + #!/usr/bin/env bash + set -euo pipefail + + failure='${local.runtime_dir}/boot-failed' + if [ -f "$failure" ]; then + cat "$failure" >&2 + exit 1 + fi + EOT metadata { key = "cpu" @@ -228,6 +258,11 @@ locals { "m7g.xlarge" = { agent = "arm64", ami = "arm64", attr = "aarch64" } } arch = local.arch_map[data.coder_parameter.instance_type.value] + + # Named here rather than read back from the amazon-init module: that module + # is given the agent's token, so an output of it cannot be used to configure + # the agent without making a cycle. + runtime_dir = "/run/coder" } # Everything about the flake: the checkout, the boot-time rebuild and the @@ -251,6 +286,7 @@ module "amazon_init" { agent_init_script = try(coder_agent.main[0].init_script, "") boot_script = module.nix.boot_script values = module.nix.values + runtime_dir = local.runtime_dir log_display_name = "NixOS" log_icon = "/icon/nix.svg" diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md index f8503481f..323e3ace2 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md @@ -71,6 +71,32 @@ The mirror image of that rule is that a boot script failure must not leave an unreachable workspace, so the agent is started even when the boot script exits non-zero. Its status is reported and propagated, never suppressed. +## Saying that the boot failed + +A failed boot script has nothing to report itself with. A deployment renders +an agent's state only once that agent has connected, and no API a script can +reach marks a build failed — so an instance that never starts an agent is +indistinguishable from a slow one until `connection_timeout` expires, and the +reason sits in a log source nobody has a reason to open. + +Two things follow from that, and both are why a failure here is visible at +all: + +- **`boot-failed`.** When the boot script exits non-zero the module writes + `$${runtime_dir}/boot-failed` (mode 0644) with a sentence about what + happened, and removes it at the start of every boot. Read it from the + agent's `startup_script` and exit non-zero — the startup script's exit + status is the only failure state a workspace will show you. The path is the + `boot_failed_path` output. +- **The fallback agent.** When the boot script has not produced a + `coder-agent.service` at all — a first boot whose configuration did not + build — the module runs `coder_agent.init_script` itself, as root, under a + transient `coder-agent-fallback.service`. It is the image's own + environment and not the system that was asked for, which is the point: the + workspace connects, the error is on screen, and there is a terminal to fix + it from. `systemctl stop coder-agent-fallback` happens as soon as a real + unit exists. Set `fallback_agent = false` to turn it off. + ## What the boot script gets Root, a child process, and these: diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf index b550742ac..bf96398cf 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf @@ -140,6 +140,26 @@ variable "log_budget_bytes" { default = 524288 } +variable "fallback_agent" { + description = <<-EOT + Start the agent from `coder_agent.init_script` when the boot script has + not produced a `coder-agent.service`. + + A deployment shows an agent's state only once it has connected, and there + is no API a script can call to fail a build, so an instance that never + starts an agent is indistinguishable from a slow one until the connection + timeout expires. The fallback runs as root under a transient unit, in the + image's own environment rather than the one the boot script was meant to + build: it exists so that a workspace whose configuration is broken can + still be opened, read and repaired. + + Turn it off for images where an agent running as root is not acceptable. + The failure is still logged, and `boot-failed` is still written. + EOT + type = bool + default = true +} + variable "hostname" { description = "Hostname to set on the instance. Defaults to the workspace name." type = string @@ -182,19 +202,20 @@ locals { LOG_SH = file("${path.module}/scripts/log.sh") FILES_SH = local.files_sh BOOT_SCRIPT = var.boot_script + INIT_SCRIPT = var.agent_init_script - ARG_ACCESS_URL = data.coder_workspace.me.access_url - ARG_AGENT_TOKEN = var.agent_token - ARG_INIT_SCRIPT_B64 = base64encode(var.agent_init_script) - ARG_RUNTIME_DIR = var.runtime_dir - ARG_PATH = var.path + ARG_ACCESS_URL = data.coder_workspace.me.access_url + ARG_AGENT_TOKEN = var.agent_token + ARG_RUNTIME_DIR = var.runtime_dir + ARG_PATH = var.path ARG_LOG_SOURCE_ID = random_uuid.log_source.result ARG_LOG_DISPLAY_NAME_B64 = base64encode(var.log_display_name) ARG_LOG_ICON = var.log_icon ARG_LOG_BUDGET = var.log_budget_bytes - ARG_HOSTNAME = local.hostname + ARG_HOSTNAME = local.hostname + ARG_FALLBACK_AGENT = tostring(var.fallback_agent) }) # EC2 caps user-data at 16 KiB and the script above plus its payloads is @@ -248,6 +269,18 @@ output "workspace_facts_path" { value = "${var.runtime_dir}/workspace.json" } +output "boot_failed_path" { + description = <<-EOT + File written when the boot script fails, holding a sentence about what + happened. Absent on a healthy boot. + + Read it from the agent's startup script and exit non-zero: that is the + only way to get an error out of a workspace whose machine came up fine + but whose configuration did not. + EOT + value = "${var.runtime_dir}/boot-failed" +} + output "bootstrap_path" { description = "Where the user-data wrapper extracts the real boot script." value = "${var.runtime_dir}/bootstrap.sh" diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl index f110003e3..805fdf692 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl @@ -7,18 +7,21 @@ # handoff, publishes the workspace's identity, runs one caller-supplied boot # script, and then starts the agent -- in that order, always. -set -euo pipefail +# -E so that the ERR trap installed below is inherited by the functions here; +# without errtrace a failure inside one of them ends this script with nothing +# said about it. +set -Eeuo pipefail export PATH="${ARG_PATH}:$PATH" export HOME=/root ACCESS_URL='${ARG_ACCESS_URL}' AGENT_TOKEN='${ARG_AGENT_TOKEN}' -INIT_SCRIPT_B64='${ARG_INIT_SCRIPT_B64}' LOG_SOURCE_ID='${ARG_LOG_SOURCE_ID}' LOG_DISPLAY_NAME_B64='${ARG_LOG_DISPLAY_NAME_B64}' LOG_ICON='${ARG_LOG_ICON}' LOG_BUDGET='${ARG_LOG_BUDGET}' HOSTNAME_='${ARG_HOSTNAME}' +FALLBACK_AGENT='${ARG_FALLBACK_AGENT}' RUNTIME_DIR='${ARG_RUNTIME_DIR}' @@ -32,13 +35,18 @@ RUNTIME_DIR='${ARG_RUNTIME_DIR}' # the point: the token rotates on every workspace start. install -d -m 0700 "$RUNTIME_DIR" -rm -f "$RUNTIME_DIR/ready" +rm -f "$RUNTIME_DIR/ready" "$RUNTIME_DIR/boot-failed" umask 077 printf 'CODER_AGENT_TOKEN=%s\nCODER_AGENT_URL=%s\n' "$AGENT_TOKEN" "$ACCESS_URL" \ >"$RUNTIME_DIR/agent.env" umask 022 -printf '%s' "$INIT_SCRIPT_B64" | base64 -d >"$RUNTIME_DIR/init.sh" +# Verbatim rather than base64: user-data has 16 KiB to fit in, the init +# script is the largest thing in it, and base64 both inflates by a third and +# destroys the redundancy the gzip wrapper lives on. +cat >"$RUNTIME_DIR/init.sh" <<'CODER_AMAZON_INIT_AGENT_SCRIPT' +${INIT_SCRIPT} +CODER_AMAZON_INIT_AGENT_SCRIPT chmod 0755 "$RUNTIME_DIR/init.sh" # coder-agent.service may run as an unprivileged user, so it has to be able to @@ -88,6 +96,53 @@ export CODER_LOG_LIBRARY="$RUNTIME_DIR/log.sh" coder_log_init "$(printf '%s' "$LOG_DISPLAY_NAME_B64" | base64 -d)" "$LOG_ICON" || true +# Nothing on the instance can report a failure except a connected agent. A +# deployment renders an agent's lifecycle only while it is connected, and +# there is no API a script can call to fail a build -- so an instance that +# never starts an agent shows "connecting" until the connection timeout +# expires and then "timeout", with the reason buried in a log source nobody +# has a reason to open. +# +# So when the boot script never produced a coder-agent.service -- a first boot +# whose configuration did not build -- start the agent from the init script +# instead, as root, under a transient unit. It runs in the image's own +# environment rather than the configuration that was asked for, which is +# exactly the point: the workspace comes up, the error is on screen, and there +# is a terminal to fix it from. +start_fallback_agent() { + [ "$FALLBACK_AGENT" = "true" ] || return 1 + command -v systemd-run >/dev/null 2>&1 || return 1 + + coder_log error "Starting an agent from the image instead, so that this workspace can still be opened." || true + coder_log error "It is NOT running the configuration you asked for. Fix the configuration and restart the workspace." || true + + # A fallback left running from an earlier attempt would fight the real agent + # over the same token, and a failed one keeps the unit name. + systemctl reset-failed coder-agent-fallback.service 2>/dev/null || true + install -d -m 0755 "$RUNTIME_DIR/bin" + + # The token is read from agent.env inside the unit rather than passed as a + # property, so that it does not end up in `systemctl show` output. + if ! systemd-run --quiet --collect --unit=coder-agent-fallback \ + --service-type=exec \ + --setenv=HOME=/root \ + --setenv=PATH="$PATH" \ + --setenv=BINARY_DIR="$RUNTIME_DIR/bin" \ + /usr/bin/env bash -c \ + "set -a; . '$RUNTIME_DIR/agent.env'; set +a; exec bash '$RUNTIME_DIR/init.sh'"; then + coder_log error "Could not start the fallback agent." || true + return 1 + fi + + sleep 3 + if systemctl is-active --quiet coder-agent-fallback; then + return 0 + fi + coder_log error "The fallback agent exited immediately." || true + journalctl -u coder-agent-fallback --no-pager --lines=30 2>/dev/null | coder_log_pipe error || true + return 1 +} + # Nothing starts coder-agent.service but this, and only once the boot script # has finished. If systemd started it at multi-user.target instead, the agent # would report the workspace ready and run its startup scripts while the boot @@ -100,9 +155,12 @@ start_agent() { fi if ! systemctl cat coder-agent >/dev/null 2>&1; then coder_log error "coder-agent.service does not exist; the boot script has never created it." || true - return 1 + start_fallback_agent || return 1 + return 0 fi + # The real unit wins whenever it exists. + systemctl stop coder-agent-fallback.service 2>/dev/null || true chown_handoff systemctl start coder-agent || true @@ -119,9 +177,20 @@ start_agent() { return 1 } +# Left for the agent to find. A connected agent is the only thing a +# deployment shows a state for, and the only state a script can choose is the +# one its startup script exits with -- so the failure is recorded here and the +# agent's startup script reads it back. Without that the workspace comes up +# looking perfectly healthy while running the wrong system. +record_failure() { + printf '%s\n' "$*" >"$RUNTIME_DIR/boot-failed" 2>/dev/null || true + chmod 0644 "$RUNTIME_DIR/boot-failed" 2>/dev/null || true +} + on_error() { local rc=$? coder_log error "Bootstrap failed (exit $rc)." || true + record_failure "The instance's boot script failed (exit $rc) before it could apply this workspace's configuration. See the boot logs for what went wrong." start_agent || true exit "$rc" } @@ -156,18 +225,23 @@ chmod 0700 "$RUNTIME_DIR/boot.sh" # Run as a child rather than sourced: the boot script is someone else's code, # and it must not be able to exit this one before the agent is started. -set +e -bash "$RUNTIME_DIR/boot.sh" -boot_rc=$? -set -e +# +# `|| boot_rc=$?` rather than turning errexit off around it: `set +e` stops +# the shell exiting but does nothing to the ERR trap, which is not subject to +# errexit at all. The trap would fire here and report this as a bootstrap +# failure, skipping everything below. In a `||` list neither happens. +boot_rc=0 +bash "$RUNTIME_DIR/boot.sh" || boot_rc=$? if [ "$boot_rc" -ne 0 ]; then coder_log error "Boot script exited $boot_rc; starting the agent anyway." || true + record_failure "This workspace's configuration could not be applied: the boot script exited $boot_rc. The machine is NOT running the configuration it was built from. See the boot logs for what went wrong." fi # `|| true` so that failing to start the agent does not re-enter the ERR trap, # which would report the bootstrap as having failed somewhere earlier. On a # first boot whose configuration does not build there is no agent unit to -# start at all, and that is the honest outcome: nothing to report it with. +# start at all; start_agent falls back to the init script so that the +# workspace is still reachable. start_agent || true exit "$boot_rc" diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl index 25e143ddd..00ba18a0c 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl @@ -8,7 +8,11 @@ # # Runs on every boot, so it must be idempotent. -set -euo pipefail +# -E, because most of what runs here runs inside a function: without errtrace +# the ERR trap below is not inherited, and a failure in the lifecycle library +# ends the script silently -- the exit status reaches the caller, but nothing +# is ever said about it in the workspace log. +set -Eeuo pipefail FLAKE_URL='${ARG_FLAKE_URL}' FLAKE_BRANCH='${ARG_FLAKE_BRANCH}' diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh index 0af2634f3..08e9c7628 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh @@ -19,6 +19,26 @@ nix_strip_ansi() { sed -e 's/\x1b\[[0-9;]*[a-zA-Z]//g' -e 's/\r$//' } +# Runs a command with its output filtered into the log, and returns the +# command's own status rather than the filter's. +# +# Not written as a pipeline: under `set -e` a failing pipeline ends the script +# there and then, before the caller can look at PIPESTATUS. That is how a +# clone of a flake reference that does not exist used to end a boot with +# nothing in the workspace log but "Cloning ..." -- the error message two +# lines further down was unreachable. +nix_run_logged() { + local rc=0 out line + out="$(mktemp)" + "$@" > "$out" 2>&1 || rc=$? + # Through nix_log rather than stdout: what git has to say about a clone it + # could not do is the whole explanation, and stdout here is the instance's + # journal, which is exactly the place the user cannot reach. + nix_filter_log < "$out" | while IFS= read -r line; do nix_log info "$line"; done + rm -f "$out" + return "$rc" +} + # --------------------------------------------------------------------------- # the checkout # --------------------------------------------------------------------------- @@ -68,17 +88,15 @@ nix_sync_checkout() { # to clone into a non-empty one. local tmp tmp="$($NIX_SUDO mktemp -d)" - # PIPESTATUS, not the pipeline's status: that is the filter's, and the - # filter always succeeds. Without this a failed clone reads as a success - # and the next step fails somewhere much less obvious. + rc=0 if [ -n "$branch" ]; then - $NIX_SUDO git clone --branch "$branch" "$url" "$tmp/repo" 2>&1 | nix_filter_log + nix_run_logged $NIX_SUDO git clone --branch "$branch" "$url" "$tmp/repo" || rc=$? else - $NIX_SUDO git clone "$url" "$tmp/repo" 2>&1 | nix_filter_log + nix_run_logged $NIX_SUDO git clone "$url" "$tmp/repo" || rc=$? fi - rc=${PIPESTATUS[0]} if [ "$rc" -ne 0 ]; then - nix_log error "Could not clone $url${branch:+ (branch $branch)}" + nix_log error "Could not clone $url${branch:+ (branch $branch)} (git exited $rc)" + nix_log error "Check the flake reference the template was pushed with: the repository has to exist and be readable from this instance." $NIX_SUDO rm -rf "$tmp" return 1 fi @@ -108,8 +126,7 @@ nix_sync_checkout() { fi # Fetching as root writes into .git, so re-assert ownership afterwards. - $NIX_SUDO git -C "$NIX_FLAKE_DIR" fetch --quiet origin "$branch" 2>&1 | nix_filter_log - if [ "${PIPESTATUS[0]}" -ne 0 ]; then + if ! nix_run_logged $NIX_SUDO git -C "$NIX_FLAKE_DIR" fetch --quiet origin "$branch"; then nix_log warn "Could not reach the remote; building the existing checkout" return 0 fi From a415ef5fb7fd185b2bbd6a810922e6c0d61d405d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Wed, 23 Sep 2026 21:27:14 +0000 Subject: [PATCH 09/24] fix(aws-nixos): report the failure with the agent instead of keeping one The fallback agent left a workspace that looked like it worked: a root agent serving the AMI's environment, with a terminal, an IDE and none of the packages, users or services the configuration asks for. It only ever existed because a deployment shows an agent's state once that agent has connected, so the agent is the only thing on the instance that can report anything. That does not mean it has to stay: `coder agent` connects, runs its startup scripts -- which fail on boot-failed -- and the state is on the deployment from then on, whether the process is still there or not. So the run is now bounded. The agent comes up under coder-agent-report.service, gets report_failure_timeout seconds (60) to connect and report, and is killed. The workspace is failed a minute into the boot rather than twenty, and has no agent, which is the truth: there is nothing on that instance that the workspace was asked to provide. Killed rather than stopped, because on SIGTERM the agent reports shutting_down and then off, over the top of the start_error it was started to report. KillSignal and RuntimeMaxSec on the transient unit so that holds even if the bootstrap is killed first. Measured on a live deployment: connected at +2s, start_error at +9s, gone at +62s, lifecycle start_error with the startup script recorded exit 1. --- .../coder-labs/templates/aws-nixos/README.md | 12 +-- .../aws-nixos/modules/amazon-init/README.md | 19 +++-- .../aws-nixos/modules/amazon-init/main.tf | 24 ++++-- .../amazon-init/scripts/bootstrap.sh.tftpl | 75 +++++++++++-------- 4 files changed, 77 insertions(+), 53 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index b4a31bdd3..57efbe51a 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -255,17 +255,17 @@ on it, which is the only way a workspace will show an error for something that w the agent existed. Read the "NixOS" log source for what actually happened, fix the flake, and restart the workspace. -On a first boot there is no agent in the AMI to start at all, so the bootstrap runs one itself as -root under `coder-agent-fallback.service`. That is why the workspace opens even though nothing was -built — the terminal is there so the flake can be fixed from inside. It has none of the packages, -users or services the configuration asks for. +On a first boot there is no agent in the AMI to run that script at all, so the bootstrap starts one +itself for a minute, lets it report, and kills it. The workspace is therefore failed rather than +"starting", but it has no agent: there is no terminal, no SSH and no IDE until the flake is fixed +and the workspace restarted. Use the serial console or SSM below to get inside it in the meantime. ### The agent never connects The boot script writes its handoff to `/run/coder` before doing anything else, so the usual cause is an instance with no route to the internet (a NixOS workspace fetches its own configuration on boot, -so it needs egress before it can report anything) — a failed rebuild connects anyway and reports -the failure. The workspace metadata shows the instance id; the AMI logs to the serial console, which +so it needs egress before it can report anything) — a failed rebuild reports itself failed instead +of hanging. The workspace metadata shows the instance id; the AMI logs to the serial console, which needs no SSH: ```console diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md index 323e3ace2..a12489d20 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md @@ -88,14 +88,19 @@ all: agent's `startup_script` and exit non-zero — the startup script's exit status is the only failure state a workspace will show you. The path is the `boot_failed_path` output. -- **The fallback agent.** When the boot script has not produced a +- **A reporting run of the agent.** When the boot script has not produced a `coder-agent.service` at all — a first boot whose configuration did not - build — the module runs `coder_agent.init_script` itself, as root, under a - transient `coder-agent-fallback.service`. It is the image's own - environment and not the system that was asked for, which is the point: the - workspace connects, the error is on screen, and there is a terminal to fix - it from. `systemctl stop coder-agent-fallback` happens as soon as a real - unit exists. Set `fallback_agent = false` to turn it off. + build — nothing is left to run that startup script, so the module runs + `coder_agent.init_script` itself for `report_failure_timeout` seconds under + a transient `coder-agent-report.service`, and then kills it. The agent + connects, its startup scripts fail on `boot-failed`, and the workspace is + failed within the minute instead of after the connection timeout. + + It is killed, not stopped: on SIGTERM the agent reports `shutting_down` and + then `off`, over the top of the `start_error` it was started to report. And + it is not left running, because it would be serving the image's environment + rather than the configuration that was asked for — a workspace that looks + like it works. Set `report_failure = false` to turn it off. ## What the boot script gets diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf index bf96398cf..f149417d3 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf @@ -140,18 +140,19 @@ variable "log_budget_bytes" { default = 524288 } -variable "fallback_agent" { +variable "report_failure" { description = <<-EOT - Start the agent from `coder_agent.init_script` when the boot script has - not produced a `coder-agent.service`. + When the boot script has not produced a `coder-agent.service`, run + `coder_agent.init_script` for `report_failure_timeout` seconds and then + kill it. A deployment shows an agent's state only once it has connected, and there is no API a script can call to fail a build, so an instance that never starts an agent is indistinguishable from a slow one until the connection - timeout expires. The fallback runs as root under a transient unit, in the - image's own environment rather than the one the boot script was meant to - build: it exists so that a workspace whose configuration is broken can - still be opened, read and repaired. + timeout expires. Running the agent briefly is the only way to say + otherwise: it connects, its startup scripts fail on `boot-failed`, and the + workspace is failed within the minute. The agent is then killed rather + than stopped, because a clean shutdown reports `off` over the top of it. Turn it off for images where an agent running as root is not acceptable. The failure is still logged, and `boot-failed` is still written. @@ -160,6 +161,12 @@ variable "fallback_agent" { default = true } +variable "report_failure_timeout" { + description = "Seconds to leave that agent running. Long enough for it to connect and run every startup script; nothing on the instance can read the state back to know." + type = number + default = 60 +} + variable "hostname" { description = "Hostname to set on the instance. Defaults to the workspace name." type = string @@ -215,7 +222,8 @@ locals { ARG_LOG_BUDGET = var.log_budget_bytes ARG_HOSTNAME = local.hostname - ARG_FALLBACK_AGENT = tostring(var.fallback_agent) + ARG_REPORT_FAILURE = tostring(var.report_failure) + ARG_REPORT_TIMEOUT = var.report_failure_timeout }) # EC2 caps user-data at 16 KiB and the script above plus its payloads is diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl index 805fdf692..494fe2196 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl @@ -21,7 +21,8 @@ LOG_DISPLAY_NAME_B64='${ARG_LOG_DISPLAY_NAME_B64}' LOG_ICON='${ARG_LOG_ICON}' LOG_BUDGET='${ARG_LOG_BUDGET}' HOSTNAME_='${ARG_HOSTNAME}' -FALLBACK_AGENT='${ARG_FALLBACK_AGENT}' +REPORT_FAILURE='${ARG_REPORT_FAILURE}' +REPORT_TIMEOUT='${ARG_REPORT_TIMEOUT}' RUNTIME_DIR='${ARG_RUNTIME_DIR}' @@ -96,51 +97,61 @@ export CODER_LOG_LIBRARY="$RUNTIME_DIR/log.sh" coder_log_init "$(printf '%s' "$LOG_DISPLAY_NAME_B64" | base64 -d)" "$LOG_ICON" || true -# Nothing on the instance can report a failure except a connected agent. A -# deployment renders an agent's lifecycle only while it is connected, and -# there is no API a script can call to fail a build -- so an instance that -# never starts an agent shows "connecting" until the connection timeout -# expires and then "timeout", with the reason buried in a log source nobody -# has a reason to open. +# Nothing on the instance can report a failure except the agent. A deployment +# renders an agent's state only once it has connected, and there is no API a +# script can call to fail a build -- so an instance that never starts an agent +# shows "connecting" until the connection timeout expires twenty minutes +# later, with the reason buried in a log source nobody has a reason to open. # -# So when the boot script never produced a coder-agent.service -- a first boot -# whose configuration did not build -- start the agent from the init script -# instead, as root, under a transient unit. It runs in the image's own -# environment rather than the configuration that was asked for, which is -# exactly the point: the workspace comes up, the error is on screen, and there -# is a terminal to fix it from. -start_fallback_agent() { - [ "$FALLBACK_AGENT" = "true" ] || return 1 +# `coder agent` can say it, though, and that is all this does: run the agent +# from the init script, let it connect and run its startup scripts -- which +# fail on the boot-failed file written above -- and then kill it. The +# workspace is failed within a minute instead of twenty, and the agent is +# gone: left running it would be serving the image's environment rather than +# the configuration that was asked for, which is a workspace that looks like +# it works. +# +# Killed rather than stopped: on SIGTERM the agent reports shutting_down and +# then off, which would overwrite the start_error it was started to report. +report_boot_failure() { + [ "$REPORT_FAILURE" = "true" ] || return 1 command -v systemd-run >/dev/null 2>&1 || return 1 - coder_log error "Starting an agent from the image instead, so that this workspace can still be opened." || true - coder_log error "It is NOT running the configuration you asked for. Fix the configuration and restart the workspace." || true + coder_log error "Starting the agent briefly to report the failure, then stopping it." || true - # A fallback left running from an earlier attempt would fight the real agent - # over the same token, and a failed one keeps the unit name. - systemctl reset-failed coder-agent-fallback.service 2>/dev/null || true + # A run left over from an earlier attempt would fight this one over the same + # token, and a failed one keeps the unit name. + systemctl reset-failed coder-agent-report.service 2>/dev/null || true install -d -m 0755 "$RUNTIME_DIR/bin" # The token is read from agent.env inside the unit rather than passed as a # property, so that it does not end up in `systemctl show` output. - if ! systemd-run --quiet --collect --unit=coder-agent-fallback \ + # + # RuntimeMaxSec and KillSignal so that the agent still goes away -- without + # reporting itself off on the way out -- if this script is killed before it + # gets to do it itself. + if ! systemd-run --quiet --collect --unit=coder-agent-report \ --service-type=exec \ + --property=RuntimeMaxSec="$((REPORT_TIMEOUT + 60))" \ + --property=KillSignal=SIGKILL \ + --property=FinalKillSignal=SIGKILL \ --setenv=HOME=/root \ --setenv=PATH="$PATH" \ --setenv=BINARY_DIR="$RUNTIME_DIR/bin" \ /usr/bin/env bash -c \ "set -a; . '$RUNTIME_DIR/agent.env'; set +a; exec bash '$RUNTIME_DIR/init.sh'"; then - coder_log error "Could not start the fallback agent." || true + coder_log error "Could not start the agent to report the failure." || true return 1 fi - sleep 3 - if systemctl is-active --quiet coder-agent-fallback; then - return 0 - fi - coder_log error "The fallback agent exited immediately." || true - journalctl -u coder-agent-fallback --no-pager --lines=30 2>/dev/null | coder_log_pipe error || true - return 1 + # Long enough for it to connect, run the startup scripts and report what + # they did. There is nothing to wait on: the state it is reporting lives on + # the deployment, and no endpoint an agent token can reach reads it back. + sleep "$REPORT_TIMEOUT" + systemctl stop coder-agent-report.service 2>/dev/null || true + + coder_log error "Reported. This workspace is failed and has no agent; fix the configuration and restart it." || true + return 0 } # Nothing starts coder-agent.service but this, and only once the boot script @@ -155,12 +166,12 @@ start_agent() { fi if ! systemctl cat coder-agent >/dev/null 2>&1; then coder_log error "coder-agent.service does not exist; the boot script has never created it." || true - start_fallback_agent || return 1 + report_boot_failure || return 1 return 0 fi - # The real unit wins whenever it exists. - systemctl stop coder-agent-fallback.service 2>/dev/null || true + # A reporting run from an earlier attempt must not fight the real agent. + systemctl stop coder-agent-report.service 2>/dev/null || true chown_handoff systemctl start coder-agent || true From bb8cf7ec7d7fcf833538f1b766a19809665440cc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Thu, 24 Sep 2026 10:09:48 +0000 Subject: [PATCH 10/24] revert(aws-nixos): stop reporting boot failures through the agent Reverts the failure-surfacing half of cfd05d4b and a415ef5f: the transient coder-agent-report run, the /run/coder/boot-failed sentinel, the agent startup script that failed on it, and troubleshooting_url. A first boot that cannot build the flake is back to what it was -- no agent, no state, "connecting" until connection_timeout -- with the error in the NixOS log source. Kept, because they are what actually made the reported failure legible and have nothing to do with the agent: - nix_run_logged, and the clone and fetch going through it. The old `git clone ... | nix_filter_log` ended the boot at the pipeline under `set -e`, before the PIPESTATUS check two lines down, so "Could not clone" was unreachable and an offline restart died instead of building the existing checkout. It also routes the command's own output through nix_log, which is what puts git's `fatal:` line in the workspace. - errtrace in both scripts, so the ERR traps are inherited by the functions the work happens in. - `|| boot_rc=$?` instead of `set +e` around the boot script: `set +e` does not disable an ERR trap, so the trap fired and reported "Bootstrap failed" over the more specific message below it. - The init script as a plain heredoc rather than base64, worth about 4 KiB of the 16 KiB user-data budget. --- .../coder-labs/templates/aws-nixos/README.md | 22 +---- .../coder-labs/templates/aws-nixos/main.tf | 36 --------- .../aws-nixos/modules/amazon-init/README.md | 31 ------- .../aws-nixos/modules/amazon-init/main.tf | 43 +--------- .../amazon-init/scripts/bootstrap.sh.tftpl | 81 +------------------ 5 files changed, 7 insertions(+), 206 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 57efbe51a..4cffa3376 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -244,28 +244,12 @@ until `nixos-rebuild switch` finishes. Watch the "NixOS" log source. A cold clos instance can take ten minutes or more. A restart with nothing to rebuild skips straight to starting the agent. -### The workspace says its startup script failed - -The configuration did not apply. The machine is up and reachable, but it is **not** running the -system it was built from — `nixos-rebuild switch` builds before it activates, so what is running is -whatever ran before: the previous generation, or on a first boot the bare AMI. - -The bootstrap records the reason in `/run/coder/boot-failed`, and the agent's startup script fails -on it, which is the only way a workspace will show an error for something that went wrong before -the agent existed. Read the "NixOS" log source for what actually happened, fix the flake, and -restart the workspace. - -On a first boot there is no agent in the AMI to run that script at all, so the bootstrap starts one -itself for a minute, lets it report, and kills it. The workspace is therefore failed rather than -"starting", but it has no agent: there is no terminal, no SSH and no IDE until the flake is fixed -and the workspace restarted. Use the serial console or SSM below to get inside it in the meantime. - ### The agent never connects The boot script writes its handoff to `/run/coder` before doing anything else, so the usual cause is -an instance with no route to the internet (a NixOS workspace fetches its own configuration on boot, -so it needs egress before it can report anything) — a failed rebuild reports itself failed instead -of hanging. The workspace metadata shows the instance id; the AMI logs to the serial console, which +a failed rebuild — or, if there are no logs at all, an instance with no route to the internet (a +NixOS workspace fetches its own configuration on boot, so it needs egress before it can report +anything). The workspace metadata shows the instance id; the AMI logs to the serial console, which needs no SSH: ```console diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index a989061ad..9388ed71a 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -49,16 +49,6 @@ variable "nixos_release" { default = "26.05" } -variable "troubleshooting_url" { - description = <<-EOT - Page the agent's "Troubleshoot" link points at. Shown when the agent has - not connected within `connection_timeout`, which on this template means - the instance never finished applying the flake. - EOT - type = string - default = "https://registry.coder.com/templates/coder-labs/aws-nixos" -} - data "coder_parameter" "instance_type" { name = "instance_type" display_name = "Instance type" @@ -143,26 +133,6 @@ resource "coder_agent" "main" { # The first boot completes a nixos-rebuild switch before the agent exists, # so the default 120s looks like a failed workspace. connection_timeout = 1200 - # Shown on the agent as a "Troubleshoot" link once it times out, which is - # the one moment the user has nothing else to go on. - troubleshooting_url = var.troubleshooting_url - - # The instance can come up perfectly while the configuration it was meant - # to run does not build, and a workspace has no way to say so: a deployment - # shows an agent's state only once it has connected, and nothing a script - # can call fails a build. What it does report is the startup script's exit - # status -- so the bootstrap leaves a file behind and this fails on it. - # See modules/amazon-init/README.md. - startup_script = <<-EOT - #!/usr/bin/env bash - set -euo pipefail - - failure='${local.runtime_dir}/boot-failed' - if [ -f "$failure" ]; then - cat "$failure" >&2 - exit 1 - fi - EOT metadata { key = "cpu" @@ -258,11 +228,6 @@ locals { "m7g.xlarge" = { agent = "arm64", ami = "arm64", attr = "aarch64" } } arch = local.arch_map[data.coder_parameter.instance_type.value] - - # Named here rather than read back from the amazon-init module: that module - # is given the agent's token, so an output of it cannot be used to configure - # the agent without making a cycle. - runtime_dir = "/run/coder" } # Everything about the flake: the checkout, the boot-time rebuild and the @@ -286,7 +251,6 @@ module "amazon_init" { agent_init_script = try(coder_agent.main[0].init_script, "") boot_script = module.nix.boot_script values = module.nix.values - runtime_dir = local.runtime_dir log_display_name = "NixOS" log_icon = "/icon/nix.svg" diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md index a12489d20..f8503481f 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md @@ -71,37 +71,6 @@ The mirror image of that rule is that a boot script failure must not leave an unreachable workspace, so the agent is started even when the boot script exits non-zero. Its status is reported and propagated, never suppressed. -## Saying that the boot failed - -A failed boot script has nothing to report itself with. A deployment renders -an agent's state only once that agent has connected, and no API a script can -reach marks a build failed — so an instance that never starts an agent is -indistinguishable from a slow one until `connection_timeout` expires, and the -reason sits in a log source nobody has a reason to open. - -Two things follow from that, and both are why a failure here is visible at -all: - -- **`boot-failed`.** When the boot script exits non-zero the module writes - `$${runtime_dir}/boot-failed` (mode 0644) with a sentence about what - happened, and removes it at the start of every boot. Read it from the - agent's `startup_script` and exit non-zero — the startup script's exit - status is the only failure state a workspace will show you. The path is the - `boot_failed_path` output. -- **A reporting run of the agent.** When the boot script has not produced a - `coder-agent.service` at all — a first boot whose configuration did not - build — nothing is left to run that startup script, so the module runs - `coder_agent.init_script` itself for `report_failure_timeout` seconds under - a transient `coder-agent-report.service`, and then kills it. The agent - connects, its startup scripts fail on `boot-failed`, and the workspace is - failed within the minute instead of after the connection timeout. - - It is killed, not stopped: on SIGTERM the agent reports `shutting_down` and - then `off`, over the top of the `start_error` it was started to report. And - it is not left running, because it would be serving the image's environment - rather than the configuration that was asked for — a workspace that looks - like it works. Set `report_failure = false` to turn it off. - ## What the boot script gets Root, a child process, and these: diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf index f149417d3..97609143c 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf @@ -140,33 +140,6 @@ variable "log_budget_bytes" { default = 524288 } -variable "report_failure" { - description = <<-EOT - When the boot script has not produced a `coder-agent.service`, run - `coder_agent.init_script` for `report_failure_timeout` seconds and then - kill it. - - A deployment shows an agent's state only once it has connected, and there - is no API a script can call to fail a build, so an instance that never - starts an agent is indistinguishable from a slow one until the connection - timeout expires. Running the agent briefly is the only way to say - otherwise: it connects, its startup scripts fail on `boot-failed`, and the - workspace is failed within the minute. The agent is then killed rather - than stopped, because a clean shutdown reports `off` over the top of it. - - Turn it off for images where an agent running as root is not acceptable. - The failure is still logged, and `boot-failed` is still written. - EOT - type = bool - default = true -} - -variable "report_failure_timeout" { - description = "Seconds to leave that agent running. Long enough for it to connect and run every startup script; nothing on the instance can read the state back to know." - type = number - default = 60 -} - variable "hostname" { description = "Hostname to set on the instance. Defaults to the workspace name." type = string @@ -221,9 +194,7 @@ locals { ARG_LOG_ICON = var.log_icon ARG_LOG_BUDGET = var.log_budget_bytes - ARG_HOSTNAME = local.hostname - ARG_REPORT_FAILURE = tostring(var.report_failure) - ARG_REPORT_TIMEOUT = var.report_failure_timeout + ARG_HOSTNAME = local.hostname }) # EC2 caps user-data at 16 KiB and the script above plus its payloads is @@ -277,18 +248,6 @@ output "workspace_facts_path" { value = "${var.runtime_dir}/workspace.json" } -output "boot_failed_path" { - description = <<-EOT - File written when the boot script fails, holding a sentence about what - happened. Absent on a healthy boot. - - Read it from the agent's startup script and exit non-zero: that is the - only way to get an error out of a workspace whose machine came up fine - but whose configuration did not. - EOT - value = "${var.runtime_dir}/boot-failed" -} - output "bootstrap_path" { description = "Where the user-data wrapper extracts the real boot script." value = "${var.runtime_dir}/bootstrap.sh" diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl index 494fe2196..dac7a4d21 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl @@ -21,8 +21,6 @@ LOG_DISPLAY_NAME_B64='${ARG_LOG_DISPLAY_NAME_B64}' LOG_ICON='${ARG_LOG_ICON}' LOG_BUDGET='${ARG_LOG_BUDGET}' HOSTNAME_='${ARG_HOSTNAME}' -REPORT_FAILURE='${ARG_REPORT_FAILURE}' -REPORT_TIMEOUT='${ARG_REPORT_TIMEOUT}' RUNTIME_DIR='${ARG_RUNTIME_DIR}' @@ -36,7 +34,7 @@ RUNTIME_DIR='${ARG_RUNTIME_DIR}' # the point: the token rotates on every workspace start. install -d -m 0700 "$RUNTIME_DIR" -rm -f "$RUNTIME_DIR/ready" "$RUNTIME_DIR/boot-failed" +rm -f "$RUNTIME_DIR/ready" umask 077 printf 'CODER_AGENT_TOKEN=%s\nCODER_AGENT_URL=%s\n' "$AGENT_TOKEN" "$ACCESS_URL" \ @@ -97,63 +95,6 @@ export CODER_LOG_LIBRARY="$RUNTIME_DIR/log.sh" coder_log_init "$(printf '%s' "$LOG_DISPLAY_NAME_B64" | base64 -d)" "$LOG_ICON" || true -# Nothing on the instance can report a failure except the agent. A deployment -# renders an agent's state only once it has connected, and there is no API a -# script can call to fail a build -- so an instance that never starts an agent -# shows "connecting" until the connection timeout expires twenty minutes -# later, with the reason buried in a log source nobody has a reason to open. -# -# `coder agent` can say it, though, and that is all this does: run the agent -# from the init script, let it connect and run its startup scripts -- which -# fail on the boot-failed file written above -- and then kill it. The -# workspace is failed within a minute instead of twenty, and the agent is -# gone: left running it would be serving the image's environment rather than -# the configuration that was asked for, which is a workspace that looks like -# it works. -# -# Killed rather than stopped: on SIGTERM the agent reports shutting_down and -# then off, which would overwrite the start_error it was started to report. -report_boot_failure() { - [ "$REPORT_FAILURE" = "true" ] || return 1 - command -v systemd-run >/dev/null 2>&1 || return 1 - - coder_log error "Starting the agent briefly to report the failure, then stopping it." || true - - # A run left over from an earlier attempt would fight this one over the same - # token, and a failed one keeps the unit name. - systemctl reset-failed coder-agent-report.service 2>/dev/null || true - install -d -m 0755 "$RUNTIME_DIR/bin" - - # The token is read from agent.env inside the unit rather than passed as a - # property, so that it does not end up in `systemctl show` output. - # - # RuntimeMaxSec and KillSignal so that the agent still goes away -- without - # reporting itself off on the way out -- if this script is killed before it - # gets to do it itself. - if ! systemd-run --quiet --collect --unit=coder-agent-report \ - --service-type=exec \ - --property=RuntimeMaxSec="$((REPORT_TIMEOUT + 60))" \ - --property=KillSignal=SIGKILL \ - --property=FinalKillSignal=SIGKILL \ - --setenv=HOME=/root \ - --setenv=PATH="$PATH" \ - --setenv=BINARY_DIR="$RUNTIME_DIR/bin" \ - /usr/bin/env bash -c \ - "set -a; . '$RUNTIME_DIR/agent.env'; set +a; exec bash '$RUNTIME_DIR/init.sh'"; then - coder_log error "Could not start the agent to report the failure." || true - return 1 - fi - - # Long enough for it to connect, run the startup scripts and report what - # they did. There is nothing to wait on: the state it is reporting lives on - # the deployment, and no endpoint an agent token can reach reads it back. - sleep "$REPORT_TIMEOUT" - systemctl stop coder-agent-report.service 2>/dev/null || true - - coder_log error "Reported. This workspace is failed and has no agent; fix the configuration and restart it." || true - return 0 -} - # Nothing starts coder-agent.service but this, and only once the boot script # has finished. If systemd started it at multi-user.target instead, the agent # would report the workspace ready and run its startup scripts while the boot @@ -166,12 +107,9 @@ start_agent() { fi if ! systemctl cat coder-agent >/dev/null 2>&1; then coder_log error "coder-agent.service does not exist; the boot script has never created it." || true - report_boot_failure || return 1 - return 0 + return 1 fi - # A reporting run from an earlier attempt must not fight the real agent. - systemctl stop coder-agent-report.service 2>/dev/null || true chown_handoff systemctl start coder-agent || true @@ -188,20 +126,9 @@ start_agent() { return 1 } -# Left for the agent to find. A connected agent is the only thing a -# deployment shows a state for, and the only state a script can choose is the -# one its startup script exits with -- so the failure is recorded here and the -# agent's startup script reads it back. Without that the workspace comes up -# looking perfectly healthy while running the wrong system. -record_failure() { - printf '%s\n' "$*" >"$RUNTIME_DIR/boot-failed" 2>/dev/null || true - chmod 0644 "$RUNTIME_DIR/boot-failed" 2>/dev/null || true -} - on_error() { local rc=$? coder_log error "Bootstrap failed (exit $rc)." || true - record_failure "The instance's boot script failed (exit $rc) before it could apply this workspace's configuration. See the boot logs for what went wrong." start_agent || true exit "$rc" } @@ -246,13 +173,11 @@ bash "$RUNTIME_DIR/boot.sh" || boot_rc=$? if [ "$boot_rc" -ne 0 ]; then coder_log error "Boot script exited $boot_rc; starting the agent anyway." || true - record_failure "This workspace's configuration could not be applied: the boot script exited $boot_rc. The machine is NOT running the configuration it was built from. See the boot logs for what went wrong." fi # `|| true` so that failing to start the agent does not re-enter the ERR trap, # which would report the bootstrap as having failed somewhere earlier. On a # first boot whose configuration does not build there is no agent unit to -# start at all; start_agent falls back to the init script so that the -# workspace is still reachable. +# start at all, and that is the honest outcome: nothing to report it with. start_agent || true exit "$boot_rc" From 8b08525fe235e76d617662781371ccfd000b147b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Thu, 24 Sep 2026 10:09:48 +0000 Subject: [PATCH 11/24] docs(aws-nixos): the upgrade timer no longer streams to the workspace Pairs with coder/nixos-example-flake@e4bfe64, which removes coder-stream-nixos-upgrade-logs.service. Boot rebuilds still stream; a scheduled upgrade now writes to the journal and nowhere else. --- registry/coder-labs/templates/aws-nixos/README.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 4cffa3376..6f6e80476 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -181,7 +181,7 @@ policy — a dirty tree or local commits are built as they are, never discarded. To rebuild immediately, on the workspace: ```console -sudo systemctl start nixos-upgrade # sync, then rebuild, streamed to the UI +sudo systemctl start nixos-upgrade # sync, then rebuild sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-x86_64 ``` @@ -196,9 +196,9 @@ under `these N derivations will be built:`, per-derivation compiler output, and /var/log/coder-nixos/rebuild-latest.log # symlink to the most recent boot rebuild ``` -Scheduled upgrades run as `nixos-upgrade.service` and write to the journal, which -`coder-stream-nixos-upgrade-logs.service` follows into the same **NixOS** log source while the -upgrade runs. `journalctl -u nixos-upgrade` has the unabridged copy. +Scheduled upgrades are the exception: `nixos-upgrade.service` writes to the journal and nowhere +else, so nothing of theirs reaches the workspace UI. `journalctl -u nixos-upgrade` has the whole +story, and the "NixOS version" metadata on the workspace shows when one has staged a generation. Keeping compiler output out of the UI is not cosmetic. Coder caps agent logs at **1 MiB per agent**, shared across every log source, and exceeding it does not truncate — the log is marked From a44efa77e423510b20c174b7abd7369f342a7b77 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 13:46:58 +0000 Subject: [PATCH 12/24] refactor(aws-nixos): apply review feedback - `coder_curl` -> `_curl` in the log library. - `NIX_SUDO` becomes a `_sudo` function. Nicer at the call sites that were relying on an empty variable disappearing by word-splitting, which is exactly the kind of thing that stops being true the first time someone quotes it. - Terraform module labels use `-`, not `_`: `amazon-init`, `aws-region`, `jetbrains-gateway`, matching `code-server` and `git-config`. - PREREQUISITES.md folded into the README as three short subsections and the IAM policy, in the shape aws-linux uses. The credential-strategy ranking, the provisioner-environment essay and the egress table are gone; what is left is the default-VPC requirement and one sentence on egress, which is the part that is actually specific to building a configuration on boot. - `modules/nix/README.md`: the flake reference section is two sentences and a link to the upstream syntax instead of a table restating it. - Comments deleted where they restated the code: both module headers, the duplicated 16 KiB rationale above the precondition (the `nonsensitive` clause survives, since nothing else explains it), the user-data wrapper section, and the git-config note. - The "why not cloud-init" opener loses the NixOS aside. While in lifecycle.sh, group the logging helpers. `nix_filter_log` was 158 lines away from the other three, under the `applying` banner, so the file read logging -> checkout -> deciding -> logging -> applying. It is now one block behind its own banner. Not split into its own file, which was the other option considered: lifecycle.sh is never a file on the instance -- it is interpolated into boot.sh, which is interpolated into the bootstrap, which is gzipped into user-data. A second Terraform payload costs 4 bytes, but it would leave lifecycle.sh referencing `nix_log` and `nix_filter_log` with no way to obtain them, so it would stop being sourceable on its own. Shipping a real file through `files{}` and sourcing it costs 1.2 KiB of the 3.1 KiB of user-data headroom, for a library with exactly one consumer. Also flips `verified: true`. --- .../templates/aws-nixos/PREREQUISITES.md | 100 ---------- .../coder-labs/templates/aws-nixos/README.md | 75 +++++++- .../coder-labs/templates/aws-nixos/main.tf | 16 +- .../aws-nixos/modules/amazon-init/README.md | 29 +-- .../aws-nixos/modules/amazon-init/main.tf | 11 -- .../modules/amazon-init/scripts/log.sh | 6 +- .../templates/aws-nixos/modules/nix/README.md | 19 +- .../templates/aws-nixos/modules/nix/main.tf | 11 -- .../modules/nix/scripts/lifecycle.sh | 176 ++++++++++-------- 9 files changed, 185 insertions(+), 258 deletions(-) delete mode 100644 registry/coder-labs/templates/aws-nixos/PREREQUISITES.md diff --git a/registry/coder-labs/templates/aws-nixos/PREREQUISITES.md b/registry/coder-labs/templates/aws-nixos/PREREQUISITES.md deleted file mode 100644 index 4b2d969be..000000000 --- a/registry/coder-labs/templates/aws-nixos/PREREQUISITES.md +++ /dev/null @@ -1,100 +0,0 @@ -# Prerequisites - -## Authentication - -This template authenticates to AWS using the provider's default [authentication methods](https://registry.terraform.io/providers/hashicorp/aws/latest/docs#authentication-and-configuration). - -The simplest way, without editing the template, is environment variables (`AWS_ACCESS_KEY_ID`, -`AWS_SECRET_ACCESS_KEY`, `AWS_REGION`) or a [credentials file](https://docs.aws.amazon.com/cli/latest/userguide/cli-configure-files.html#cli-configure-files-format). -If you are running Coder on a VM, that file must be at `/home/coder/aws/credentials`. - -Credentials belong in the environment of the **provisioner process** — `coder server`, or your -external provisioner — and not in Terraform variables. Template variables surface in workspace -parameters and build logs, so a credential passed that way is readable by anyone who can view a -build. Restart the provisioner after changing them. - -Prefer, in order: - -1. An **instance profile** (Coder on EC2) or **IRSA** (Coder on EKS). No long-lived secret exists. -2. A long-lived, low-privilege identity plus `assume_role` in the provider block. -3. Static access keys. - -Avoid `AWS_SESSION_TOKEN` from STS for a provisioner: it expires, and it will expire in the middle -of a build. - -## A default VPC in the selected region - -Like the `aws-linux` template, this one launches into the default VPC of the region chosen by the -`aws_region` parameter and does not take a subnet or security group. Regions without a default VPC -fail at apply with `VPCIdNotSpecified`. Either pick a region that has one, or create one with -`aws ec2 create-default-vpc --region `. - -## Required permissions / policy - -The following sample policy allows Coder to create EC2 instances and modify instances it -provisioned. - -```json -{ - "Version": "2012-10-17", - "Statement": [ - { - "Sid": "VisualEditor0", - "Effect": "Allow", - "Action": [ - "ec2:GetDefaultCreditSpecification", - "ec2:DescribeIamInstanceProfileAssociations", - "ec2:DescribeTags", - "ec2:DescribeInstances", - "ec2:DescribeInstanceTypes", - "ec2:DescribeInstanceStatus", - "ec2:CreateTags", - "ec2:RunInstances", - "ec2:DescribeInstanceCreditSpecifications", - "ec2:DescribeImages", - "ec2:ModifyDefaultCreditSpecification", - "ec2:DescribeVolumes" - ], - "Resource": "*" - }, - { - "Sid": "CoderResources", - "Effect": "Allow", - "Action": [ - "ec2:DescribeInstanceAttribute", - "ec2:UnmonitorInstances", - "ec2:TerminateInstances", - "ec2:StartInstances", - "ec2:StopInstances", - "ec2:DeleteTags", - "ec2:MonitorInstances", - "ec2:CreateTags", - "ec2:RunInstances", - "ec2:ModifyInstanceAttribute", - "ec2:ModifyInstanceCreditSpecification" - ], - "Resource": "arn:aws:ec2:*:*:instance/*", - "Condition": { - "StringEquals": { - "aws:ResourceTag/Coder_Provisioned": "true" - } - } - } - ] -} -``` - -## Network egress from the workspace - -Unlike the stock AWS templates, workspaces here must reach more than the Coder deployment. A NixOS -instance resolves and builds its own configuration on boot, so it needs outbound HTTPS to: - -| Host | Why | -| ------------------------------------ | -------------------------------------------------------- | -| your Coder access URL | agent connection and the log API | -| `cache.nixos.org` | binary cache; without it everything is built from source | -| `github.com` / `codeload.github.com` | fetching your flake and its nixpkgs input | -| any extra substituters you configure | binary caches declared in your flake | - -No inbound rules are required. If you plan to use the SSH rescue path described in the README, open -port 22 from your own address — the default VPC security group does not allow it. diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 6f6e80476..b5eb46a68 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -2,7 +2,7 @@ display_name: AWS EC2 (NixOS) description: Provision NixOS EC2 VMs as Coder workspaces from a flake icon: ../../../../.icons/nixos.svg -verified: false +verified: true tags: [vm, linux, aws, nixos, persistent-vm] --- @@ -22,9 +22,76 @@ point the template at your own fork to control the environment. This template authenticates to AWS using the provider's default [authentication methods](https://registry.terraform.io/providers/hashicorp/aws/latest/docs#authentication-and-configuration). -The simplest way is to set `AWS_ACCESS_KEY_ID` and `AWS_SECRET_ACCESS_KEY` in the environment of the -Coder provisioner. See [PREREQUISITES.md](./PREREQUISITES.md) for the IAM policy, the default VPC -requirement, and the reasons credentials must not be passed as template variables. +The simplest way, without editing the template, is environment variables (e.g. `AWS_ACCESS_KEY_ID`) +or a [credentials file](https://docs.aws.amazon.com/cli/latest/userguide/cli-configure-files.html#cli-configure-files-format), +set for the provisioner process rather than as template variables. If you are running Coder on a VM, +that file must be at `/home/coder/aws/credentials`. + +### The region needs a default VPC + +Like `aws-linux`, this template takes no subnet or security group and launches into the default VPC +of the selected region. Regions without one fail at apply with `VPCIdNotSpecified`. + +### Egress from the workspace + +A NixOS instance fetches and builds its own configuration on boot, so workspaces need outbound +HTTPS to your Coder access URL, `cache.nixos.org`, and wherever the flake is hosted. No inbound +rules are required. + +## Required permissions / policy + +The following sample policy allows Coder to create EC2 instances and modify instances provisioned by +Coder: + +```json +{ + "Version": "2012-10-17", + "Statement": [ + { + "Sid": "VisualEditor0", + "Effect": "Allow", + "Action": [ + "ec2:GetDefaultCreditSpecification", + "ec2:DescribeIamInstanceProfileAssociations", + "ec2:DescribeTags", + "ec2:DescribeInstances", + "ec2:DescribeInstanceTypes", + "ec2:DescribeInstanceStatus", + "ec2:CreateTags", + "ec2:RunInstances", + "ec2:DescribeInstanceCreditSpecifications", + "ec2:DescribeImages", + "ec2:ModifyDefaultCreditSpecification", + "ec2:DescribeVolumes" + ], + "Resource": "*" + }, + { + "Sid": "CoderResources", + "Effect": "Allow", + "Action": [ + "ec2:DescribeInstanceAttribute", + "ec2:UnmonitorInstances", + "ec2:TerminateInstances", + "ec2:StartInstances", + "ec2:StopInstances", + "ec2:DeleteTags", + "ec2:MonitorInstances", + "ec2:CreateTags", + "ec2:RunInstances", + "ec2:ModifyInstanceAttribute", + "ec2:ModifyInstanceCreditSpecification" + ], + "Resource": "arn:aws:ec2:*:*:instance/*", + "Condition": { + "StringEquals": { + "aws:ResourceTag/Coder_Provisioned": "true" + } + } + } + ] +} +``` ## How it works diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index 9388ed71a..7f6d03210 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -10,14 +10,14 @@ terraform { } } -module "aws_region" { +module "aws-region" { source = "registry.coder.com/coder/aws-region/coder" version = "~> 1.0" default = "eu-west-3" } provider "aws" { - region = module.aws_region.value + region = module.aws-region.value } variable "flake_ref" { @@ -183,7 +183,7 @@ module "code-server" { # The IDE backend is a dynamically linked download that Gateway unpacks into # the workspace and execs, which on NixOS needs `programs.nix-ld`. The # reference flake enables it. -module "jetbrains_gateway" { +module "jetbrains-gateway" { count = data.coder_workspace.me.start_count source = "registry.coder.com/coder/jetbrains-gateway/coder" version = "~> 1.2" @@ -204,10 +204,6 @@ module "jetbrains_gateway" { order = 2 } -# Git authorship, which the NixOS configuration deliberately does not set: it -# is per-workspace state a flake has no pure way to learn. -# -# See https://registry.coder.com/modules/coder/git-config module "git-config" { count = data.coder_workspace.me.start_count source = "registry.coder.com/coder/git-config/coder" @@ -244,7 +240,7 @@ module "nix" { # Gets Coder onto the instance and runs one script on every boot. It knows # nothing about Nix: `boot_script` is an opaque string to it, and the flake is # applied entirely inside that string. See ./modules/amazon-init/README.md. -module "amazon_init" { +module "amazon-init" { source = "./modules/amazon-init" agent_token = try(coder_agent.main[0].token, "") @@ -258,9 +254,9 @@ module "amazon_init" { resource "aws_instance" "dev" { ami = data.aws_ami.nixos.id - availability_zone = "${module.aws_region.value}a" + availability_zone = "${module.aws-region.value}a" instance_type = data.coder_parameter.instance_type.value - user_data = module.amazon_init.user_data + user_data = module.amazon-init.user_data # The agent token is inside user-data and rotates on every workspace start, # so user-data changes on every start. With replacement enabled, every diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md index f8503481f..0c16f917a 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md @@ -8,7 +8,7 @@ agent. It does not know or care what that script does — `boot_script` is an opaque string, and nothing in this module reads it. ```tf -module "amazon_init" { +module "amazon-init" { source = "./modules/amazon-init" agent_token = coder_agent.main.token @@ -17,7 +17,7 @@ module "amazon_init" { } resource "aws_instance" "dev" { - user_data = module.amazon_init.user_data + user_data = module.amazon-init.user_data } ``` @@ -26,10 +26,9 @@ Workspace and owner identity are read from `coder_workspace` and ## Why this is not cloud-init -Some AMIs — the official NixOS images among them — do not ship cloud-init. -They run `amazon-init.service`, which reads `/etc/ec2-metadata/user-data` and -execs it as a shell script when it begins with `#!` — after -`multi-user.target`, **on every boot**. There is no `runcmd`, no +Some AMIs do not ship cloud-init, they run `amazon-init.service`, which reads +`/etc/ec2-metadata/user-data` and execs it as a shell script when it begins +with `#!` — after `multi-user.target`, **on every boot**. There is no `runcmd`, no `write_files`, no per-boot/once distinction and no ordering hooks. Three things follow: @@ -116,21 +115,3 @@ Two things it handles that are easy to get wrong: own first generation, before anything has been rebuilt. An image without one simply gets no logs: every function here fails closed, because logging must never be the reason a boot fails. - -## The user-data wrapper - -EC2 caps user-data at 16 KiB, and the bootstrap script plus its payloads is -past that. So the output is a six-line self-extracting wrapper around a -gzipped copy. - -That is transparent to `amazon-init`: it only checks the first two bytes for -`#!` before exec'ing the blob. The wrapper extracts to -`$${runtime_dir}/bootstrap.sh`, so the real script is on disk when a boot needs -debugging. - -The 16 KiB limit is asserted here, as a `precondition` on the `user_data` -output, so the plan fails in the module that decides what goes into user-data -rather than at apply time with an EC2 error that names no cause. Anything -passed through `files` counts against it — and note that those contents are -gzipped before user-data is gzipped again, which buys nothing, so `files` is -for small text. diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf index 97609143c..bbe44ebda 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf @@ -1,10 +1,3 @@ -# Coder on an AMI that runs amazon-init instead of cloud-init. -# -# Renders EC2 user-data that publishes the agent handoff, publishes the -# workspace's identity, runs one boot script supplied by the caller, and then -# starts the agent. What that boot script does is none of this module's -# business: it is a string, and the module never looks inside it. - terraform { required_version = ">= 1.0" @@ -221,10 +214,6 @@ output "user_data" { value = local.user_data sensitive = true - # EC2 rejects user-data over 16 KiB, and it does so at apply time with an - # error that says nothing about which part grew. Checking here fails the - # plan instead, in the module that decides what goes in. - # # nonsensitive because the length is sensitive by propagation, and Terraform # suppresses error messages derived from sensitive values. precondition { diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh index b0258457f..b806bce87 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh @@ -20,7 +20,7 @@ CODER_LOG_READY="${CODER_LOG_READY:-0}" # Resolved once. An image without curl gets no logs: every function here then # fails closed, which is the right trade -- logging must never be the reason a # boot fails. -coder_curl() { +_curl() { if [ -z "${CODER_CURL:-}" ]; then command -v curl > /dev/null 2>&1 || return 1 CODER_CURL=$(command -v curl) @@ -54,7 +54,7 @@ coder_log_init() { while [ "$attempt" -lt 40 ]; do code=$( - coder_curl -sS -o /dev/null -w '%{http_code}' -X POST \ + _curl -sS -o /dev/null -w '%{http_code}' -X POST \ "$CODER_ACCESS_URL/api/v2/workspaceagents/me/log-source" \ -H "Coder-Session-Token: $CODER_AGENT_TOKEN" \ -H 'Content-Type: application/json' \ @@ -95,7 +95,7 @@ coder_log_send() { [ "$(coder_log_budget_left)" -gt "$size" ] || return 0 coder_log_budget_add "$size" - coder_curl -sS -o /dev/null -X PATCH \ + _curl -sS -o /dev/null -X PATCH \ "$CODER_ACCESS_URL/api/v2/workspaceagents/me/logs" \ -H "Coder-Session-Token: $CODER_AGENT_TOKEN" \ -H 'Content-Type: application/json' \ diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/README.md b/registry/coder-labs/templates/aws-nixos/modules/nix/README.md index a6ad75a1a..b7ef1c1df 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/README.md @@ -3,15 +3,10 @@ The flake lifecycle: keep a checkout in sync with a Git remote, decide whether the running system is out of date, and rebuild it. -Nothing here knows about EC2, user-data, or how the instance came to exist. The boot path is a **string** the caller hands to whatever runs scripts on the machine. On AWS the caller is [`../amazon-init`](../amazon-init/README.md), but nothing depends on that. -Keeping the machine current _after_ boot is not this module's job either. That -belongs to the configuration, as `system.autoUpgrade` on a systemd timer the -machine's owner can read and change. - ```tf module "nix" { source = "./modules/nix" @@ -24,15 +19,11 @@ module "nix" { ## The flake reference -One string, in the form `nix` itself accepts. `?ref=` carries the branch, and -without it the remote's default branch is used — resolved on the instance, -since Terraform cannot know it without talking to the remote. - -| `flake_ref` | builds | -| ------------------------------------- | --------------- | -| `https://host/org/repo` | default branch | -| `git+https://host/org/repo?ref=dev` | `dev` | -| `git+ssh://git@host/org/repo?ref=dev` | `dev`, over SSH | +`flake_ref` is what you would pass to `nixos-rebuild --flake`: a +[flake reference](https://nix.dev/manual/nix/latest/command-ref/new-cli/nix3-flake#flake-references), +with the branch in `?ref=` and the remote's default branch when it is absent — +resolved on the instance, since Terraform cannot know it without talking to the +remote. `$ARCH` in `flake_attr` is replaced with `arch`, so one template can offer both architectures without the attribute and the machine disagreeing. diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf index 2efdd00f8..9cd1a9d70 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf @@ -1,14 +1,3 @@ -# The flake lifecycle at boot: clone or fast-forward the checkout, decide -# whether the running system is out of date, and rebuild it. -# -# Nothing here knows about EC2, user-data or how the instance was started -- -# the boot path is exposed as a string for whatever puts scripts on the -# machine, which on AWS is ../amazon-init. -# -# Keeping the machine current *afterwards* is not this module's job either. -# That is `system.autoUpgrade` in the configuration itself, on a systemd timer -# the machine's owner can read and change. - terraform { required_version = ">= 1.3" } diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh index 08e9c7628..6bd7cae8b 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh @@ -5,11 +5,24 @@ export NIX_CONFIG="experimental-features = nix-command flakes" -# Callers run as root from user-data and as the workspace user from a -# coder_script, so privilege handling belongs here rather than in each. -NIX_SUDO="" -[ "$(id -u)" -eq 0 ] || NIX_SUDO="sudo" +# Callers run as root from user-data, and by hand as the workspace user, so +# privilege handling belongs here rather than in each of them. +_sudo() { + if [ "$(id -u)" -eq 0 ]; then + "$@" + else + sudo "$@" + fi +} +# --------------------------------------------------------------------------- +# logging +# --------------------------------------------------------------------------- +# +# Everything below writes through nix_log, which the caller is expected to +# define -- see README.md. The stub is what a machine with no Coder runtime +# gets, and it is guarded rather than unconditional so that a caller which +# has already defined the real one wins. command -v nix_log > /dev/null 2>&1 || nix_log() { shift printf '%s\n' "$*" @@ -19,6 +32,57 @@ nix_strip_ansi() { sed -e 's/\x1b\[[0-9;]*[a-zA-Z]//g' -e 's/\r$//' } +# Drop the things that are noise rather than progress; everything else reaches +# the caller as nix wrote it. Nix's plain output is already the readable +# output -- every --log-format is byte-identical off a TTY. +nix_filter_log() { + local line prefix + while IFS= read -r line || [ -n "$line" ]; do + case "$line" in + *$'\033'*) line=$(printf '%s' "$line" | nix_strip_ansi) ;; + esac + case "$line" in + "") continue ;; + # The enumerated store paths under "these N derivations will be built:" + " "*/nix/store/*) continue ;; + # Those list headers end in a colon promising the list we just dropped, + # which reads as truncated output. Restate them as complete sentences. + "these "*" will be "*: | "this "*" will be "*:) + case "$line" in + "these "*) line=${line#these } ;; + "this "*) line="1 ${line#this }" ;; + esac + printf '%s\n' "${line%:}" + continue + ;; + # git's fetch progress, including the indented ref-update line. + "remote: "* | "From "* | " "*".."*"->"*) continue ;; + # Per-derivation build output, "> text". The "building '...'" + # line already marks it and the text is in the transcript. Requiring a + # space-free prefix stops this eating lines that contain "> ". + *"> "*) + prefix=${line%%> *} + case "$prefix" in + "" | *[[:space:]]*) ;; + *) continue ;; + esac + ;; + esac + + # Nix writes its progress in lowercase ("building the system + # configuration..."), and in the workspace UI those lines sit among + # sentences from every other log source. Capitalise the first letter -- + # but only when the first word is a plain word, so that program names + # ("nixos-rebuild: ...") and paths stay exactly as they were written. + case "${line%% *}" in + [a-z]*[!a-zA-Z:]*) ;; + [a-z]*) line="${line^}" ;; + esac + + printf '%s\n' "$line" + done +} + # Runs a command with its output filtered into the log, and returns the # command's own status rather than the filter's. # @@ -53,11 +117,11 @@ nix_run_logged() { # on its tracking branch. A workspace whose configuration silently reverted on # restart would be worse than one that drifts. nix_checkout_dirty() { - [ -n "$($NIX_SUDO git -C "$NIX_FLAKE_DIR" status --porcelain 2> /dev/null | head -1)" ] + [ -n "$(_sudo git -C "$NIX_FLAKE_DIR" status --porcelain 2> /dev/null | head -1)" ] } nix_checkout_rev() { - $NIX_SUDO git -C "$NIX_FLAKE_DIR" rev-parse HEAD 2> /dev/null || true + _sudo git -C "$NIX_FLAKE_DIR" rev-parse HEAD 2> /dev/null || true } # Keep the whole tree owned by whoever owns the directory. We clone and fetch @@ -67,14 +131,14 @@ nix_checkout_rev() { nix_own_checkout() { local owner owner=$(stat -c %U "$NIX_FLAKE_DIR" 2> /dev/null || echo root) - $NIX_SUDO chown -R "$owner" "$NIX_FLAKE_DIR" 2> /dev/null || true + _sudo chown -R "$owner" "$NIX_FLAKE_DIR" 2> /dev/null || true } # The branch the checkout is actually on. With no `?ref=` in the flake # reference there is nothing to ask but the checkout itself, and after a clone # that is the remote's default branch. nix_checkout_branch() { - $NIX_SUDO git -C "$NIX_FLAKE_DIR" rev-parse --abbrev-ref HEAD 2> /dev/null || true + _sudo git -C "$NIX_FLAKE_DIR" rev-parse --abbrev-ref HEAD 2> /dev/null || true } nix_sync_checkout() { @@ -82,26 +146,26 @@ nix_sync_checkout() { if [ ! -e "$NIX_FLAKE_DIR/flake.nix" ]; then nix_log info "Cloning $url into $NIX_FLAKE_DIR" - $NIX_SUDO install -d -m 0755 "$NIX_FLAKE_DIR" + _sudo install -d -m 0755 "$NIX_FLAKE_DIR" # Clone into a temporary directory and move the contents, because the # directory already exists (systemd-tmpfiles creates it) and git refuses # to clone into a non-empty one. local tmp - tmp="$($NIX_SUDO mktemp -d)" + tmp="$(_sudo mktemp -d)" rc=0 if [ -n "$branch" ]; then - nix_run_logged $NIX_SUDO git clone --branch "$branch" "$url" "$tmp/repo" || rc=$? + nix_run_logged _sudo git clone --branch "$branch" "$url" "$tmp/repo" || rc=$? else - nix_run_logged $NIX_SUDO git clone "$url" "$tmp/repo" || rc=$? + nix_run_logged _sudo git clone "$url" "$tmp/repo" || rc=$? fi if [ "$rc" -ne 0 ]; then nix_log error "Could not clone $url${branch:+ (branch $branch)} (git exited $rc)" nix_log error "Check the flake reference the template was pushed with: the repository has to exist and be readable from this instance." - $NIX_SUDO rm -rf "$tmp" + _sudo rm -rf "$tmp" return 1 fi - $NIX_SUDO sh -c "cd '$tmp/repo' && tar cf - ." | $NIX_SUDO tar xf - -C "$NIX_FLAKE_DIR" - $NIX_SUDO rm -rf "$tmp" + _sudo sh -c "cd '$tmp/repo' && tar cf - ." | _sudo tar xf - -C "$NIX_FLAKE_DIR" + _sudo rm -rf "$tmp" nix_own_checkout return 0 fi @@ -126,20 +190,20 @@ nix_sync_checkout() { fi # Fetching as root writes into .git, so re-assert ownership afterwards. - if ! nix_run_logged $NIX_SUDO git -C "$NIX_FLAKE_DIR" fetch --quiet origin "$branch"; then + if ! nix_run_logged _sudo git -C "$NIX_FLAKE_DIR" fetch --quiet origin "$branch"; then nix_log warn "Could not reach the remote; building the existing checkout" return 0 fi local_rev="$(nix_checkout_rev)" - upstream_rev="$($NIX_SUDO git -C "$NIX_FLAKE_DIR" rev-parse FETCH_HEAD 2> /dev/null || true)" + upstream_rev="$(_sudo git -C "$NIX_FLAKE_DIR" rev-parse FETCH_HEAD 2> /dev/null || true)" [ -n "$upstream_rev" ] || return 0 [ "$local_rev" != "$upstream_rev" ] || return 0 # Fast-forward only. A checkout carrying local commits is left alone. - if $NIX_SUDO git -C "$NIX_FLAKE_DIR" merge-base --is-ancestor "$local_rev" "$upstream_rev" 2> /dev/null; then + if _sudo git -C "$NIX_FLAKE_DIR" merge-base --is-ancestor "$local_rev" "$upstream_rev" 2> /dev/null; then nix_log info "Updating $NIX_FLAKE_DIR to ${upstream_rev:0:12}" - $NIX_SUDO git -C "$NIX_FLAKE_DIR" reset --hard --quiet "$upstream_rev" + _sudo git -C "$NIX_FLAKE_DIR" reset --hard --quiet "$upstream_rev" nix_own_checkout else nix_log info "$NIX_FLAKE_DIR has local commits; building those instead of $branch" @@ -155,8 +219,8 @@ nix_recorded_rev() { } nix_record_rev() { - $NIX_SUDO install -d -m 0755 "$NIX_STATE_DIR" - printf '%s\n' "$1" | $NIX_SUDO tee "$NIX_STATE_DIR/flake.rev" > /dev/null + _sudo install -d -m 0755 "$NIX_STATE_DIR" + printf '%s\n' "$1" | _sudo tee "$NIX_STATE_DIR/flake.rev" > /dev/null } # True when a generation is the boot default but is not the running system, @@ -192,57 +256,6 @@ nix_needs_rebuild() { # applying # --------------------------------------------------------------------------- -# Drop the things that are noise rather than progress; everything else reaches -# the caller as nix wrote it. Nix's plain output is already the readable -# output -- every --log-format is byte-identical off a TTY. -nix_filter_log() { - local line prefix - while IFS= read -r line || [ -n "$line" ]; do - case "$line" in - *$'\033'*) line=$(printf '%s' "$line" | nix_strip_ansi) ;; - esac - case "$line" in - "") continue ;; - # The enumerated store paths under "these N derivations will be built:" - " "*/nix/store/*) continue ;; - # Those list headers end in a colon promising the list we just dropped, - # which reads as truncated output. Restate them as complete sentences. - "these "*" will be "*: | "this "*" will be "*:) - case "$line" in - "these "*) line=${line#these } ;; - "this "*) line="1 ${line#this }" ;; - esac - printf '%s\n' "${line%:}" - continue - ;; - # git's fetch progress, including the indented ref-update line. - "remote: "* | "From "* | " "*".."*"->"*) continue ;; - # Per-derivation build output, "> text". The "building '...'" - # line already marks it and the text is in the transcript. Requiring a - # space-free prefix stops this eating lines that contain "> ". - *"> "*) - prefix=${line%%> *} - case "$prefix" in - "" | *[[:space:]]*) ;; - *) continue ;; - esac - ;; - esac - - # Nix writes its progress in lowercase ("building the system - # configuration..."), and in the workspace UI those lines sit among - # sentences from every other log source. Capitalise the first letter -- - # but only when the first word is a plain word, so that program names - # ("nixos-rebuild: ...") and paths stay exactly as they were written. - case "${line%% *}" in - [a-z]*[!a-zA-Z:]*) ;; - [a-z]*) line="${line^}" ;; - esac - - printf '%s\n' "$line" - done -} - # `switch` and `boot` both build before activating, so a configuration that # fails to build never touches the running system; a separate `nix build` # gate would add nothing. @@ -253,16 +266,16 @@ nix_filter_log() { nix_apply() { local operation="$1" transcript="$2" rc - $NIX_SUDO install -d -m 0755 "$NIX_LOG_DIR" - $NIX_SUDO install -m 0644 /dev/null "$transcript" - $NIX_SUDO ln -sfn "$transcript" "$NIX_LOG_DIR/rebuild-latest.log" + _sudo install -d -m 0755 "$NIX_LOG_DIR" + _sudo install -m 0644 /dev/null "$transcript" + _sudo ln -sfn "$transcript" "$NIX_LOG_DIR/rebuild-latest.log" # stdbuf so output streams instead of arriving in one burst at the end. set +e - $NIX_SUDO nixos-rebuild "$operation" \ + _sudo nixos-rebuild "$operation" \ --flake "$NIX_FLAKE_DIR#$NIX_FLAKE_ATTR" \ --print-build-logs \ - 2>&1 | stdbuf -oL $NIX_SUDO tee -a "$transcript" | nix_filter_log \ + 2>&1 | stdbuf -oL _sudo tee -a "$transcript" | nix_filter_log \ | while IFS= read -r line; do nix_log info "$line"; done rc=${PIPESTATUS[0]} set -e @@ -270,14 +283,15 @@ nix_apply() { return "$rc" } -# coder_script does not prevent overlapping cron runs, and two concurrent -# nixos-rebuild processes are a bad time. +# Two concurrent nixos-rebuild processes are a bad time, and the boot path is +# not the only thing that rebuilds this machine: someone at a terminal can run +# nixos-rebuild by hand while a boot is still in progress. nix_lock() { local lock="$NIX_STATE_DIR/rebuild.lock" - $NIX_SUDO install -d -m 0755 "$NIX_STATE_DIR" + _sudo install -d -m 0755 "$NIX_STATE_DIR" # Mode 0666 so the workspace user can take the same advisory lock as root. # It carries no data, only the lock. - [ -e "$lock" ] || $NIX_SUDO install -m 0666 /dev/null "$lock" + [ -e "$lock" ] || _sudo install -m 0666 /dev/null "$lock" exec 9> "$lock" if [ "${1:-wait}" = "nowait" ]; then flock -n 9 From 6d9b3afb59a485b09d15cf6c4a920dfc472b5728 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 13:59:37 +0000 Subject: [PATCH 13/24] docs(aws-nixos): no upgrade timer, and the modules moved out Pairs with coder/nixos-example-flake@e7ff0ad, which deleted `modules/coder/auto-upgrade.nix` and now imports the Coder integration from coder/nixos-modules. So the template's claim that the cadence "belongs to the machine" is down to one sentence: a workspace rebuilds when it boots, and anything else is `system.autoUpgrade` in someone's own flake. The `coder.autoUpgrade` block, `systemctl list-timers nixos-upgrade` and the journal-only note about scheduled upgrades all described units that no longer exist. The "NixOS version" metadata is unchanged and still correct -- it compares /run/current-system against the system profile, so it reports any staged generation, whatever staged it. --- .../coder-labs/templates/aws-nixos/README.md | 49 +++++++------------ .../coder-labs/templates/aws-nixos/main.tf | 10 ++-- 2 files changed, 22 insertions(+), 37 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index b5eb46a68..5e8fa6e61 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -11,8 +11,10 @@ tags: [vm, linux, aws, nixos, persistent-vm] Provision NixOS EC2 instances as [Coder workspaces](https://coder.com/docs/workspaces), configured declaratively from a flake in a Git repository. The Coder agent is declared as a NixOS systemd unit, so it survives `nixos-rebuild` and the workspace stays reachable across configuration changes. The -reference configuration lives at [coder/nixos-example-flake](https://github.com/coder/nixos-example-flake); -point the template at your own fork to control the environment. +reference configuration lives at [coder/nixos-example-flake](https://github.com/coder/nixos-example-flake), +which imports the agent, the workspace user and the shutdown hook from +[coder/nixos-modules](https://github.com/coder/nixos-modules). Point the template at your own fork +to control the environment. @@ -220,37 +222,21 @@ A flake built from a git checkout ignores untracked files — `git add` a new ## Keeping workspaces up to date -The schedule belongs to the machine, not to this template. The reference flake enables NixOS's own -[`system.autoUpgrade`](https://search.nixos.org/options?query=system.autoUpgrade), wrapped as -`coder.autoUpgrade` so the defaults make sense for a workspace: +A workspace rebuilds from the flake when it **boots**, and that is the only schedule there is: to +pick up a change, restart the workspace, or run the rebuild yourself: -```nix -coder.autoUpgrade = { - dates = "04:40"; # systemd OnCalendar, not cron - operation = "boot"; # or "switch" -}; +```console +sudo nixos-rebuild switch # /etc/nixos/flake.nix is found on its own ``` -- **`boot` (default)** — builds the new configuration and makes it the boot default without - activating it. Nothing restarts while you are working; the change lands on your next workspace - restart, and the "NixOS version" metric shows `(restart to apply update)` until then. -- **`switch`** — activates immediately, restarting any service whose definition changed. - -Set `coder.autoUpgrade.enable = false` to turn it off, and edit `/etc/nixos` on the workspace to -change any of it — this is a normal NixOS timer, so `systemctl list-timers nixos-upgrade` and -`systemd-analyze calendar ''` tell you what will happen and when. - -The timer is **not** `Persistent`: a schedule missed while the workspace was stopped is not made up -on the next boot, because booting already rebuilds from the checkout. Before each run the -configuration fast-forwards the checkout under the same lock the boot path uses, with the same -policy — a dirty tree or local commits are built as they are, never discarded. - -To rebuild immediately, on the workspace: +Adding a timer is the configuration's business, not this template's — `system.autoUpgrade` is +upstream's and goes in the flake, where whoever owns the machine can see it. Two things it will not +do for you: order itself behind the boot-time rebuild (`amazon-init.service`), and sync the +checkout first, since `--refresh` does nothing for a local path flake. -```console -sudo systemctl start nixos-upgrade # sync, then rebuild -sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-x86_64 -``` +The "NixOS version" metadata shows `(restart to apply update)` whenever a generation has been built +and made the boot default without being activated — which is what the shutdown staging hook does, +and what `nixos-rebuild boot` does by hand. ## Where the logs are @@ -263,9 +249,8 @@ under `these N derivations will be built:`, per-derivation compiler output, and /var/log/coder-nixos/rebuild-latest.log # symlink to the most recent boot rebuild ``` -Scheduled upgrades are the exception: `nixos-upgrade.service` writes to the journal and nowhere -else, so nothing of theirs reaches the workspace UI. `journalctl -u nixos-upgrade` has the whole -story, and the "NixOS version" metadata on the workspace shows when one has staged a generation. +A rebuild you start yourself is the exception: it goes wherever you ran it, and to the journal, not +to the workspace UI. Only the boot path streams. Keeping compiler output out of the UI is not cosmetic. Coder caps agent logs at **1 MiB per agent**, shared across every log source, and exceeding it does not truncate — the log is marked diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index 7f6d03210..2c1d267d5 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -155,11 +155,11 @@ resource "coder_agent" "main" { timeout = 30 script = "coder stat disk --path $HOME" } - # Makes a staged generation visible. `coder.autoUpgrade.operation = "boot"` - # in the flake stages rather than activates, which otherwise looks exactly - # like updates being ignored. The command comes from the nix module -- - # metadata has to be declared on the agent, but what it means to be up to - # date is not this file's business. + # Makes a staged generation visible: a configuration that was built and made + # the boot default without being activated otherwise looks exactly like + # updates being ignored. The command comes from the nix module -- metadata + # has to be declared on the agent, but what it means to be up to date is not + # this file's business. metadata { key = "nixos" display_name = "NixOS version" From 6d47f254ae222d37fbbf093cbdaaa59b3d14b93f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 14:08:46 +0000 Subject: [PATCH 14/24] fix(aws-nixos): run stdbuf under _sudo, not around it `stdbuf -oL _sudo tee -a "$transcript"` cannot work: `_sudo` is a shell function and stdbuf execs what it is given, so it exited 127 with "failed to run command '_sudo'". That empties the middle of the rebuild pipeline, nixos-rebuild writes into a broken pipe, and the boot fails with an empty transcript and nothing in the workspace log but "nixos-rebuild switch failed". Introduced two commits ago by the `NIX_SUDO` -> `_sudo` change, where the variable used to expand to `sudo` or to nothing -- both things stdbuf can exec. Caught on a live instance, which is the only place it shows up: shellcheck, terraform validate and a rendered-script syntax check are all happy with it. `_sudo stdbuf -oL tee` instead, which also matches what the rest of the file does with redirections that need privilege. --- .../templates/aws-nixos/modules/nix/scripts/lifecycle.sh | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh index 6bd7cae8b..aa0d676a8 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh @@ -270,12 +270,16 @@ nix_apply() { _sudo install -m 0644 /dev/null "$transcript" _sudo ln -sfn "$transcript" "$NIX_LOG_DIR/rebuild-latest.log" - # stdbuf so output streams instead of arriving in one burst at the end. + # stdbuf so output streams instead of arriving in one burst at the end, and + # under _sudo rather than around it: _sudo is a shell function, and stdbuf + # execs what it is given. `stdbuf -oL _sudo tee` dies with "failed to run + # command '_sudo'", which empties the pipeline's middle, breaks the pipe + # under nixos-rebuild and fails the rebuild with an empty transcript. set +e _sudo nixos-rebuild "$operation" \ --flake "$NIX_FLAKE_DIR#$NIX_FLAKE_ATTR" \ --print-build-logs \ - 2>&1 | stdbuf -oL _sudo tee -a "$transcript" | nix_filter_log \ + 2>&1 | _sudo stdbuf -oL tee -a "$transcript" | nix_filter_log \ | while IFS= read -r line; do nix_log info "$line"; done rc=${PIPESTATUS[0]} set -e From ca834aa63d7bcc2df9cb8ed9b0e78f7923034218 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 14:20:14 +0000 Subject: [PATCH 15/24] docs(aws-nixos): name the attribute in the rebuild-by-hand example Verified on a live workspace: with no `#`, nixos-rebuild looks for a configuration named after the hostname, which on EC2 is the DHCP name. The note twenty lines up already said so; the example did not. --- registry/coder-labs/templates/aws-nixos/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 5e8fa6e61..d9fd995df 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -226,7 +226,7 @@ A workspace rebuilds from the flake when it **boots**, and that is the only sche pick up a change, restart the workspace, or run the rebuild yourself: ```console -sudo nixos-rebuild switch # /etc/nixos/flake.nix is found on its own +sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-x86_64 ``` Adding a timer is the configuration's business, not this template's — `system.autoUpgrade` is From d50a01c7b1c92b7d0df3dfa3a21cb382aad8616d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 14:45:02 +0000 Subject: [PATCH 16/24] refactor(aws-nixos): take the region's AZ and the instance catalog from modules Two parameters the template was doing by hand now come from the shared modules, which is also where the knowledge belongs. **The availability zone.** `"${module.aws-region.value}a"` becomes `module.aws-region.default_availability_zone`, added by coder/registry#1138. Worth being honest about what that buys: the module computes the same `region + "a"` string, so the behaviour is identical and it is still a guess -- AZ names are per-account aliases and some accounts are never offered the `a` one. What changes is that there is now a single place to fix it, instead of four templates mashing strings. aws-envbuilder still has the old form and a `# TODO: provide a way to pick the availability zone` to go with it. **The instance type.** The template's own `coder_parameter` and, more to the point, the seven-row `arch_map` next to it both go. coder/registry#1136's `aws-ec2-instance-type` publishes an `instances` catalog carrying `coder_arch` (amd64/arm64) and `ami` (x86_64/arm64) per type, which is two of the three spellings this template needs; the third is Nix's `aarch64`, derived from the second. So the comment that used to say the map existed "so the AMI architecture, coder_agent.arch and the flake attribute cannot disagree" is now true by construction rather than by maintenance. Everything under 4 GiB is excluded, which is the rule the old curated list was expressing: no swap, Nix store on the root volume, and a rebuild that compiles anything exhausts a 1-2 GiB instance. Both sources are `git::` refs, with `depth=1` so a `terraform init` clones 48 MiB of coder/registry rather than 92. Neither version exists on registry.coder.com yet -- aws-region 1.1.0 is merged but untagged (newest published is 1.0.31, which has no `default_availability_zone`), and #1136 is an open PR. Both carry a TODO, the README says so, and this template cannot be released until both are re-pointed. Note the parameter renames from `instance_type` to `aws_ec2_instance_type`, which is the module's name for it. Verified on a live deployment, both architectures for the first time: t3.medium x86_64 agent amd64 coder-workspace-x86_64 eu-west-3a t4g.medium aarch64 agent arm64 coder-workspace-aarch64 eu-west-3a both reaching lifecycle=ready with a healthy agent. --- .../coder-labs/templates/aws-nixos/README.md | 34 +++++--- .../coder-labs/templates/aws-nixos/main.tf | 85 ++++++++----------- 2 files changed, 58 insertions(+), 61 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index d9fd995df..03571343a 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -149,9 +149,9 @@ nixos-rebuild switch --flake 'github:your-org/config#coder-workspace-x86_64' ``` `$ARCH` in `flake_attr` is replaced with `x86_64` or `aarch64` to match the chosen instance type. -That keeps the AMI architecture, `coder_agent.arch` and the flake attribute in agreement — they are -all derived from one map in `main.tf`, so a Graviton instance type cannot accidentally boot an -x86 configuration. If you keep a single configuration instead, set `flake_attr` to a fixed name and +That keeps the AMI architecture, `coder_agent.arch` and the flake attribute in agreement — all +three are read off the selected instance type, so a Graviton instance type cannot accidentally boot +an x86 configuration. If you keep a single configuration instead, set `flake_attr` to a fixed name and only offer instance types of the matching architecture. Any reference `nixos-rebuild --flake` understands works, including `github:owner/repo`, @@ -278,14 +278,16 @@ volume is picked up on the next restart. ## Architecture support -Both `x86_64` and `arm64` (Graviton) instance types are offered. The AMI filter, -`coder_agent.arch` and the flake attribute are all derived from the instance type, so they cannot -disagree — but your flake must expose a configuration for the architecture you select. The -reference flake ships `coder-workspace-x86_64` and `coder-workspace-aarch64`. +Both `x86_64` and `arm64` (Graviton) instance types are offered. The instance type is a +[aws-ec2-instance-type](https://registry.coder.com/modules/coder/aws-ec2-instance-type) parameter, +and the AMI filter, `coder_agent.arch` and the flake attribute are all looked up from the same +catalog entry — so they cannot disagree. Your flake still has to expose a configuration for the +architecture you select; the reference flake ships `coder-workspace-x86_64` and +`coder-workspace-aarch64`. -The smallest instance type offered is `t3.medium` on purpose: the NixOS AMI configures no swap and -the Nix store shares the root volume, so a rebuild that has to compile anything will exhaust a -1–2 GiB instance. +Anything under 4 GiB is excluded on purpose: the NixOS AMI configures no swap and the Nix store +shares the root volume, so a rebuild that has to compile anything will exhaust a 1–2 GiB instance. +The smallest option is therefore `t3.medium`. ## Troubleshooting @@ -354,15 +356,23 @@ store, where every process on the workspace can read it. ## Extending the template -Three registry modules are included: +Five registry modules are included: +[aws-region](https://registry.coder.com/modules/coder/aws-region), +[aws-ec2-instance-type](https://registry.coder.com/modules/coder/aws-ec2-instance-type), [code-server](https://registry.coder.com/modules/coder/code-server), [jetbrains-gateway](https://registry.coder.com/modules/coder/jetbrains-gateway) and [git-config](https://registry.coder.com/modules/coder/git-config). +> [!NOTE] +> The first two are currently sourced from Git rather than the registry, because the versions this +> template needs are not published yet: `aws-region`'s `default_availability_zone` is merged but +> untagged, and `aws-ec2-instance-type` is still an open pull request. Both `source` lines carry a +> TODO and must be re-pointed at `registry.coder.com` before this template is released. + The first two push a dynamically linked binary into the workspace and exec it, so they work only because the reference flake sets `programs.nix-ld.enable = true` — remove that and both fail with a misleading "No such file or directory". Gateway is also told which architecture to fetch, from the -same instance-type map that picks the AMI, and is restricted to the IDEs JetBrains publishes an +same catalog entry that picks the AMI, and is restricted to the IDEs JetBrains publishes an `aarch64` backend for. > [!NOTE] diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index 2c1d267d5..3d2793e6f 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -11,8 +11,12 @@ terraform { } module "aws-region" { - source = "registry.coder.com/coder/aws-region/coder" - version = "~> 1.0" + # TODO: back to `registry.coder.com/coder/aws-region/coder` once 1.1.0 is + # published. `default_availability_zone` landed in coder/registry#1138 and + # the newest published version is still 1.0.31, which has only `value`. + # depth=1 because the source is the whole registry repo: 48 MiB rather + # than 92, on every `terraform init` the provisioner runs. + source = "git::https://github.com/coder/registry.git//registry/coder/modules/aws-region?ref=main&depth=1" default = "eu-west-3" } @@ -49,45 +53,30 @@ variable "nixos_release" { default = "26.05" } -data "coder_parameter" "instance_type" { - name = "instance_type" - display_name = "Instance type" - description = <<-EOT +module "aws-ec2-instance-type" { + # TODO: back to `registry.coder.com/coder/aws-ec2-instance-type/coder` once + # coder/registry#1136 merges and publishes. Until then this template cannot + # be released: it points at a branch, which is mutable. + source = "git::https://github.com/coder/registry.git//registry/coder/modules/aws-ec2-instance-type?ref=phorcys/aws-ec2-instance-type&depth=1" + + default = "t3.medium" + description = trimspace(<<-EOT The smallest option is t3.medium on purpose: the NixOS AMI configures no swap and the Nix store shares the root volume, so a rebuild that has to compile anything will exhaust a 1-2 GiB instance. EOT - default = "t3.medium" - mutable = false + ) - option { - name = "2 vCPU, 4 GiB RAM" - value = "t3.medium" - } - option { - name = "2 vCPU, 8 GiB RAM" - value = "t3.large" - } - option { - name = "4 vCPU, 16 GiB RAM" - value = "t3.xlarge" - } - option { - name = "8 vCPU, 32 GiB RAM" - value = "t3.2xlarge" - } - option { - name = "2 vCPU, 4 GiB RAM (Graviton)" - value = "t4g.medium" - } - option { - name = "2 vCPU, 8 GiB RAM (Graviton)" - value = "m7g.large" - } - option { - name = "4 vCPU, 16 GiB RAM (Graviton)" - value = "m7g.xlarge" - } + # Everything under 4 GiB, for the reason above. The rest of the general + # category is left alone -- the floor is the rule, not a curated list. + exclude = [ + "t3.nano", + "t3.micro", + "t3.small", + "t4g.nano", + "t4g.micro", + "t4g.small", + ] } data "coder_parameter" "root_volume_size" { @@ -212,18 +201,16 @@ module "git-config" { } locals { - # One map so the AMI architecture, coder_agent.arch and the flake attribute - # cannot disagree. - arch_map = { - "t3.medium" = { agent = "amd64", ami = "x86_64", attr = "x86_64" } - "t3.large" = { agent = "amd64", ami = "x86_64", attr = "x86_64" } - "t3.xlarge" = { agent = "amd64", ami = "x86_64", attr = "x86_64" } - "t3.2xlarge" = { agent = "amd64", ami = "x86_64", attr = "x86_64" } - "t4g.medium" = { agent = "arm64", ami = "arm64", attr = "aarch64" } - "m7g.large" = { agent = "arm64", ami = "arm64", attr = "aarch64" } - "m7g.xlarge" = { agent = "arm64", ami = "arm64", attr = "aarch64" } + instance = module.aws-ec2-instance-type.instances[module.aws-ec2-instance-type.value] + + # The AMI architecture, coder_agent.arch and the flake attribute all come + # from the instance type, so they cannot disagree. The module publishes the + # first two spellings; the third is Nix's, and is the same distinction. + arch = { + agent = local.instance.coder_arch + ami = local.instance.ami + attr = local.instance.ami == "arm64" ? "aarch64" : "x86_64" } - arch = local.arch_map[data.coder_parameter.instance_type.value] } # Everything about the flake: the checkout, the boot-time rebuild and the @@ -254,8 +241,8 @@ module "amazon-init" { resource "aws_instance" "dev" { ami = data.aws_ami.nixos.id - availability_zone = "${module.aws-region.value}a" - instance_type = data.coder_parameter.instance_type.value + availability_zone = module.aws-region.default_availability_zone + instance_type = module.aws-ec2-instance-type.value user_data = module.amazon-init.user_data # The agent token is inside user-data and rotates on every workspace start, From 57ac6364b09357f7833c9586ee5f566424e64e83 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 16:39:25 +0000 Subject: [PATCH 17/24] refactor(aws-nixos): offer every size, and name the families Drops the `exclude` list. Nothing is filtered out of the picker now: it runs from t3.nano up, and the sub-4 GiB options will fail the first time they have to build something that is not in the binary cache. `t3.medium` is still the default and the description still says why. `include = ["t3", "t4g", "m7g"]` in the same breath, because the module was rewritten under this template while the change was in flight: `type_category` and `exclude` are gone, replaced by an `include` of instance *families* defaulting to `["t3"]`. Passing neither would have quietly made the template x86-only -- and since the AMI, the agent's arch and the flake attribute all follow the instance type, that is not a cosmetic loss, it is the Graviton half of the template disappearing from the menu. Checked against the deployment rather than assumed: 22 options, 7 amd64 and 15 arm64, labelled with the architecture. Also fixes a sentence I broke in the last commit: "The first two push a dynamically linked binary" stopped being true when aws-region and aws-ec2-instance-type went to the top of that list. --- .../coder-labs/templates/aws-nixos/README.md | 17 +++++++++----- .../coder-labs/templates/aws-nixos/main.tf | 23 +++++++++---------- 2 files changed, 22 insertions(+), 18 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 03571343a..aa77cb839 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -285,9 +285,14 @@ catalog entry — so they cannot disagree. Your flake still has to expose a conf architecture you select; the reference flake ships `coder-workspace-x86_64` and `coder-workspace-aarch64`. -Anything under 4 GiB is excluded on purpose: the NixOS AMI configures no swap and the Nix store -shares the root volume, so a rebuild that has to compile anything will exhaust a 1–2 GiB instance. -The smallest option is therefore `t3.medium`. +Every size of `t3`, `t4g` and `m7g` is offered — nothing is filtered out, so the list runs from +`t3.nano` upwards. The default is `t3.medium` because that is the smallest one that works: the +NixOS AMI configures no swap and the Nix store shares the root volume, so a rebuild that has to +compile anything will exhaust a 1–2 GiB instance. The smaller options are selectable and will fail +the first time they have to build something that is not in the binary cache. + +Add families with the module's `include` — it defaults to `t3` alone, so the Graviton families are +named explicitly here to keep both architectures on offer. ## Troubleshooting @@ -369,9 +374,9 @@ Five registry modules are included: > untagged, and `aws-ec2-instance-type` is still an open pull request. Both `source` lines carry a > TODO and must be re-pointed at `registry.coder.com` before this template is released. -The first two push a dynamically linked binary into the workspace and exec it, so they work only -because the reference flake sets `programs.nix-ld.enable = true` — remove that and both fail with a -misleading "No such file or directory". Gateway is also told which architecture to fetch, from the +code-server and JetBrains Gateway push a dynamically linked binary into the workspace and exec it, +so they work only because the reference flake sets `programs.nix-ld.enable = true` — remove that +and both fail with a misleading "No such file or directory". Gateway is also told which architecture to fetch, from the same catalog entry that picks the AMI, and is restricted to the IDEs JetBrains publishes an `aarch64` backend for. diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index 3d2793e6f..5a72e074b 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -61,21 +61,20 @@ module "aws-ec2-instance-type" { default = "t3.medium" description = trimspace(<<-EOT - The smallest option is t3.medium on purpose: the NixOS AMI configures no - swap and the Nix store shares the root volume, so a rebuild that has to - compile anything will exhaust a 1-2 GiB instance. + t3.medium is the smallest that works: the NixOS AMI configures no swap and + the Nix store shares the root volume, so a rebuild that has to compile + anything will exhaust a 1-2 GiB instance. EOT ) - # Everything under 4 GiB, for the reason above. The rest of the general - # category is left alone -- the floor is the rule, not a curated list. - exclude = [ - "t3.nano", - "t3.micro", - "t3.small", - "t4g.nano", - "t4g.micro", - "t4g.small", + # Nothing is excluded, but the families are named: the module offers `t3` + # alone by default, and this template supports both architectures -- the + # AMI, the agent and the flake attribute all follow the instance type, so + # dropping the Graviton families would quietly make it x86-only. + include = [ + "t3", + "t4g", + "m7g", ] } From 313777cd28f1ea0c8b1f558ed60d6d8c10080cb6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 18:50:16 +0000 Subject: [PATCH 18/24] refactor(coder-labs/aws-nixos): follow the flake's renamed configurations The reference flake's outputs are now `coder-workspace-ec2-x86_64` and `coder-workspace-ec2-aarch64` -- every configuration in it imports hardware/ec2.nix, so the generic names were a promise it did not keep. `flake_attr` defaults here follow, in the template and in the nix module. A default is only read when a template is created or a variable is left unset, so an existing template keeps the old stored value and its workspaces would fail their next rebuild. Documented in the README, with how to update it. --- .../coder-labs/templates/aws-nixos/README.md | 18 +++++++++++++----- .../coder-labs/templates/aws-nixos/main.tf | 2 +- .../templates/aws-nixos/modules/nix/main.tf | 2 +- 3 files changed, 15 insertions(+), 7 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index aa77cb839..3baa6650e 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -145,7 +145,7 @@ committed: a Git flake reference only ever sees committed files. selection you would make by hand: ```console -nixos-rebuild switch --flake 'github:your-org/config#coder-workspace-x86_64' +nixos-rebuild switch --flake 'github:your-org/config#coder-workspace-ec2-x86_64' ``` `$ARCH` in `flake_attr` is replaced with `x86_64` or `aarch64` to match the chosen instance type. @@ -154,6 +154,14 @@ three are read off the selected instance type, so a Graviton instance type canno an x86 configuration. If you keep a single configuration instead, set `flake_attr` to a fixed name and only offer instance types of the matching architecture. +> [!NOTE] +> The reference flake's configurations were renamed from `coder-workspace-` to +> `coder-workspace-ec2-`, and the `flake_attr` default here follows. A default is only read +> when a template is created or a variable is left unset, so a template already pushed with the old +> value keeps it — update the stored `flake_attr` (push again with an explicit value, or edit it in +> the template settings) before workspaces rebuild against the renamed flake, otherwise the next +> rebuild fails on a missing attribute. + Any reference `nixos-rebuild --flake` understands works, including `github:owner/repo`, `git+ssh://` for private repositories, and `?dir=subdir` for a flake in a subdirectory. The configuration must be committed: a Git flake reference only ever sees committed files. @@ -198,7 +206,7 @@ The configuration is a git checkout at `/etc/nixos`, owned by the workspace user, and that is what the template builds. So the command is the ordinary one: ```console -sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-x86_64 +sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-ec2-x86_64 ``` No overrides, no `--impure`, no injected inputs — what you get by hand is @@ -226,7 +234,7 @@ A workspace rebuilds from the flake when it **boots**, and that is the only sche pick up a change, restart the workspace, or run the rebuild yourself: ```console -sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-x86_64 +sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-ec2-x86_64 ``` Adding a timer is the configuration's business, not this template's — `system.autoUpgrade` is @@ -282,8 +290,8 @@ Both `x86_64` and `arm64` (Graviton) instance types are offered. The instance ty [aws-ec2-instance-type](https://registry.coder.com/modules/coder/aws-ec2-instance-type) parameter, and the AMI filter, `coder_agent.arch` and the flake attribute are all looked up from the same catalog entry — so they cannot disagree. Your flake still has to expose a configuration for the -architecture you select; the reference flake ships `coder-workspace-x86_64` and -`coder-workspace-aarch64`. +architecture you select; the reference flake ships `coder-workspace-ec2-x86_64` and +`coder-workspace-ec2-aarch64`. Every size of `t3`, `t4g` and `m7g` is offered — nothing is filtered out, so the list runs from `t3.nano` upwards. The default is `t3.medium` because that is the smallest one that works: the diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index 5a72e074b..c9764a065 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -40,7 +40,7 @@ variable "flake_ref" { variable "flake_attr" { description = "`nixosConfigurations` attribute to build. `$ARCH` is replaced with `x86_64` or `aarch64` to match the instance type." type = string - default = "coder-workspace-$ARCH" + default = "coder-workspace-ec2-$ARCH" } variable "nixos_release" { diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf index 9cd1a9d70..550aa6b80 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf @@ -22,7 +22,7 @@ variable "flake_ref" { variable "flake_attr" { description = "`nixosConfigurations` attribute to build. `$ARCH` is replaced with `arch`." type = string - default = "coder-workspace-$ARCH" + default = "coder-workspace-ec2-$ARCH" } variable "arch" { From 6dd42e71730767e92f21a120b68aec9d07816e47 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 18:53:20 +0000 Subject: [PATCH 19/24] fix(coder-labs/aws-nixos): the instance catalog renamed ami to arch --- registry/coder-labs/templates/aws-nixos/main.tf | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index c9764a065..f668d9dea 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -207,8 +207,8 @@ locals { # first two spellings; the third is Nix's, and is the same distinction. arch = { agent = local.instance.coder_arch - ami = local.instance.ami - attr = local.instance.ami == "arm64" ? "aarch64" : "x86_64" + ami = local.instance.arch + attr = local.instance.arch == "arm64" ? "aarch64" : "x86_64" } } From ac4ba33483994dc624e30893e66f97f2ab764a98 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 19:43:56 +0000 Subject: [PATCH 20/24] refactor(coder-labs/aws-nixos): take aws-ec2-instance-type from the registry coder/registry#1136 merged and 1.0.0 is published, so the module comes from registry.coder.com instead of a branch of this repository -- which was mutable, and the reason the template could not be released. aws-region stays on Git: `default_availability_zone` is tagged as 1.1.0 but the registry still serves 1.0.31, whose only output is `value`. The pin moves from `main` to the release tag, so it is at least immutable. --- registry/coder-labs/templates/aws-nixos/README.md | 8 ++++---- registry/coder-labs/templates/aws-nixos/main.tf | 11 +++++------ 2 files changed, 9 insertions(+), 10 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 3baa6650e..268838212 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -377,10 +377,10 @@ Five registry modules are included: [git-config](https://registry.coder.com/modules/coder/git-config). > [!NOTE] -> The first two are currently sourced from Git rather than the registry, because the versions this -> template needs are not published yet: `aws-region`'s `default_availability_zone` is merged but -> untagged, and `aws-ec2-instance-type` is still an open pull request. Both `source` lines carry a -> TODO and must be re-pointed at `registry.coder.com` before this template is released. +> `aws-region` is still sourced from Git, because the version this template needs is not served by +> the registry yet: `default_availability_zone` landed in 1.1.0 and the newest published version is +> 1.0.31. The pin is the release tag, so it is immutable, but the `source` line carries a TODO and +> must be re-pointed at `registry.coder.com` before this template is released. code-server and JetBrains Gateway push a dynamically linked binary into the workspace and exec it, so they work only because the reference flake sets `programs.nix-ld.enable = true` — remove that diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index f668d9dea..a6e70ae6e 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -13,10 +13,11 @@ terraform { module "aws-region" { # TODO: back to `registry.coder.com/coder/aws-region/coder` once 1.1.0 is # published. `default_availability_zone` landed in coder/registry#1138 and - # the newest published version is still 1.0.31, which has only `value`. + # is tagged, but the newest version the registry serves is 1.0.31, which + # has only `value`. The tag is at least immutable, unlike a branch. # depth=1 because the source is the whole registry repo: 48 MiB rather # than 92, on every `terraform init` the provisioner runs. - source = "git::https://github.com/coder/registry.git//registry/coder/modules/aws-region?ref=main&depth=1" + source = "git::https://github.com/coder/registry.git//registry/coder/modules/aws-region?ref=release/coder/aws-region/v1.1.0&depth=1" default = "eu-west-3" } @@ -54,10 +55,8 @@ variable "nixos_release" { } module "aws-ec2-instance-type" { - # TODO: back to `registry.coder.com/coder/aws-ec2-instance-type/coder` once - # coder/registry#1136 merges and publishes. Until then this template cannot - # be released: it points at a branch, which is mutable. - source = "git::https://github.com/coder/registry.git//registry/coder/modules/aws-ec2-instance-type?ref=phorcys/aws-ec2-instance-type&depth=1" + source = "registry.coder.com/coder/aws-ec2-instance-type/coder" + version = "~> 1.0" default = "t3.medium" description = trimspace(<<-EOT From 7d4accf98e7aa3b9773e26ae276eef84f080b988 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 19:51:45 +0000 Subject: [PATCH 21/24] refactor(coder-labs/aws-nixos): take aws-region from the registry too 1.1.0 is served now, so `default_availability_zone` is available from registry.coder.com and the last Git source is gone. The template no longer depends on an unreleased module, and no longer shallow-clones this repository on every provisioner init. --- registry/coder-labs/templates/aws-nixos/README.md | 6 ------ registry/coder-labs/templates/aws-nixos/main.tf | 10 +++------- 2 files changed, 3 insertions(+), 13 deletions(-) diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 268838212..1741ff066 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -376,12 +376,6 @@ Five registry modules are included: [jetbrains-gateway](https://registry.coder.com/modules/coder/jetbrains-gateway) and [git-config](https://registry.coder.com/modules/coder/git-config). -> [!NOTE] -> `aws-region` is still sourced from Git, because the version this template needs is not served by -> the registry yet: `default_availability_zone` landed in 1.1.0 and the newest published version is -> 1.0.31. The pin is the release tag, so it is immutable, but the `source` line carries a TODO and -> must be re-pointed at `registry.coder.com` before this template is released. - code-server and JetBrains Gateway push a dynamically linked binary into the workspace and exec it, so they work only because the reference flake sets `programs.nix-ld.enable = true` — remove that and both fail with a misleading "No such file or directory". Gateway is also told which architecture to fetch, from the diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index a6e70ae6e..83653525e 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -11,13 +11,9 @@ terraform { } module "aws-region" { - # TODO: back to `registry.coder.com/coder/aws-region/coder` once 1.1.0 is - # published. `default_availability_zone` landed in coder/registry#1138 and - # is tagged, but the newest version the registry serves is 1.0.31, which - # has only `value`. The tag is at least immutable, unlike a branch. - # depth=1 because the source is the whole registry repo: 48 MiB rather - # than 92, on every `terraform init` the provisioner runs. - source = "git::https://github.com/coder/registry.git//registry/coder/modules/aws-region?ref=release/coder/aws-region/v1.1.0&depth=1" + # 1.1.0 for `default_availability_zone`. + source = "registry.coder.com/coder/aws-region/coder" + version = "~> 1.1" default = "eu-west-3" } From 481b07169d7b9c972d156a1daf16f6e4272dce0c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 23:32:19 +0000 Subject: [PATCH 22/24] Harden and simplify AWS NixOS flake lifecycle module --- .../templates/aws-nixos/modules/nix/README.md | 81 ++--------- .../templates/aws-nixos/modules/nix/main.tf | 53 +++++--- .../aws-nixos/modules/nix/main.tftest.hcl | 46 +++++++ .../modules/nix/scripts/boot.sh.tftpl | 34 ++--- .../modules/nix/scripts/lifecycle.sh | 127 +++--------------- .../modules/nix/scripts/lifecycle.test.sh | 67 +++++++++ 6 files changed, 181 insertions(+), 227 deletions(-) create mode 100644 registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl create mode 100755 registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.test.sh diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/README.md b/registry/coder-labs/templates/aws-nixos/modules/nix/README.md index b7ef1c1df..9f9891f3c 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/README.md @@ -1,83 +1,18 @@ # nix -The flake lifecycle: keep a checkout in sync with a Git remote, decide whether -the running system is out of date, and rebuild it. - -The boot path is a **string** the caller hands to whatever runs scripts on the -machine. On AWS the caller is [`../amazon-init`](../amazon-init/README.md), but -nothing depends on that. +This local module renders a root-run `boot_script` before the Coder agent starts. It clones a Git flake and runs `nixos-rebuild switch` when the checkout, attribute, or active generation changes. ```tf module "nix" { - source = "./modules/nix" - - agent_id = try(coder_agent.main[0].id, "") - flake_ref = "git+https://github.com/coder/nixos-example-flake?ref=main" - arch = "x86_64" + source = "./modules/nix" + flake_ref = "git+https://github.com/coder/nixos-example-flake?ref=main" + flake_attr = "coder-workspace-ec2-$ARCH" + arch = "x86_64" } ``` -## The flake reference - -`flake_ref` is what you would pass to `nixos-rebuild --flake`: a -[flake reference](https://nix.dev/manual/nix/latest/command-ref/new-cli/nix3-flake#flake-references), -with the branch in `?ref=` and the remote's default branch when it is absent — -resolved on the instance, since Terraform cannot know it without talking to the -remote. - -`$ARCH` in `flake_attr` is replaced with `arch`, so one template can offer both -architectures without the attribute and the machine disagreeing. - -## What runs where - -`boot_script` runs as root on every boot, before the agent is started. It syncs -the checkout, and rebuilds only if the configuration changed: - -- a **clean** checkout is fast-forwarded to the remote -- a **dirty** one, or one carrying local commits, is left alone and built as it - is — someone is working on it -- a checkout on a different branch than requested says so rather than silently - building the wrong thing - -A `flock` in `state_dir` is the lock everything rebuilding this machine should -take, including the configuration's own upgrade timer, so that two rebuilds -never race for the system profile. - -## Logging - -`scripts/lifecycle.sh` sends output through a single `nix_log ` -hook. Define it and progress goes wherever you want; leave it undefined and it -prints to stdout. - -`boot_script` sets that hook up from `CODER_LOG_LIBRARY` when the bootstrapper -provides one — that is how output reaches the workspace UI before an agent -exists — and falls back to plain `echo` when it does not. - -Raw `nix` output is filtered before it is logged: store-path lists, -per-derivation build output and lock-file noise are dropped, and list headers -that promised a list are rewritten as sentences. Transcripts in `log_dir` keep -everything, unfiltered. - -## Inputs and outputs - -`flake_dir`, `state_dir` and `log_dir` default to `/etc/nixos`, -`/var/lib/coder-nixos` and `/var/log/coder-nixos`, and both scripts take them -from here — the paths are defined once. - -Outputs exist for the things a caller genuinely cannot do itself: -`boot_script`, `flake_uri` and `flake_attr` for display, `log_dir` and -`flake_dir` for pointing people at, and `version_command` for a `coder_agent` -metadata block — which has to be declared inline on the agent, though what it -means for a NixOS machine to be up to date does not belong in a template. - -`values` passes straight through to `values` on the bootstrapper, so a caller -has one wire for runtime facts and a Nix-specific fact would have an obvious -home. There are none today: the configuration already knows its checkout, -attribute and directories, because it is what sets them. +References accept HTTP(S) or SSH Git URLs, optional `git+` and `?ref=`. Without `ref`, Git follows the default branch. `$ARCH` expands to `arch`. HTTP URL userinfo is rejected: migrate embedded passwords or tokens to root-managed authentication. The instance requires outbound Git and Nix input/substituter access, root privileges, systemd, and NixOS. -## Moving this to the registry +A clean checkout fast-forwards; tracked edits and local commits remain untouched. Untracked files do not trigger builds: Git flakes ignore them. The `state_dir` lock prevents races only with callers that take it. Rebuild transcripts live in `log_dir`. First-boot failures may require AWS logs before the agent exists. -It is a self-contained Terraform module already. Publishing it as -`registry.coder.com/coder/nix` needs `.tftest.hcl` coverage and a decision -about `nixos-rebuild` on non-NixOS hosts (`nix profile` would be the -equivalent), not untangling. +The caller runs `boot_script` as root before agent startup. Display outputs include `flake_uri`, `flake_attr`, `flake_dir`, `log_dir` and `version_command`; `values` passes through unchanged. diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf index 550aa6b80..f3dc3496d 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf @@ -14,8 +14,8 @@ variable "flake_ref" { type = string validation { - condition = can(regex("^(git\\+)?(https?|ssh)://", var.flake_ref)) - error_message = "flake_ref must be an http(s) or ssh Git URL, optionally prefixed with git+." + condition = can(regex("^(git\\+)?(https?|ssh)://", var.flake_ref)) && !can(regex("^(git\\+)?https?://[^/?#]*@", var.flake_ref)) && !can(regex("[[:cntrl:]]", var.flake_ref)) + error_message = "flake_ref must be an http(s) or ssh Git URL without HTTP credentials or control characters. Use root-managed Git authentication instead of URL userinfo." } } @@ -23,6 +23,11 @@ variable "flake_attr" { description = "`nixosConfigurations` attribute to build. `$ARCH` is replaced with `arch`." type = string default = "coder-workspace-ec2-$ARCH" + + validation { + condition = !can(regex("[[:cntrl:]]", var.flake_attr)) + error_message = "flake_attr must not contain control characters." + } } variable "arch" { @@ -54,42 +59,43 @@ variable "flake_dir" { description = "Checkout to build. Owned by the workspace user so the configuration can be edited in place." type = string default = "/etc/nixos" + + validation { + condition = !can(regex("[[:cntrl:]]", var.flake_dir)) + error_message = "flake_dir must not contain control characters." + } } variable "state_dir" { description = "Revision marker and rebuild lock." type = string default = "/var/lib/coder-nixos" + + validation { + condition = !can(regex("[[:cntrl:]]", var.state_dir)) + error_message = "state_dir must not contain control characters." + } } variable "log_dir" { description = "Rebuild transcripts." type = string default = "/var/log/coder-nixos" + + validation { + condition = !can(regex("[[:cntrl:]]", var.log_dir)) + error_message = "log_dir must not contain control characters." + } } locals { - # `git clone` is what runs on the instance, so reduce the reference to a - # plain remote: strip a `git+` scheme prefix and any query string. flake_url = replace(replace(var.flake_ref, "/^git\\+/", ""), "/\\?.*$/", "") - # An absent `?ref=` means the remote's default branch, resolved on the - # instance -- Terraform cannot know it without talking to the remote. flake_branch = try(regex("[?&]ref=([^&#]+)", var.flake_ref)[0], "") flake_attr = replace(var.flake_attr, "$ARCH", var.arch) lifecycle_sh = file("${path.module}/scripts/lifecycle.sh") - - script_args = { - LIFECYCLE_SH = local.lifecycle_sh - ARG_FLAKE_URL = local.flake_url - ARG_FLAKE_BRANCH = local.flake_branch - ARG_FLAKE_ATTR = local.flake_attr - ARG_FLAKE_DIR = var.flake_dir - ARG_STATE_DIR = var.state_dir - ARG_LOG_DIR = var.log_dir - } } output "values" { @@ -99,7 +105,15 @@ output "values" { output "boot_script" { description = "Applies the flake. Hand this to whatever runs a script on every boot; it expects to run as root." - value = templatefile("${path.module}/scripts/boot.sh.tftpl", local.script_args) + value = templatefile("${path.module}/scripts/boot.sh.tftpl", { + lifecycle_sh = local.lifecycle_sh + flake_url = base64encode(local.flake_url) + flake_branch = base64encode(local.flake_branch) + flake_attr = base64encode(local.flake_attr) + flake_dir = base64encode(var.flake_dir) + state_dir = base64encode(var.state_dir) + log_dir = base64encode(var.log_dir) + }) } output "flake_uri" { @@ -128,10 +142,7 @@ output "version_command" { staged but not yet booted. For a `coder_agent` metadata block, which has to be declared inline on the agent. EOT - # /run/current-system is the activated system; /run/booted-system is what - # the kernel booted and still points at the previous generation after a - # switch, which would mark every new workspace as needing a restart. - value = <<-EOT + value = <<-EOT version=$(nixos-version 2>/dev/null || echo unknown) if [ "$(readlink -f /run/current-system)" = "$(readlink -f /nix/var/nix/profiles/system)" ]; then echo "$version" diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl new file mode 100644 index 000000000..6b20b85af --- /dev/null +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl @@ -0,0 +1,46 @@ +run "safe_render" { + command = plan + + variables { + flake_ref = "git+ssh://git@example.org/flake?ref=feature" + flake_attr = "host-$ARCH'$(touch /tmp/should-not-run)" + flake_dir = "/etc/nixos'$(touch /tmp/should-not-run)" + } + + assert { + condition = output.flake_uri == "ssh://git@example.org/flake?ref=feature#host-x86_64'$(touch /tmp/should-not-run)" + error_message = "SSH userinfo or reference parsing changed." + } + assert { + condition = strcontains(output.boot_script, base64encode("/etc/nixos'$(touch /tmp/should-not-run)")) && !strcontains(output.boot_script, "FLAKE_DIR='/etc/nixos'") + error_message = "The boot script must encode untrusted arguments." + } +} + +run "default_branch" { + command = plan + variables { + flake_ref = "https://example.org/flake" + arch = "aarch64" + } + assert { + condition = output.flake_uri == "https://example.org/flake#coder-workspace-ec2-aarch64" + error_message = "The default branch or ARM attribute changed." + } +} + +run "reject_http_userinfo" { + command = plan + variables { + flake_ref = "git+https://user:token@example.org/flake?ref=main" + } + expect_failures = [var.flake_ref] +} + +run "reject_control_characters" { + command = plan + variables { + flake_ref = "https://example.org/flake?ref=main\nmalicious" + } + expect_failures = [var.flake_ref] +} diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl index 00ba18a0c..cdce3bae8 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/boot.sh.tftpl @@ -1,32 +1,18 @@ #!/usr/bin/env bash -# Applies the flake on every boot, as root. -# -# Handed to whatever bootstraps the instance as an opaque string. On AWS that -# is the amazon-init module, which runs this after the agent handoff exists -# and before the agent is started -- so everything here happens while the -# workspace is still building, and the agent never sees a half-applied system. -# -# Runs on every boot, so it must be idempotent. -# -E, because most of what runs here runs inside a function: without errtrace -# the ERR trap below is not inherited, and a failure in the lifecycle library -# ends the script silently -- the exit status reaches the caller, but nothing -# is ever said about it in the workspace log. set -Eeuo pipefail -FLAKE_URL='${ARG_FLAKE_URL}' -FLAKE_BRANCH='${ARG_FLAKE_BRANCH}' -FLAKE_ATTR='${ARG_FLAKE_ATTR}' - -FLAKE_DIR='${ARG_FLAKE_DIR}' -STATE_DIR='${ARG_STATE_DIR}' -LOG_DIR='${ARG_LOG_DIR}' +# The dot sentinel preserves trailing newlines through Bash command substitution. +decode_arg() { local value; value="$(printf '%s' "$1" | base64 -d; printf '.')"; printf '%s' "$${value%.}"; } +FLAKE_URL=$(decode_arg '${flake_url}') +FLAKE_BRANCH=$(decode_arg '${flake_branch}') +FLAKE_ATTR=$(decode_arg '${flake_attr}') +FLAKE_DIR=$(decode_arg '${flake_dir}') +STATE_DIR=$(decode_arg '${state_dir}') +LOG_DIR=$(decode_arg '${log_dir}') install -d -m 0755 "$STATE_DIR" "$LOG_DIR" -# Optional. A bootstrapper that can stream to the workspace UI before the -# agent exists points CODER_LOG_LIBRARY at a library providing these; without -# one, output still reaches whatever captured this script's stdout. if [ -r "$${CODER_LOG_LIBRARY:-}" ]; then # shellcheck source=/dev/null . "$CODER_LOG_LIBRARY" @@ -50,7 +36,7 @@ NIX_FLAKE_ATTR="$FLAKE_ATTR" NIX_STATE_DIR="$STATE_DIR" NIX_LOG_DIR="$LOG_DIR" -${LIFECYCLE_SH} +${lifecycle_sh} TRANSCRIPT="$LOG_DIR/rebuild-$(date -u +%Y%m%dT%H%M%SZ).log" @@ -81,8 +67,6 @@ if ! nix_apply switch "$TRANSCRIPT"; then fi nix_record_rev "$REV" -# The workspace user exists now, so systemd-tmpfiles has set the flake -# directory's owner; make the tree match it. nix_own_checkout coder_log info "Switch complete." || true diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh index aa0d676a8..6ac03de63 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.sh @@ -1,12 +1,7 @@ # shellcheck shell=bash -# -# Nix flake lifecycle: sync a checkout, decide, apply. No EC2 and no Coder API -# calls. Inputs and contract: see README.md in this directory. export NIX_CONFIG="experimental-features = nix-command flakes" -# Callers run as root from user-data, and by hand as the workspace user, so -# privilege handling belongs here rather than in each of them. _sudo() { if [ "$(id -u)" -eq 0 ]; then "$@" @@ -15,38 +10,29 @@ _sudo() { fi } -# --------------------------------------------------------------------------- -# logging -# --------------------------------------------------------------------------- -# -# Everything below writes through nix_log, which the caller is expected to -# define -- see README.md. The stub is what a machine with no Coder runtime -# gets, and it is guarded rather than unconditional so that a caller which -# has already defined the real one wins. command -v nix_log > /dev/null 2>&1 || nix_log() { shift printf '%s\n' "$*" } +nix_redact_url() { + printf '%s' "$1" | sed -E 's#(https?://)[^/@[:space:]]+@#\1[redacted]@#g' +} + nix_strip_ansi() { sed -e 's/\x1b\[[0-9;]*[a-zA-Z]//g' -e 's/\r$//' } -# Drop the things that are noise rather than progress; everything else reaches -# the caller as nix wrote it. Nix's plain output is already the readable -# output -- every --log-format is byte-identical off a TTY. nix_filter_log() { local line prefix while IFS= read -r line || [ -n "$line" ]; do + line=$(nix_redact_url "$line") case "$line" in *$'\033'*) line=$(printf '%s' "$line" | nix_strip_ansi) ;; esac case "$line" in "") continue ;; - # The enumerated store paths under "these N derivations will be built:" " "*/nix/store/*) continue ;; - # Those list headers end in a colon promising the list we just dropped, - # which reads as truncated output. Restate them as complete sentences. "these "*" will be "*: | "this "*" will be "*:) case "$line" in "these "*) line=${line#these } ;; @@ -55,11 +41,7 @@ nix_filter_log() { printf '%s\n' "${line%:}" continue ;; - # git's fetch progress, including the indented ref-update line. "remote: "* | "From "* | " "*".."*"->"*) continue ;; - # Per-derivation build output, "> text". The "building '...'" - # line already marks it and the text is in the transcript. Requiring a - # space-free prefix stops this eating lines that contain "> ". *"> "*) prefix=${line%%> *} case "$prefix" in @@ -69,11 +51,6 @@ nix_filter_log() { ;; esac - # Nix writes its progress in lowercase ("building the system - # configuration..."), and in the workspace UI those lines sit among - # sentences from every other log source. Capitalise the first letter -- - # but only when the first word is a plain word, so that program names - # ("nixos-rebuild: ...") and paths stay exactly as they were written. case "${line%% *}" in [a-z]*[!a-zA-Z:]*) ;; [a-z]*) line="${line^}" ;; @@ -83,60 +60,29 @@ nix_filter_log() { done } -# Runs a command with its output filtered into the log, and returns the -# command's own status rather than the filter's. -# -# Not written as a pipeline: under `set -e` a failing pipeline ends the script -# there and then, before the caller can look at PIPESTATUS. That is how a -# clone of a flake reference that does not exist used to end a boot with -# nothing in the workspace log but "Cloning ..." -- the error message two -# lines further down was unreachable. nix_run_logged() { local rc=0 out line out="$(mktemp)" "$@" > "$out" 2>&1 || rc=$? - # Through nix_log rather than stdout: what git has to say about a clone it - # could not do is the whole explanation, and stdout here is the instance's - # journal, which is exactly the place the user cannot reach. nix_filter_log < "$out" | while IFS= read -r line; do nix_log info "$line"; done rm -f "$out" return "$rc" } -# --------------------------------------------------------------------------- -# the checkout -# --------------------------------------------------------------------------- -# -# The configuration is a real git checkout at NIX_FLAKE_DIR (/etc/nixos), and -# that is what gets built. It is what makes a bare `nixos-rebuild switch` -# work, since nixos-rebuild looks for /etc/nixos/flake.nix on its own, and it -# means the configuration running the machine is something the user can read -# and edit. -# -# Local work is never discarded: we fast-forward only a clean checkout sitting -# on its tracking branch. A workspace whose configuration silently reverted on -# restart would be worse than one that drifts. nix_checkout_dirty() { - [ -n "$(_sudo git -C "$NIX_FLAKE_DIR" status --porcelain 2> /dev/null | head -1)" ] + [ -n "$(_sudo git -C "$NIX_FLAKE_DIR" status --porcelain --untracked-files=no 2> /dev/null | head -1)" ] } nix_checkout_rev() { _sudo git -C "$NIX_FLAKE_DIR" rev-parse HEAD 2> /dev/null || true } -# Keep the whole tree owned by whoever owns the directory. We clone and fetch -# as root, which otherwise leaves .git root-owned inside a user-owned -# directory -- the user can edit files but every git command fails on -# .git/index.lock. nix_own_checkout() { local owner owner=$(stat -c %U "$NIX_FLAKE_DIR" 2> /dev/null || echo root) _sudo chown -R "$owner" "$NIX_FLAKE_DIR" 2> /dev/null || true } -# The branch the checkout is actually on. With no `?ref=` in the flake -# reference there is nothing to ask but the checkout itself, and after a clone -# that is the remote's default branch. nix_checkout_branch() { _sudo git -C "$NIX_FLAKE_DIR" rev-parse --abbrev-ref HEAD 2> /dev/null || true } @@ -145,11 +91,8 @@ nix_sync_checkout() { local url="$1" branch="${2:-}" upstream_rev local_rev current_branch rc if [ ! -e "$NIX_FLAKE_DIR/flake.nix" ]; then - nix_log info "Cloning $url into $NIX_FLAKE_DIR" + nix_log info "Cloning $(nix_redact_url "$url") into $NIX_FLAKE_DIR" _sudo install -d -m 0755 "$NIX_FLAKE_DIR" - # Clone into a temporary directory and move the contents, because the - # directory already exists (systemd-tmpfiles creates it) and git refuses - # to clone into a non-empty one. local tmp tmp="$(_sudo mktemp -d)" rc=0 @@ -159,20 +102,18 @@ nix_sync_checkout() { nix_run_logged _sudo git clone "$url" "$tmp/repo" || rc=$? fi if [ "$rc" -ne 0 ]; then - nix_log error "Could not clone $url${branch:+ (branch $branch)} (git exited $rc)" + nix_log error "Could not clone $(nix_redact_url "$url")${branch:+ (branch $branch)} (git exited $rc)" nix_log error "Check the flake reference the template was pushed with: the repository has to exist and be readable from this instance." _sudo rm -rf "$tmp" return 1 fi - _sudo sh -c "cd '$tmp/repo' && tar cf - ." | _sudo tar xf - -C "$NIX_FLAKE_DIR" + _sudo tar -C "$tmp/repo" -cf - . | _sudo tar -C "$NIX_FLAKE_DIR" -xf - _sudo rm -rf "$tmp" nix_own_checkout return 0 fi current_branch="$(nix_checkout_branch)" - # An empty branch means "whatever this checkout tracks", which after the - # clone above is the remote's default. [ -n "$branch" ] || branch="$current_branch" if nix_checkout_dirty; then @@ -180,17 +121,17 @@ nix_sync_checkout() { return 0 fi - # Asking for a different branch than the checkout is on is not something to - # resolve silently: the fast-forward below would fail its ancestry test and - # report local commits, which is not what happened. if [ -n "$current_branch" ] && [ "$current_branch" != "$branch" ]; then nix_log warn "$NIX_FLAKE_DIR is on $current_branch, not $branch; building $current_branch" nix_log warn "Check out $branch there, or delete $NIX_FLAKE_DIR to start from the remote" return 0 fi - # Fetching as root writes into .git, so re-assert ownership afterwards. - if ! nix_run_logged _sudo git -C "$NIX_FLAKE_DIR" fetch --quiet origin "$branch"; then + # Root fetch writes into .git even when it fails. + rc=0 + nix_run_logged _sudo git -C "$NIX_FLAKE_DIR" fetch --quiet origin "$branch" || rc=$? + nix_own_checkout + if [ "$rc" -ne 0 ]; then nix_log warn "Could not reach the remote; building the existing checkout" return 0 fi @@ -210,10 +151,6 @@ nix_sync_checkout() { fi } -# --------------------------------------------------------------------------- -# deciding -# --------------------------------------------------------------------------- - nix_recorded_rev() { cat "$NIX_STATE_DIR/flake.rev" 2> /dev/null || true } @@ -223,46 +160,28 @@ nix_record_rev() { printf '%s\n' "$1" | _sudo tee "$NIX_STATE_DIR/flake.rev" > /dev/null } -# True when a generation is the boot default but is not the running system, -# i.e. `boot` was used. A revision check alone would call that up to date. -# # Compares against /run/current-system, the *activated* system, and not -# /run/booted-system, which is whatever the kernel booted. After a switch the -# booted symlink still points at the previous generation -- so using it here -# reports every freshly created workspace as needing a restart. nix_pending_generation() { [ "$(readlink -f /run/current-system)" != "$(readlink -f /nix/var/nix/profiles/system)" ] } -# Echoes a revision (or "dirty") and returns 0 when a rebuild is needed. nix_needs_rebuild() { local rev if nix_checkout_dirty; then - # Uncommitted work has no revision to compare, so always rebuild. - printf 'dirty' + printf 'dirty#%s' "$NIX_FLAKE_ATTR" return 0 fi rev="$(nix_checkout_rev)" [ -n "$rev" ] || rev="unknown" - printf '%s' "$rev" + printf '%s#%s' "$rev" "$NIX_FLAKE_ATTR" - [ "$rev" = "$(nix_recorded_rev)" ] || return 0 + [ "$rev#$NIX_FLAKE_ATTR" = "$(nix_recorded_rev)" ] || return 0 nix_pending_generation } -# --------------------------------------------------------------------------- -# applying -# --------------------------------------------------------------------------- - -# `switch` and `boot` both build before activating, so a configuration that -# fails to build never touches the running system; a separate `nix build` -# gate would add nothing. -# -# No --override-input, no --refresh, no --no-write-lock-file: the flake is a -# local checkout that we fetched explicitly, so the command below is exactly -# what a user can type by hand. +# Build the local checkout as-is; no remote override or lock-file suppression. nix_apply() { local operation="$1" transcript="$2" rc @@ -270,11 +189,7 @@ nix_apply() { _sudo install -m 0644 /dev/null "$transcript" _sudo ln -sfn "$transcript" "$NIX_LOG_DIR/rebuild-latest.log" - # stdbuf so output streams instead of arriving in one burst at the end, and - # under _sudo rather than around it: _sudo is a shell function, and stdbuf - # execs what it is given. `stdbuf -oL _sudo tee` dies with "failed to run - # command '_sudo'", which empties the pipeline's middle, breaks the pipe - # under nixos-rebuild and fails the rebuild with an empty transcript. + # stdbuf must run under _sudo; it cannot exec a shell function. set +e _sudo nixos-rebuild "$operation" \ --flake "$NIX_FLAKE_DIR#$NIX_FLAKE_ATTR" \ @@ -287,14 +202,10 @@ nix_apply() { return "$rc" } -# Two concurrent nixos-rebuild processes are a bad time, and the boot path is -# not the only thing that rebuilds this machine: someone at a terminal can run -# nixos-rebuild by hand while a boot is still in progress. nix_lock() { local lock="$NIX_STATE_DIR/rebuild.lock" _sudo install -d -m 0755 "$NIX_STATE_DIR" # Mode 0666 so the workspace user can take the same advisory lock as root. - # It carries no data, only the lock. [ -e "$lock" ] || _sudo install -m 0666 /dev/null "$lock" exec 9> "$lock" if [ "${1:-wait}" = "nowait" ]; then diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.test.sh b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.test.sh new file mode 100755 index 000000000..0bc4a1ba1 --- /dev/null +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.test.sh @@ -0,0 +1,67 @@ +#!/usr/bin/env bash +set -Eeuo pipefail + +here=$(cd "$(dirname "$0")" && pwd) +root=$(mktemp -d) +trap 'rm -rf "$root"' EXIT +export GIT_CONFIG_NOSYSTEM=1 +export GIT_CONFIG_GLOBAL=/dev/null + +git init -q -b main "$root/remote" +git -C "$root/remote" config user.email test@example.org +git -C "$root/remote" config user.name Test +printf 'initial\n' > "$root/remote/flake.nix" +git -C "$root/remote" add flake.nix +git -C "$root/remote" commit -qm initial + +# Source the production functions while replacing privileged ownership with a probe. +# shellcheck source=lifecycle.sh +source "$here/lifecycle.sh" +nix_pending_generation() { false; } +_sudo() { "$@"; } +ownership_calls=0 +nix_own_checkout() { ownership_calls=$((ownership_calls + 1)); } +NIX_FLAKE_DIR="$root/checkout" +NIX_STATE_DIR="$root/state" +NIX_FLAKE_ATTR=host-one +mkdir -p "$NIX_FLAKE_DIR" + +nix_sync_checkout "file://$root/remote" main +[ -f "$NIX_FLAKE_DIR/flake.nix" ] +[ "$ownership_calls" -eq 1 ] +rev=$(nix_needs_rebuild) +[ "$rev" = "$(git -C "$NIX_FLAKE_DIR" rev-parse HEAD)#host-one" ] +nix_record_rev "$rev" +if nix_needs_rebuild > /dev/null; then echo 'unexpected rebuild' >&2; exit 1; fi +NIX_FLAKE_ATTR=host-two +nix_needs_rebuild > /dev/null +nix_record_rev "$(nix_needs_rebuild)" +if nix_needs_rebuild > /dev/null; then echo 'unexpected attribute rebuild' >&2; exit 1; fi + +printf 'ignored by Git flake\n' > "$NIX_FLAKE_DIR/untracked" +if nix_needs_rebuild > /dev/null; then echo 'untracked file caused rebuild' >&2; exit 1; fi +printf 'modified\n' > "$NIX_FLAKE_DIR/flake.nix" +nix_checkout_dirty +nix_needs_rebuild > /dev/null +git -C "$NIX_FLAKE_DIR" checkout -q -- flake.nix + +printf 'updated\n' > "$root/remote/flake.nix" +git -C "$root/remote" commit -qam updated +nix_sync_checkout "file://$root/remote" main +[ "$(cat "$NIX_FLAKE_DIR/flake.nix")" = updated ] +[ "$ownership_calls" -eq 3 ] + +before=$ownership_calls +nix_sync_checkout "file://$root/remote" main +[ "$ownership_calls" -eq "$((before + 1))" ] +before=$ownership_calls +# A failed fetch must still restore ownership. +git -C "$NIX_FLAKE_DIR" remote set-url origin file:///does-not-exist +nix_sync_checkout "file://$root/remote" main +[ "$ownership_calls" -eq "$((before + 1))" ] + +redacted=$(printf 'fatal: https://user:token@example.org/repo\n' | nix_filter_log) +[[ "$redacted" == *'https://[redacted]@example.org/repo'* ]] +[[ "$redacted" != *token* ]] +[[ "$(nix_redact_url 'https://user:token@example.org/repo')" != *token* ]] +printf 'lifecycle tests passed\n' From 75017c5f7912d688c19b4d3081a95f5b65c95035 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Mon, 28 Sep 2026 23:43:20 +0000 Subject: [PATCH 23/24] refactor(aws-nixos): streamline bootstrap and template onboarding Move bootstrap file generation into the template, protect root-owned scripts, encode shell arguments and log JSON, and tighten flake reference/path validation. Keep architecture selection small, prevent disk downsizing, and report the retained AMI. Shorten the three READMEs and add mocked integration, boot, log, and lifecycle tests. --- .../coder-labs/templates/aws-nixos/README.md | 409 ++---------------- .../coder-labs/templates/aws-nixos/main.tf | 110 ++--- .../aws-nixos/modules/amazon-init/README.md | 118 +---- .../aws-nixos/modules/amazon-init/main.tf | 90 ++-- .../modules/amazon-init/main.tftest.hcl | 128 ++++++ .../amazon-init/scripts/bootstrap.sh.tftpl | 114 ++--- .../amazon-init/scripts/bootstrap.test.py | 91 ++++ .../modules/amazon-init/scripts/log.sh | 147 ++++--- .../modules/amazon-init/scripts/log.test.py | 64 +++ .../templates/aws-nixos/modules/nix/main.tf | 13 +- .../aws-nixos/modules/nix/main.tftest.hcl | 8 + .../modules/nix/scripts/lifecycle.test.sh | 15 +- .../aws-nixos/tests/architecture.tftest.hcl | 119 +++++ 13 files changed, 646 insertions(+), 780 deletions(-) create mode 100644 registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tftest.hcl create mode 100644 registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.test.py create mode 100644 registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.test.py create mode 100644 registry/coder-labs/templates/aws-nixos/tests/architecture.tftest.hcl diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 1741ff066..23e293e2f 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -6,407 +6,56 @@ verified: true tags: [vm, linux, aws, nixos, persistent-vm] --- -# Remote development on NixOS AWS EC2 VMs +# NixOS workspaces on AWS EC2 -Provision NixOS EC2 instances as [Coder workspaces](https://coder.com/docs/workspaces), configured -declaratively from a flake in a Git repository. The Coder agent is declared as a NixOS systemd unit, -so it survives `nixos-rebuild` and the workspace stays reachable across configuration changes. The -reference configuration lives at [coder/nixos-example-flake](https://github.com/coder/nixos-example-flake), -which imports the agent, the workspace user and the shutdown hook from -[coder/nixos-modules](https://github.com/coder/nixos-modules). Point the template at your own fork -to control the environment. +Boot an EC2 workspace from a Git flake. NixOS rebuilds before the agent starts. - +## Before you start -## Prerequisites +- Give the Coder provisioner AWS credentials through the usual provider credential chain. + See the [EC2 policy example](../../../coder/templates/aws-linux/PREREQUISITES.md); its `RunInstances` and `CreateTags` permissions cover more than tagged resources. +- Keep a default VPC/subnet. Allow provisioner egress to AWS/JetBrains and VM egress to Coder/Git/Nix caches. +- Use a supported Git URL: `https://host/org/repo` or `git+ssh://git@host/org/repo`, with optional `?ref=branch`. + `github:` and `?dir=` are not supported. Commit your flake before starting a workspace. -### Authentication +## Choose a flake -This template authenticates to AWS using the provider's default [authentication methods](https://registry.terraform.io/providers/hashicorp/aws/latest/docs#authentication-and-configuration). +**Start with the [example flake](https://github.com/coder/nixos-example-flake):** -The simplest way, without editing the template, is environment variables (e.g. `AWS_ACCESS_KEY_ID`) -or a [credentials file](https://docs.aws.amazon.com/cli/latest/userguide/cli-configure-files.html#cli-configure-files-format), -set for the provisioner process rather than as template variables. If you are running Coder on a VM, -that file must be at `/home/coder/aws/credentials`. +1. Fork it, edit `configuration.nix`, and commit your changes and `flake.lock`. +2. Set `flake_ref` to your fork's Git URL. Keep the default `flake_attr = "coder-workspace-ec2-$ARCH"`. +3. Push the template and create a workspace. Its default instance type is `t3.medium`. -### The region needs a default VPC +**Bring an existing flake:** -Like `aws-linux`, this template takes no subnet or security group and launches into the default VPC -of the selected region. Regions without one fail at apply with `VPCIdNotSpecified`. +1. Add `github:coder/nixos-modules` as an input. Import `coder-modules.nixosModules.default` in each workspace host. +2. Include EC2 hardware support; the [example hardware module](https://github.com/coder/nixos-example-flake/blob/main/hardware/ec2.nix) imports the boot-critical NixOS Amazon image module. +3. Set each host's `nixpkgs.hostPlatform` and `coder.flakeAttr` to its own `nixosConfigurations` name. +4. Export hosts for the instance types you offer. Use `$ARCH` in `flake_attr` for paired `x86_64`/`aarch64` names, or a fixed name for one architecture. +5. Set `flake_ref` and `flake_attr`, push the template, then create a workspace. -### Egress from the workspace +Instance type determines AMI, agent architecture, and `$ARCH`. Small sizes may lack build memory. -A NixOS instance fetches and builds its own configuration on boot, so workspaces need outbound -HTTPS to your Coder access URL, `cache.nixos.org`, and wherever the flake is hosted. No inbound -rules are required. +## Work with the workspace -## Required permissions / policy - -The following sample policy allows Coder to create EC2 instances and modify instances provisioned by -Coder: - -```json -{ - "Version": "2012-10-17", - "Statement": [ - { - "Sid": "VisualEditor0", - "Effect": "Allow", - "Action": [ - "ec2:GetDefaultCreditSpecification", - "ec2:DescribeIamInstanceProfileAssociations", - "ec2:DescribeTags", - "ec2:DescribeInstances", - "ec2:DescribeInstanceTypes", - "ec2:DescribeInstanceStatus", - "ec2:CreateTags", - "ec2:RunInstances", - "ec2:DescribeInstanceCreditSpecifications", - "ec2:DescribeImages", - "ec2:ModifyDefaultCreditSpecification", - "ec2:DescribeVolumes" - ], - "Resource": "*" - }, - { - "Sid": "CoderResources", - "Effect": "Allow", - "Action": [ - "ec2:DescribeInstanceAttribute", - "ec2:UnmonitorInstances", - "ec2:TerminateInstances", - "ec2:StartInstances", - "ec2:StopInstances", - "ec2:DeleteTags", - "ec2:MonitorInstances", - "ec2:CreateTags", - "ec2:RunInstances", - "ec2:ModifyInstanceAttribute", - "ec2:ModifyInstanceCreditSpecification" - ], - "Resource": "arn:aws:ec2:*:*:instance/*", - "Condition": { - "StringEquals": { - "aws:ResourceTag/Coder_Provisioned": "true" - } - } - } - ] -} -``` - -## How it works - -The template does not build anything itself. It launches an official NixOS AMI and hands the -instance three things: the agent token, the agent init script, and a flake reference. The instance -then applies that flake. - -```text -Terraform ──user-data──▶ amazon-init ──▶ nixos-rebuild switch ──▶ coder-agent.service - (every boot) (from your flake) (started once it finishes) -``` - -The NixOS AMI does not run cloud-init. It runs `amazon-init.service`, which reads -`/etc/ec2-metadata/user-data` and execs it as a shell script when it begins with `#!` — after -`multi-user.target`, on every boot. The template's script writes the agent handoff, generates the -per-workspace Nix values, and runs one `nixos-rebuild switch`. - -User-data itself is a short self-extracting wrapper: the boot script, its two libraries and the -agent init script come to roughly 19 KiB against EC2's 16 KiB limit, so it ships compressed and -unpacks to `/run/coder/bootstrap.sh` — which is also where to look when debugging a boot. - -**The agent only starts once the rebuild has finished**, so a build that has work to do shows no -agent while it runs — minutes on a first create, usually seconds afterwards. Progress is streamed -to a workspace log source named "NixOS" while that happens, and `connection_timeout` is raised to -20 minutes so a normal first build does not look like a failure. - -That delay is deliberate. `coder-agent.service` is declared without `wantedBy`, so systemd never -starts it on its own and the boot script starts it explicitly after the switch. Left to systemd the -agent would come up at `multi-user.target`, which on every boot after the first is _before_ -`amazon-init` has fetched the new commit and rebuilt: the workspace would be reported ready and its -startup scripts would install into a generation that is about to be replaced. A workspace that is -late is better than a workspace that is wrong. - -If the rebuild fails, the boot script starts the agent anyway — a workspace whose flake does not -build is exactly the one you need a terminal on. - -## Choosing which configuration is applied - -Two Terraform variables control this: `flake_ref`, the Git reference to the configuration, and -`flake_attr`, the `nixosConfigurations` attribute to apply. - -`flake_ref` takes the reference in the form `nix` itself accepts — a `git+` prefix is optional and -`?ref=` selects a branch, so `https://host/org/repo`, -`git+https://host/org/repo?ref=dev` and `git+ssh://git@host/org/repo?ref=dev` all work. Without -`?ref=` the remote's default branch is used, resolved on the instance. The configuration must be -committed: a Git flake reference only ever sees committed files. - -`flake_attr` is the part after `#` in a flake reference, so between them this is exactly the -selection you would make by hand: - -```console -nixos-rebuild switch --flake 'github:your-org/config#coder-workspace-ec2-x86_64' -``` - -`$ARCH` in `flake_attr` is replaced with `x86_64` or `aarch64` to match the chosen instance type. -That keeps the AMI architecture, `coder_agent.arch` and the flake attribute in agreement — all -three are read off the selected instance type, so a Graviton instance type cannot accidentally boot -an x86 configuration. If you keep a single configuration instead, set `flake_attr` to a fixed name and -only offer instance types of the matching architecture. - -> [!NOTE] -> The reference flake's configurations were renamed from `coder-workspace-` to -> `coder-workspace-ec2-`, and the `flake_attr` default here follows. A default is only read -> when a template is created or a variable is left unset, so a template already pushed with the old -> value keeps it — update the stored `flake_attr` (push again with an explicit value, or edit it in -> the template settings) before workspaces rebuild against the renamed flake, otherwise the next -> rebuild fails on a missing attribute. - -Any reference `nixos-rebuild --flake` understands works, including `github:owner/repo`, -`git+ssh://` for private repositories, and `?dir=subdir` for a flake in a subdirectory. The -configuration must be committed: a Git flake reference only ever sees committed files. - -## Values passed into the flake - -None. The flake is evaluated exactly as written, with no `--override-input`, no `--impure` and no -injected inputs — which is what makes the command the template runs reproducible by hand. - -Per-workspace facts are published as a runtime file instead: - -```json -// /run/coder/workspace.json, mode 0644 -{ - "workspace": "my-workspace", - "owner": "jane", - "owner_name": "Jane Doe", - "owner_email": "jane@example.com", - "access_url": "https://coder.example.com", - "hostname": "my-workspace" -} -``` - -It cannot be an evaluation input: a pure flake may not read an absolute path outside itself, so -consuming it at eval time would need `--impure` and would stop `nixos-rebuild switch` from -reproducing what the template applied. A configuration that wants these values reads the file from -a service at runtime. - -Git identity is not in the flake's hands either — it comes from the -[git-config](https://registry.coder.com/modules/coder/git-config) module, which configures the -workspace user's `~/.gitconfig` after the rebuild. - -> [!IMPORTANT] -> Never pass a secret into Nix — not as an input, `--argstr` or `builtins.getEnv`. It is copied -> into `/nix/store`, which is world-readable to every process on the workspace and persists across -> generations and past rotation. The agent token is deliberately handed over through `/run/coder` -> at runtime instead, at mode 0600 on a tmpfs, so that it never reaches Nix. - -## Rebuilding by hand - -The configuration is a git checkout at `/etc/nixos`, owned by the workspace -user, and that is what the template builds. So the command is the ordinary one: +Boot syncs `/etc/nixos` and rebuilds. Dirty trees and local commits stay untouched. To rebuild manually: ```console sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-ec2-x86_64 ``` -No overrides, no `--impure`, no injected inputs — what you get by hand is -exactly what the template applies. The attribute is shown in the workspace's -`Flake URI` metadata and in the boot log. - -> [!NOTE] -> A bare `sudo nixos-rebuild switch` only works if your flake exposes a -> configuration named after the machine's hostname, which is what -> `nixos-rebuild` defaults to. The example flake uses per-architecture names -> instead, so pass `--flake /etc/nixos#`. If your workspaces have stable -> names, naming the configuration after the hostname makes the bare form work. - -On every boot the template syncs the checkout: it clones if missing, -fast-forwards a clean checkout on its tracking branch, and **leaves a dirty -tree or local commits alone** and builds those instead. So edits survive a -restart, and the machine tracks upstream until you change something. - -A flake built from a git checkout ignores untracked files — `git add` a new -`.nix` file or the rebuild will not see it. - -## Keeping workspaces up to date - -A workspace rebuilds from the flake when it **boots**, and that is the only schedule there is: to -pick up a change, restart the workspace, or run the rebuild yourself: - -```console -sudo nixos-rebuild switch --flake /etc/nixos#coder-workspace-ec2-x86_64 -``` - -Adding a timer is the configuration's business, not this template's — `system.autoUpgrade` is -upstream's and goes in the flake, where whoever owns the machine can see it. Two things it will not -do for you: order itself behind the boot-time rebuild (`amazon-init.service`), and sync the -checkout first, since `--refresh` does nothing for a local path flake. - -The "NixOS version" metadata shows `(restart to apply update)` whenever a generation has been built -and made the boot default without being activated — which is what the shutdown staging hook does, -and what `nixos-rebuild boot` does by hand. - -## Where the logs are - -The workspace UI streams `nixos-rebuild`'s own output under a log source named **NixOS** — the same -lines you would see in a terminal. Three kinds of noise are dropped: the enumerated store paths -under `these N derivations will be built:`, per-derivation compiler output, and the expected -`not writing modified lock file` notice. The complete transcript is on the instance: - -```console -/var/log/coder-nixos/rebuild-latest.log # symlink to the most recent boot rebuild -``` - -A rebuild you start yourself is the exception: it goes wherever you ran it, and to the journal, not -to the workspace UI. Only the boot path streams. - -Keeping compiler output out of the UI is not cosmetic. Coder caps agent logs at **1 MiB per -agent**, shared across every log source, and exceeding it does not truncate — the log is marked -overflowed and all later logs for that agent are dropped permanently. The template budgets itself -to half the cap and goes quiet with a pointer to the transcript if it ever gets there. - -## Persistence - -The instance is stopped and started rather than destroyed and recreated, so the root volume — and -with it `/home` and the Nix store — persists across workspace restarts. - -Two lifecycle settings make that safe, and both matter: - -- `ignore_changes = [ami]` — the official NixOS AMIs are republished weekly and garbage-collected - after 90 days. Without this, a new AMI id would replace every live workspace and destroy its root - volume. The side effect is that **changing `nixos_release` only affects newly created - workspaces**; existing ones keep their AMI and get their packages from your flake's nixpkgs pin - anyway. -- `user_data_replace_on_change = false` — the agent token is inside user-data and rotates on every - start, so user-data changes on every start. With replacement enabled, every restart would destroy - the volume. +Use your host's attribute instead of the example name. The root disk and Nix store survive stop/start, **not** instance deletion or replacement. A larger root disk can be selected later; EBS cannot shrink it. -`root_volume_size` is mutable: the AMI enables `boot.growPartition` and `autoResize`, so a larger -volume is picked up on the next restart. - -## Architecture support - -Both `x86_64` and `arm64` (Graviton) instance types are offered. The instance type is a -[aws-ec2-instance-type](https://registry.coder.com/modules/coder/aws-ec2-instance-type) parameter, -and the AMI filter, `coder_agent.arch` and the flake attribute are all looked up from the same -catalog entry — so they cannot disagree. Your flake still has to expose a configuration for the -architecture you select; the reference flake ships `coder-workspace-ec2-x86_64` and -`coder-workspace-ec2-aarch64`. - -Every size of `t3`, `t4g` and `m7g` is offered — nothing is filtered out, so the list runs from -`t3.nano` upwards. The default is `t3.medium` because that is the smallest one that works: the -NixOS AMI configures no swap and the Nix store shares the root volume, so a rebuild that has to -compile anything will exhaust a 1–2 GiB instance. The smaller options are selectable and will fail -the first time they have to build something that is not in the binary cache. - -Add families with the module's `include` — it defaults to `t3` alone, so the Graviton families are -named explicitly here to keep both architectures on offer. - -## Troubleshooting - -### The workspace has been building for a long time - -Expected whenever there is a rebuild to do, and always on first create: the agent is not started -until `nixos-rebuild switch` finishes. Watch the "NixOS" log source. A cold closure on a small -instance can take ten minutes or more. A restart with nothing to rebuild skips straight to starting -the agent. - -### The agent never connects - -The boot script writes its handoff to `/run/coder` before doing anything else, so the usual cause is -a failed rebuild — or, if there are no logs at all, an instance with no route to the internet (a -NixOS workspace fetches its own configuration on boot, so it needs egress before it can report -anything). The workspace metadata shows the instance id; the AMI logs to the serial console, which -needs no SSH: +Watch the **NixOS** workspace log or `/var/log/coder-nixos/rebuild-latest.log`. If first boot has no agent, use EC2 console output: ```console -aws ec2 get-console-output --instance-id i-0123456789abcdef0 --output text +aws ec2 get-console-output --instance-id --output text ``` -On the instance: - -```console -systemctl status amazon-init coder-agent -journalctl -u amazon-init -b -cat /var/log/coder-nixos/rebuild-latest.log -``` - -### A configuration change broke the workspace - -`nixos-rebuild switch` builds before it activates, so a configuration that fails to _build_ never -touches the running system — the previous generation keeps running and the failure appears in the -logs. If a configuration builds but misbehaves, roll back: - -```console -sudo nixos-rebuild switch --rollback -``` - -There is deliberately no automatic rollback: on a fresh instance the previous generation is the bare -AMI, which has no Coder agent at all, so rolling back automatically would trade a visible failure -for an unreachable workspace. - -### Recovering an unreachable instance - -`aws ec2 get-console-output` above needs no access to the instance at all and is usually enough. - -For a shell, the AMI enables OpenSSH and `amazon-ssm-agent`. Neither is reachable out of the box — -the template attaches no key pair and the default security group allows no inbound traffic — so -attach what you need for the session with the AWS CLI and detach it afterwards: - -```console -aws ec2 create-security-group --group-name coder-debug --description "temporary SSH" --vpc-id -aws ec2 authorize-security-group-ingress --group-id --protocol tcp --port 22 --cidr /32 -aws ec2 modify-instance-attribute --instance-id --groups -``` - -### Private flake repositories - -Nix fetches flakes as root via the daemon, so credentials must be readable by root rather than by -the workspace user. Prefer `nix.settings.netrc-file` pointing at a file the boot script writes at -mode 0600, or an `!include` of a private file from `/etc/nix/nix.conf`. Do **not** put a token in -`nix.settings.access-tokens` directly: that renders it into `/etc/nix/nix.conf` by way of the Nix -store, where every process on the workspace can read it. - -## Extending the template - -Five registry modules are included: -[aws-region](https://registry.coder.com/modules/coder/aws-region), -[aws-ec2-instance-type](https://registry.coder.com/modules/coder/aws-ec2-instance-type), -[code-server](https://registry.coder.com/modules/coder/code-server), -[jetbrains-gateway](https://registry.coder.com/modules/coder/jetbrains-gateway) and -[git-config](https://registry.coder.com/modules/coder/git-config). - -code-server and JetBrains Gateway push a dynamically linked binary into the workspace and exec it, -so they work only because the reference flake sets `programs.nix-ld.enable = true` — remove that -and both fail with a misleading "No such file or directory". Gateway is also told which architecture to fetch, from the -same catalog entry that picks the AMI, and is restricted to the IDEs JetBrains publishes an -`aarch64` backend for. - -> [!NOTE] -> A registry module whose script starts with `#!/bin/bash` cannot run here. NixOS puts nothing in -> `/bin` but `sh`, so the kernel fails the exec before anything runs and the agent reports exit -> 255 with an empty log. `#!/usr/bin/env bash` works. Check the module before adding it. - -The `coder` CLI itself is put on `PATH` by the flake's Coder module, which the `coder stat` -metadata scripts depend on. The agent prepends its own directory to the `PATH` it hands scripts, -but on NixOS those run through a login shell and `/etc/profile` rebuilds `PATH` from the system -environment, dropping it — so without that wrapper every CPU/memory/disk metric reads -`coder: command not found`. - -For anything heavier, prefer declaring the tool in your flake and exposing it with a -[`coder_app`](https://registry.terraform.io/providers/coder/coder/latest/docs/resources/app): -it is reproducible and avoids the dynamic-linking problem entirely. +## Secrets and limitations -The template is two halves that do not know about each other: +Do not put secrets in Nix expressions: the Nix store is readable on the VM. Workspace facts and optional bootstrap files are not secret storage. The agent token is kept out of Nix, but EC2 user-data and Terraform state contain it; restrict access to both. Processes with instance-metadata access can read user-data. -- [`modules/amazon-init/`](./modules/amazon-init/README.md) gets Coder onto an EC2 instance whose - AMI runs `amazon-init` instead of cloud-init. It publishes the agent handoff and the workspace - identity, streams logs before an agent exists, runs one script, and starts the agent last. - Nothing in it mentions Nix — the script it runs is an opaque string. -- [`modules/nix/`](./modules/nix/README.md) is the flake lifecycle: sync a checkout, decide whether - a rebuild is needed, apply it, filter the output. Nothing in it mentions EC2 or user-data. +Private repos need root Git credentials before first boot and root Nix input credentials; this template supplies neither. HTTP URL credentials in `flake_ref` are rejected. NixOS scripts need `#!/usr/bin/env bash`; downloaded IDE binaries need `programs.nix-ld`. -The seam is one string: the nix module renders a boot script and exports it, and the amazon-init -module runs it without looking inside. Either half can be lifted into a standalone registry module -without untangling it from the other, and `main.tf` is left with the things that are genuinely -about this template — the AMI, the instance, the agent and the IDE modules. +Existing templates may retain a stored legacy `flake_attr`; update that variable explicitly before rebuilding against renamed example-flake hosts. diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index 83653525e..901abbd5f 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -11,7 +11,6 @@ terraform { } module "aws-region" { - # 1.1.0 for `default_availability_zone`. source = "registry.coder.com/coder/aws-region/coder" version = "~> 1.1" default = "eu-west-3" @@ -22,16 +21,14 @@ provider "aws" { } variable "flake_ref" { - description = <<-EOT - Git reference to the NixOS configuration, in the form `nix` itself - accepts: `https://host/org/repo`, optionally with a `git+` prefix and a - `?ref=` branch. Without `?ref=` the remote's default branch is used. - - The configuration must be committed -- a Git flake reference only ever - sees committed files. - EOT + description = "Git URL of the NixOS flake; optional ?ref= selects a branch. No credentials in the URL." type = string default = "https://github.com/coder/nixos-example-flake" + + validation { + condition = can(regex("^(git\\+)?(https?|ssh)://", var.flake_ref)) && !can(regex("^(git\\+)?https?://[^/?#]*@", var.flake_ref)) && !can(regex("[[:cntrl:]]", var.flake_ref)) && !strcontains(var.flake_ref, "#") && (!strcontains(var.flake_ref, "?") || can(regex("\\?ref=[^&#?]+$", var.flake_ref))) + error_message = "Use an http(s) or ssh Git URL without HTTP credentials, control characters, fragments, or query parameters other than ?ref=." + } } variable "flake_attr" { @@ -54,18 +51,8 @@ module "aws-ec2-instance-type" { source = "registry.coder.com/coder/aws-ec2-instance-type/coder" version = "~> 1.0" - default = "t3.medium" - description = trimspace(<<-EOT - t3.medium is the smallest that works: the NixOS AMI configures no swap and - the Nix store shares the root volume, so a rebuild that has to compile - anything will exhaust a 1-2 GiB instance. - EOT - ) - - # Nothing is excluded, but the families are named: the module offers `t3` - # alone by default, and this template supports both architectures -- the - # AMI, the agent and the flake attribute all follow the instance type, so - # dropping the Graviton families would quietly make it x86-only. + default = "t3.medium" + description = "Choose enough memory for NixOS builds. The smallest sizes may run out of memory." include = [ "t3", "t4g", @@ -82,8 +69,9 @@ data "coder_parameter" "root_volume_size" { mutable = true validation { - min = 40 - max = 2000 + min = 40 + max = 2000 + monotonic = "increasing" } } @@ -98,23 +86,18 @@ data "aws_ami" "nixos" { } filter { name = "architecture" - values = [local.arch.ami] + values = [local.instance.arch] } - # Required: aws_ami matches on a name pattern, and AMI names are not unique - # across accounts. Without an owner filter, most_recent would happily pick - # a stranger's image named nixos/... and boot it with the agent token. - owners = ["427812963091"] # NixOS + # Restrict the name match to official NixOS images. + owners = ["427812963091"] } resource "coder_agent" "main" { count = data.coder_workspace.me.start_count - arch = local.arch.agent + arch = local.instance.coder_arch os = "linux" - # Token rather than instance identity: the boot script needs a bearer token - # for the agent log API anyway. - auth = "token" - # The first boot completes a nixos-rebuild switch before the agent exists, - # so the default 120s looks like a failed workspace. + auth = "token" + # The first boot rebuilds NixOS before starting the agent. connection_timeout = 1200 metadata { @@ -138,11 +121,7 @@ resource "coder_agent" "main" { timeout = 30 script = "coder stat disk --path $HOME" } - # Makes a staged generation visible: a configuration that was built and made - # the boot default without being activated otherwise looks exactly like - # updates being ignored. The command comes from the nix module -- metadata - # has to be declared on the agent, but what it means to be up to date is not - # this file's business. + # Include staged-but-not-activated generations in the agent metadata. metadata { key = "nixos" display_name = "NixOS version" @@ -152,7 +131,6 @@ resource "coder_agent" "main" { } } -# See https://registry.coder.com/modules/coder/code-server module "code-server" { count = data.coder_workspace.me.start_count source = "registry.coder.com/coder/code-server/coder" @@ -161,30 +139,20 @@ module "code-server" { order = 1 } -# See https://registry.coder.com/modules/coder/jetbrains-gateway -# -# The IDE backend is a dynamically linked download that Gateway unpacks into -# the workspace and execs, which on NixOS needs `programs.nix-ld`. The -# reference flake enables it. +# IDE binaries need nix-ld in the flake; the example enables it. module "jetbrains-gateway" { count = data.coder_workspace.me.start_count source = "registry.coder.com/coder/jetbrains-gateway/coder" version = "~> 1.2" agent_id = coder_agent.main[0].id agent_name = "main" - arch = local.arch.agent - # The flake names the workspace user; `coder` is the reference flake's - # default and what the rest of this template assumes. - folder = "/home/coder" - # Restricted to the IDEs JetBrains publishes an aarch64 backend for, since - # half the instance types here are Graviton. + arch = local.instance.coder_arch + folder = "/home/coder" + # All listed IDEs have both x86 and ARM builds. jetbrains_ides = ["IU", "PY", "GO", "WS"] default = "IU" - # Without this the module hands Gateway its pinned 2024.3 build numbers. - # The cost is that a workspace build now asks data.services.jetbrains.com - # for the current release. - latest = true - order = 2 + latest = true + order = 2 } module "git-config" { @@ -196,31 +164,17 @@ module "git-config" { locals { instance = module.aws-ec2-instance-type.instances[module.aws-ec2-instance-type.value] - - # The AMI architecture, coder_agent.arch and the flake attribute all come - # from the instance type, so they cannot disagree. The module publishes the - # first two spellings; the third is Nix's, and is the same distinction. - arch = { - agent = local.instance.coder_arch - ami = local.instance.arch - attr = local.instance.arch == "arm64" ? "aarch64" : "x86_64" - } + nix_arch = local.instance.arch == "arm64" ? "aarch64" : "x86_64" } -# Everything about the flake: the checkout, the boot-time rebuild and the -# periodic one. It knows nothing about EC2 -- `boot_script` is a string for -# whoever runs scripts on the machine. See ./modules/nix/README.md. module "nix" { source = "./modules/nix" flake_ref = var.flake_ref flake_attr = var.flake_attr - arch = local.arch.attr + arch = local.nix_arch } -# Gets Coder onto the instance and runs one script on every boot. It knows -# nothing about Nix: `boot_script` is an opaque string to it, and the flake is -# applied entirely inside that string. See ./modules/amazon-init/README.md. module "amazon-init" { source = "./modules/amazon-init" @@ -239,9 +193,7 @@ resource "aws_instance" "dev" { instance_type = module.aws-ec2-instance-type.value user_data = module.amazon-init.user_data - # The agent token is inside user-data and rotates on every workspace start, - # so user-data changes on every start. With replacement enabled, every - # restart would destroy the root volume and with it /home and the Nix store. + # Rotating the user-data token must not replace the persistent root disk. user_data_replace_on_change = false root_block_device { @@ -251,14 +203,12 @@ resource "aws_instance" "dev" { } tags = { - Name = "coder-${data.coder_workspace_owner.me.name}-${data.coder_workspace.me.name}" - # Required if you are using our example policy, see template README + Name = "coder-${data.coder_workspace_owner.me.name}-${data.coder_workspace.me.name}" Coder_Provisioned = "true" } lifecycle { - # NixOS AMIs are republished weekly and garbage-collected after 90 days. - # Without this, a new AMI id replaces every live workspace. + # New AMI releases must not replace an existing workspace and its disk. ignore_changes = [ami] } } @@ -267,7 +217,7 @@ resource "coder_metadata" "workspace_info" { resource_id = aws_instance.dev.id item { key = "AMI" - value = data.aws_ami.nixos.name + value = aws_instance.dev.ami } item { key = "Flake URI" diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md index 0c16f917a..50f896801 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md @@ -1,117 +1,25 @@ # amazon-init -Coder on an EC2 AMI that runs `amazon-init` instead of cloud-init. - -It renders user-data that publishes the agent handoff, publishes the -workspace's identity, runs **one script you supply**, and then starts the -agent. It does not know or care what that script does — `boot_script` is an -opaque string, and nothing in this module reads it. +`amazon-init.service` runs this user-data every boot. It writes the agent +handoff and public facts, runs `boot_script` as root, then starts the agent +even on failure. Do not enable the agent unit at boot. ```tf module "amazon-init" { - source = "./modules/amazon-init" - + source = "./modules/amazon-init" agent_token = coder_agent.main.token agent_init_script = coder_agent.main.init_script boot_script = local.my_boot_script } - -resource "aws_instance" "dev" { - user_data = module.amazon-init.user_data -} ``` -Workspace and owner identity are read from `coder_workspace` and -`coder_workspace_owner` here, so the caller passes neither. - -## Why this is not cloud-init - -Some AMIs do not ship cloud-init, they run `amazon-init.service`, which reads -`/etc/ec2-metadata/user-data` and execs it as a shell script when it begins -with `#!` — after `multi-user.target`, **on every boot**. There is no `runcmd`, no -`write_files`, no per-boot/once distinction and no ordering hooks. Three -things follow: - -- Everything must be idempotent, because it all runs again on every restart. -- It is the only hook available, so the agent handoff, the workspace facts and - whatever the image needs doing all have to live in it. -- Nothing can be ordered `After` it. A boot script that reconfigures the - machine may start units synchronously, and `amazon-init` cannot become - active until it exits — so a unit that waits for `amazon-init` deadlocks the - first boot. - -## Order of operations - -1. **Handoff** — `agent.env` (0600), `init.sh`, then `ready`, into - `runtime_dir` (a tmpfs). Written before anything that can fail, so the - agent can still be started from a failure path. -2. **Hostname**, from `hostname` or the workspace name. -3. **Logging** — the shell library is written to `$${runtime_dir}/log.sh` and - sourced, and the log source is registered. -4. **Facts** — `$${runtime_dir}/workspace.json`, mode 0644, no secrets: the - workspace's identity, the deployment URL, the `log_source_id` a service on - the instance needs in order to log anywhere the user will see, and anything - the caller added through `values`. Rendered by Terraform, so a full name - with a quote in it cannot break the document. -5. **Files** — anything in `files`, a map of absolute path to text, written - mode 0644 with parent directories created. -6. **Your boot script**, as a child process. -7. **The agent**, always, whatever step 6 did. - -## Starting the agent last is the point - -`coder-agent.service` must not be `wantedBy` anything; this module starts it, -and only once the boot script has finished. Left to systemd, the agent comes -up at `multi-user.target` — before the boot script has finished changing the -machine — reports the workspace ready, and runs its startup scripts against a -system that is about to be replaced under them. - -The mirror image of that rule is that a boot script failure must not leave an -unreachable workspace, so the agent is started even when the boot script -exits non-zero. Its status is reported and propagated, never suppressed. - -## What the boot script gets - -Root, a child process, and these: - -| Variable | Meaning | -| ----------------------- | ------------------------------ | -| `CODER_LOG_LIBRARY` | path to source for `coder_log` | -| `CODER_RUNTIME_DIR` | the tmpfs this module owns | -| `CODER_WORKSPACE_FACTS` | path to `workspace.json` | -| `CODER_ACCESS_URL` | deployment URL | -| `CODER_AGENT_TOKEN` | agent token | -| `CODER_LOG_SOURCE_ID` | log source to write to | - -The log source is registered before the boot script starts, and `CODER_LOG_READY` -is exported, so sourcing the library is enough — `coder_log` works from the -first line and the boot script must not register the source again. - -## Logging before the agent exists - -The agent is the normal way to get output into the workspace UI, and on a -first boot it does not exist for as long as the boot script runs. So -`scripts/log.sh` calls the log API directly with the agent's own token: -`coder_log `, `coder_log_pipe ` for a stream, and -`coder_log_tail ` for the end of a transcript. - -The library is internal to this module — it is not an output. On the instance -it is at `$${runtime_dir}/log.sh`, which is where anything the agent runs later -should source it from, and `CODER_LOG_READY` is exported so a process that -sources it does not register the source a second time. - -The source id is a `random_uuid` held in Terraform state: stable across -stop/start, new only when the workspace is recreated, by which point the agent -and its logs are new anyway. - -Two things it handles that are easy to get wrong: +`runtime_dir` must be tmpfs. Root-owned scripts cannot be replaced by the agent; +only `agent.env` (0600) and `init.sh` (0700) pass to it. The token also lives in +Terraform state and EC2 user-data: restrict IMDS access. Never put secrets in +public `values` or `files` (0644). -- **The 1 MiB cap.** Coder caps agent logs at 1 MiB per agent across every - source. Overflowing does not truncate — the agent is flagged overflowed and - every later log from every source is dropped permanently. The library tracks - its own usage and goes quiet at `log_budget_bytes`, which defaults to half - the cap. -- **No curl.** The image must have one on `PATH` — the NixOS AMI does, in its - own first generation, before anything has been rebuilt. An image without one - simply gets no logs: every function here fails closed, because logging must - never be the reason a boot fails. +The root boot script receives `CODER_RUNTIME_DIR`, `CODER_WORKSPACE_FACTS`, +`CODER_ACCESS_URL`, `CODER_AGENT_TOKEN`, `CODER_LOG_SOURCE_ID`, and `CODER_LOG_LIBRARY`. +Early logs require outbound Coder access and `curl`. The shared 1 MiB +agent log cap is budgeted. Use trusted file directories: symlinked ancestors +can redirect writes despite lexical path checks. diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf index bbe44ebda..0081139e9 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf @@ -13,12 +13,7 @@ terraform { } } -# The log source has to keep the same id for the life of the workspace -- -# Coder treats a repeat POST of a known id as a no-op, and a value that -# changed every plan would churn user-data on every start. Terraform state is -# exactly the right place for that: stable across stop/start, new only when -# the workspace is recreated, by which point the agent and its logs are new -# too. +# Keep the log source stable across workspace restarts. resource "random_uuid" "log_source" {} data "coder_workspace" "me" {} @@ -74,8 +69,15 @@ variable "files" { default = {} validation { - condition = alltrue([for path in keys(var.files) : startswith(path, "/")]) - error_message = "File paths must be absolute." + condition = alltrue([ + for path in keys(var.files) : + startswith(path, "/") && abspath(path) == path && + !can(regex("[\\x00-\\x1f\\x7f]", path)) && + path != var.runtime_dir && !startswith(path, "${var.runtime_dir}/") && + !(startswith(var.runtime_dir, "/run/") && startswith(path, "/var${var.runtime_dir}/")) && + !(startswith(var.runtime_dir, "/var/run/") && startswith(path, "${trimprefix(var.runtime_dir, "/var")}/")) + ]) + error_message = "File paths must be absolute, canonical files outside runtime_dir (including /var/run aliases)." } } @@ -100,6 +102,11 @@ variable "runtime_dir" { description = "Directory for the agent handoff and this module's own state. Must be on a tmpfs: it holds the token." type = string default = "/run/coder" + + validation { + condition = startswith(var.runtime_dir, "/") && var.runtime_dir != "/" && abspath(var.runtime_dir) == var.runtime_dir && !can(regex("[\\x00-\\x1f\\x7f]", var.runtime_dir)) + error_message = "runtime_dir must be a canonical absolute directory path without control characters." + } } variable "path" { @@ -142,20 +149,12 @@ variable "hostname" { locals { hostname = var.hostname != "" ? var.hostname : lower(data.coder_workspace.me.name) - # Written by the bootstrap script before the boot script runs. Carried - # gzipped and base64-encoded so that no content can terminate the heredoc - # that writes it. - files_sh = join("\n", [ - for path, content in var.files : <<-SH - install -d -m 0755 "$(dirname '${path}')" - printf '%s' '${base64gzip(content)}' | base64 -d | gzip -dc >'${path}' - chmod 0644 '${path}' - SH - ]) - - # Everything the machine is told about itself. Merged so that what this - # module knows wins: a caller cannot accidentally rewrite the workspace's - # own identity through `values`. + files = [for path, content in var.files : { + path = base64encode(path) + content = base64gzip(content) + }] + + # Module-owned identity wins over caller-provided facts. facts = merge(var.values, { workspace = data.coder_workspace.me.name owner = data.coder_workspace_owner.me.name @@ -164,8 +163,6 @@ locals { access_url = data.coder_workspace.me.access_url hostname = local.hostname - # The one thing a configuration on the instance cannot work out for - # itself, and needs in order to log anywhere the user will see. log_source_id = random_uuid.log_source.result }) @@ -173,39 +170,39 @@ locals { FACTS_JSON = jsonencode(local.facts) LOG_SH = file("${path.module}/scripts/log.sh") - FILES_SH = local.files_sh + FILES = local.files BOOT_SCRIPT = var.boot_script INIT_SCRIPT = var.agent_init_script - ARG_ACCESS_URL = data.coder_workspace.me.access_url - ARG_AGENT_TOKEN = var.agent_token - ARG_RUNTIME_DIR = var.runtime_dir - ARG_PATH = var.path + ARG_ACCESS_URL = base64encode(data.coder_workspace.me.access_url) + ARG_AGENT_TOKEN = base64encode(var.agent_token) + ARG_RUNTIME_DIR = base64encode(var.runtime_dir) + ARG_PATH = base64encode(var.path) - ARG_LOG_SOURCE_ID = random_uuid.log_source.result - ARG_LOG_DISPLAY_NAME_B64 = base64encode(var.log_display_name) - ARG_LOG_ICON = var.log_icon - ARG_LOG_BUDGET = var.log_budget_bytes + ARG_LOG_SOURCE_ID = random_uuid.log_source.result + ARG_LOG_BUDGET = var.log_budget_bytes + ARG_LOG_REGISTRATION_B64 = base64encode(jsonencode({ + id = random_uuid.log_source.result + display_name = var.log_display_name + icon = var.log_icon + })) - ARG_HOSTNAME = local.hostname + ARG_HOSTNAME = base64encode(local.hostname) }) - # EC2 caps user-data at 16 KiB and the script above plus its payloads is - # comfortably past that, so user-data is a six-line self-extracting wrapper - # around a compressed copy. - # - # This is transparent to amazon-init: it only inspects the first two bytes - # for `#!` before exec'ing the blob, and it has no decompression step of its - # own. Extracting to a fixed path also means the real script is on disk when - # something needs debugging. + # EC2 caps user-data at 16 KiB; compress the bootstrap before sending it. user_data = <<-SH #!/usr/bin/env bash set -eu - install -d -m 0700 ${var.runtime_dir} - base64 -d <<'CODER_PAYLOAD' | gzip -dc >${var.runtime_dir}/bootstrap.sh + runtime_dir=$(printf %s '${base64encode(var.runtime_dir)}' | base64 -d) + install -d -m 0700 -- "$runtime_dir" + chown root:root -- "$runtime_dir" + chmod 0700 -- "$runtime_dir" + base64 -d <<'CODER_PAYLOAD' | gzip -dc >"$runtime_dir/bootstrap.sh" ${base64gzip(local.bootstrap)} CODER_PAYLOAD - exec bash ${var.runtime_dir}/bootstrap.sh + chmod 0700 -- "$runtime_dir/bootstrap.sh" + exec bash "$runtime_dir/bootstrap.sh" SH } @@ -214,8 +211,7 @@ output "user_data" { value = local.user_data sensitive = true - # nonsensitive because the length is sensitive by propagation, and Terraform - # suppresses error messages derived from sensitive values. + # Terraform suppresses error messages derived from sensitive values. precondition { condition = nonsensitive(length(local.user_data)) < 16384 error_message = "Rendered user-data is ${nonsensitive(length(local.user_data))} bytes; EC2 allows at most 16384." diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tftest.hcl b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tftest.hcl new file mode 100644 index 000000000..645389013 --- /dev/null +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tftest.hcl @@ -0,0 +1,128 @@ +mock_provider "coder" { + mock_data "coder_workspace" { + defaults = { + name = "test" + access_url = "https://coder.example" + } + } + mock_data "coder_workspace_owner" { + defaults = { + name = "owner" + full_name = "Owner" + email = "owner@example.org" + } + } +} + +mock_provider "random" { + mock_resource "random_uuid" { + defaults = { + result = "11111111-1111-4111-8111-111111111111" + } + } +} + +run "safe_render" { + command = apply + variables { + agent_token = "token" + agent_init_script = "echo init" + files = { "/tmp/quote'$(touch injected) 🐈" = "safe text" } + log_display_name = "quoted \" name 🐈" + } + assert { + condition = length(nonsensitive(output.user_data)) < 16384 && startswith(nonsensitive(output.user_data), "#!/usr/bin/env bash") + error_message = "User-data must remain a runnable, EC2-sized shell script." + } +} + +run "reject_relative_path" { + command = plan + variables { + agent_token = "token" + agent_init_script = "echo init" + files = { "relative/file" = "bad" } + } + expect_failures = [var.files] +} + +run "reject_parent_traversal" { + command = plan + variables { + agent_token = "token" + agent_init_script = "echo init" + files = { "/tmp/../sneaky" = "bad" } + } + expect_failures = [var.files] +} + +run "reject_newline_path" { + command = plan + variables { + agent_token = "token" + agent_init_script = "echo init" + files = { "/tmp/new\nline" = "bad" } + } + expect_failures = [var.files] +} + +run "reject_handoff_overwrite" { + command = plan + variables { + agent_token = "token" + agent_init_script = "echo init" + files = { "/run/coder/agent.env" = "bad" } + } + expect_failures = [var.files] +} + +run "reject_duplicate_separator" { + command = plan + variables { + agent_token = "token" + agent_init_script = "echo init" + files = { "/run//coder/agent.env" = "bad" } + } + expect_failures = [var.files] +} + +run "reject_dot_segment" { + command = plan + variables { + agent_token = "token" + agent_init_script = "echo init" + files = { "/run/./coder/agent.env" = "bad" } + } + expect_failures = [var.files] +} + +run "reject_var_run_alias" { + command = plan + variables { + agent_token = "token" + agent_init_script = "echo init" + files = { "/var/run/coder/agent.env" = "bad" } + } + expect_failures = [var.files] +} + +run "reject_reverse_var_run_alias" { + command = plan + variables { + agent_token = "token" + agent_init_script = "echo init" + runtime_dir = "/var/run/coder" + files = { "/run/coder/agent.env" = "bad" } + } + expect_failures = [var.files] +} + +run "reject_noncanonical_runtime" { + command = plan + variables { + agent_token = "token" + agent_init_script = "echo init" + runtime_dir = "/run//coder" + } + expect_failures = [var.runtime_dir] +} diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl index dac7a4d21..13ccc8609 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.sh.tftpl @@ -1,65 +1,37 @@ #!/usr/bin/env bash -# EC2 user-data for an AMI that runs amazon-init.service rather than -# cloud-init. amazon-init execs this as a shell script when it starts with -# `#!` -- after multi-user.target, on every boot. So it must be idempotent. -# -# It knows nothing about what the instance is for. It publishes the agent -# handoff, publishes the workspace's identity, runs one caller-supplied boot -# script, and then starts the agent -- in that order, always. - -# -E so that the ERR trap installed below is inherited by the functions here; -# without errtrace a failure inside one of them ends this script with nothing -# said about it. +# amazon-init runs this on every boot; the agent must start after the boot script. set -Eeuo pipefail -export PATH="${ARG_PATH}:$PATH" +decode() { printf '%s' "$1" | base64 -d; } +export PATH="$(decode '${ARG_PATH}'):$PATH" export HOME=/root -ACCESS_URL='${ARG_ACCESS_URL}' -AGENT_TOKEN='${ARG_AGENT_TOKEN}' +ACCESS_URL=$(decode '${ARG_ACCESS_URL}') +AGENT_TOKEN=$(decode '${ARG_AGENT_TOKEN}') LOG_SOURCE_ID='${ARG_LOG_SOURCE_ID}' -LOG_DISPLAY_NAME_B64='${ARG_LOG_DISPLAY_NAME_B64}' -LOG_ICON='${ARG_LOG_ICON}' LOG_BUDGET='${ARG_LOG_BUDGET}' -HOSTNAME_='${ARG_HOSTNAME}' +HOSTNAME_=$(decode '${ARG_HOSTNAME}') +RUNTIME_DIR=$(decode '${ARG_RUNTIME_DIR}') -RUNTIME_DIR='${ARG_RUNTIME_DIR}' - -# --- 1. Hand the agent its token ------------------------------------------- -# -# Done first, before anything that can fail, so that the agent can still be -# started from a later failure path -- an instance whose boot script broke is -# exactly the instance someone needs a terminal on. -# -# RUNTIME_DIR is a tmpfs, so nothing written here survives a reboot. That is -# the point: the token rotates on every workspace start. - -install -d -m 0700 "$RUNTIME_DIR" +# The token stays on tmpfs; only handoff files are owned by the agent. +install -d -m 0700 -- "$RUNTIME_DIR" +chown root:root -- "$RUNTIME_DIR" +chmod 0711 -- "$RUNTIME_DIR" rm -f "$RUNTIME_DIR/ready" umask 077 printf 'CODER_AGENT_TOKEN=%s\nCODER_AGENT_URL=%s\n' "$AGENT_TOKEN" "$ACCESS_URL" \ >"$RUNTIME_DIR/agent.env" umask 022 -# Verbatim rather than base64: user-data has 16 KiB to fit in, the init -# script is the largest thing in it, and base64 both inflates by a third and -# destroys the redundancy the gzip wrapper lives on. cat >"$RUNTIME_DIR/init.sh" <<'CODER_AMAZON_INIT_AGENT_SCRIPT' ${INIT_SCRIPT} CODER_AMAZON_INIT_AGENT_SCRIPT -chmod 0755 "$RUNTIME_DIR/init.sh" - -# coder-agent.service may run as an unprivileged user, so it has to be able to -# traverse the directory and read the token. Modes stay 0600/0700; only the -# owner changes. -# -# The owner is asked of the unit rather than guessed, so an image we do not -# control cannot disagree with us. On the very first boot the unit does not -# exist yet, so this is a no-op and is retried before the agent is started. +chmod 0700 "$RUNTIME_DIR/init.sh" + chown_handoff() { local target target=$(systemctl show coder-agent -p User --value 2>/dev/null || true) [ -n "$target" ] || return 0 - chown -R "$target" "$RUNTIME_DIR" 2>/dev/null || true + chown "$target" "$RUNTIME_DIR/agent.env" "$RUNTIME_DIR/init.sh" 2>/dev/null || true } chown_handoff touch "$RUNTIME_DIR/ready" @@ -67,24 +39,12 @@ chown_handoff [ -z "$HOSTNAME_" ] || hostnamectl set-hostname "$HOSTNAME_" || true -# --- 2. Logging ------------------------------------------------------------- -# -# The agent is the usual way to get output into the workspace UI, and it does -# not exist yet -- on a first boot it will not exist for as long as the boot -# script runs. So the log API is called directly, with the agent's own token. - export CODER_ACCESS_URL="$ACCESS_URL" export CODER_AGENT_TOKEN="$AGENT_TOKEN" export CODER_LOG_SOURCE_ID="$LOG_SOURCE_ID" export CODER_LOG_STATE_DIR="$RUNTIME_DIR" export CODER_LOG_BUDGET="$LOG_BUDGET" -# Written to disk as well as sourced, so the boot script and anything the -# agent runs later can use the same functions without a second copy. -# -# Interpolated as text rather than base64: user-data is gzipped and has 16 KiB -# to fit in, and base64 both inflates by a third and destroys the redundancy -# gzip lives on. The delimiters are long enough not to occur in the payload. cat >"$RUNTIME_DIR/log.sh" <<'CODER_AMAZON_INIT_LOG_LIBRARY' ${LOG_SH} CODER_AMAZON_INIT_LOG_LIBRARY @@ -93,14 +53,11 @@ export CODER_LOG_LIBRARY="$RUNTIME_DIR/log.sh" # shellcheck source=/dev/null . "$CODER_LOG_LIBRARY" -coder_log_init "$(printf '%s' "$LOG_DISPLAY_NAME_B64" | base64 -d)" "$LOG_ICON" || true +CODER_LOG_REGISTRATION=$(decode '${ARG_LOG_REGISTRATION_B64}') +export CODER_LOG_REGISTRATION +coder_log_init || true -# Nothing starts coder-agent.service but this, and only once the boot script -# has finished. If systemd started it at multi-user.target instead, the agent -# would report the workspace ready and run its startup scripts while the boot -# script was still changing the machine underneath them. -# -# So every exit path from here on has to go through this. +# Agent startup follows every boot-script outcome, including failures. start_agent() { if systemctl is-active --quiet coder-agent; then return 0 @@ -113,8 +70,6 @@ start_agent() { chown_handoff systemctl start coder-agent || true - # Type=simple, so systemd reports the unit active as soon as it has forked. - # Give a start that is going to fail immediately the chance to do so. sleep 3 if systemctl is-active --quiet coder-agent; then coder_log info "coder-agent is active." || true @@ -134,17 +89,7 @@ on_error() { } trap on_error ERR -# --- 3. Facts about this workspace ----------------------------------------- -# -# The runtime interface: who owns this workspace, where the deployment is, and -# anything the caller added through `values`. Mode 0644, because unprivileged -# services read it -- so no secrets. The token lives in agent.env at 0600 and -# never appears here. -# -# Rendered as JSON by Terraform rather than assembled in shell: a full name -# with a quote in it is somebody's Tuesday, and this way it cannot break the -# document. - +# Public facts must never contain secrets. cat >"$RUNTIME_DIR/workspace.json" <<'CODER_AMAZON_INIT_FACTS' ${FACTS_JSON} CODER_AMAZON_INIT_FACTS @@ -152,22 +97,19 @@ chmod 0644 "$RUNTIME_DIR/workspace.json" export CODER_WORKSPACE_FACTS="$RUNTIME_DIR/workspace.json" export CODER_RUNTIME_DIR="$RUNTIME_DIR" -# --- 4. Files and the boot script ------------------------------------------- - -${FILES_SH} +%{ for file in FILES ~} +file_path=$(decode '${file.path}') +install -d -m 0755 -- "$(dirname -- "$file_path")" +printf '%s' '${file.content}' | base64 -d | gzip -dc >"$file_path" +chmod 0644 -- "$file_path" +%{ endfor ~} cat >"$RUNTIME_DIR/boot.sh" <<'CODER_AMAZON_INIT_BOOT_SCRIPT' ${BOOT_SCRIPT} CODER_AMAZON_INIT_BOOT_SCRIPT chmod 0700 "$RUNTIME_DIR/boot.sh" -# Run as a child rather than sourced: the boot script is someone else's code, -# and it must not be able to exit this one before the agent is started. -# -# `|| boot_rc=$?` rather than turning errexit off around it: `set +e` stops -# the shell exiting but does nothing to the ERR trap, which is not subject to -# errexit at all. The trap would fire here and report this as a bootstrap -# failure, skipping everything below. In a `||` list neither happens. +# A child cannot exit the bootstrap before the agent-start attempt. boot_rc=0 bash "$RUNTIME_DIR/boot.sh" || boot_rc=$? @@ -175,9 +117,5 @@ if [ "$boot_rc" -ne 0 ]; then coder_log error "Boot script exited $boot_rc; starting the agent anyway." || true fi -# `|| true` so that failing to start the agent does not re-enter the ERR trap, -# which would report the bootstrap as having failed somewhere earlier. On a -# first boot whose configuration does not build there is no agent unit to -# start at all, and that is the honest outcome: nothing to report it with. start_agent || true exit "$boot_rc" diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.test.py b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.test.py new file mode 100644 index 000000000..b0b179089 --- /dev/null +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.test.py @@ -0,0 +1,91 @@ +"""Run directly: python3 scripts/bootstrap.test.py (needs passwordless sudo).""" +import base64 +import gzip +import os +from pathlib import Path +import re +import subprocess +import tempfile + +TEMPLATE = Path(__file__).with_name("bootstrap.sh.tftpl") +LIBRARY = Path(__file__).with_name("log.sh") + + +def b64(text): + return base64.b64encode((text.encode() if isinstance(text, str) else text)).decode() + + +with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + root.chmod(0o755) + fake = root / "bin" + fake.mkdir() + runtime = root / "runtime" + marker = root / "unit" + active = root / "active" + injected = root / "injected" + file_path = root / "quote'$(touch injected) 🐈" + (fake / "systemctl").write_text(f'''#!/usr/bin/env bash +case "$1" in + show) if test -e '{marker}'; then echo nobody; fi ;; + cat) test -e '{marker}' ;; + is-active) test -e '{active}' ;; + start) touch '{active}' ;; +esac +''') + (fake / "hostnamectl").write_text("#!/usr/bin/env bash\nexit 0\n") + (fake / "curl").write_text("#!/usr/bin/env bash\ncase \" $* \" in *POST*) printf 201;; esac\n") + (fake / "sleep").write_text("#!/usr/bin/env bash\nexit 0\n") + for command in fake.iterdir(): + command.chmod(0o755) + source = TEMPLATE.read_text() + substitutions = { + "ARG_PATH": b64(str(fake)), + "ARG_ACCESS_URL": b64("https://coder.example"), + "ARG_AGENT_TOKEN": b64("token"), + "ARG_RUNTIME_DIR": b64(str(runtime)), + "ARG_LOG_SOURCE_ID": "id", + "ARG_LOG_BUDGET": "4096", + "ARG_HOSTNAME": b64("test-host"), + "ARG_LOG_REGISTRATION_B64": b64('{"id":"id","display_name":"Boot","icon":"/icon"}'), + "INIT_SCRIPT": "#!/usr/bin/env bash\necho init", + "LOG_SH": LIBRARY.read_text(), + "FACTS_JSON": '{"workspace":"test"}', + "BOOT_SCRIPT": f"#!/usr/bin/env bash\ntouch '{marker}'", + } + pattern = r"%\{ for file in FILES ~\}(.*?)%\{ endfor ~\}" + loop = re.search(pattern, source, re.S) + assert loop + source = source[:loop.start()] + loop.group(1).replace("${file.path}", b64(str(file_path))).replace("${file.content}", b64(gzip.compress(b"safe text"))) + source[loop.end():] + for key, value in substitutions.items(): + source = source.replace("${" + key + "}", value) + assert not re.search(r"\$\{(?:ARG_|file\.)", source) + runtime.mkdir(mode=0o700) + bootstrap = runtime / "bootstrap.sh" + bootstrap.write_text(source) + bootstrap.chmod(0o700) + subprocess.run(["sudo", "-n", "chown", "root:root", str(runtime), str(bootstrap)], check=True) + env = os.environ | {"PATH": str(fake) + ":" + os.environ["PATH"]} + try: + for boot in range(2): + subprocess.run(["sudo", "-n", "env", "PATH=" + env["PATH"], "bash", str(bootstrap)], check=True, timeout=15) + assert file_path.read_text() == "safe text" + assert not injected.exists() + assert runtime.stat().st_uid == 0 and runtime.stat().st_mode & 0o777 == 0o711 + assert bootstrap.stat().st_uid == 0 and (runtime / "boot.sh").stat().st_uid == 0 + assert (runtime / "boot.sh").stat().st_mode & 0o077 == 0 + for handoff in ("agent.env", "init.sh"): + assert (runtime / handoff).stat().st_uid == 65534 + for attempt in (f"printf hacked >'{runtime}/boot.sh'", f"ln -sf /etc/passwd '{runtime}/boot.sh'"): + failed = subprocess.run(["sudo", "-n", "-u", "nobody", "bash", "-c", attempt], capture_output=True) + assert failed.returncode != 0, attempt + subprocess.run(["sudo", "-n", "-u", "nobody", "cat", str(runtime / "agent.env")], stdout=subprocess.DEVNULL, check=True) + # On a failed later boot the wrapper still attempts to start the agent. + broken = source.replace(f"touch '{marker}'", "exit 7") + subprocess.run(["sudo", "-n", "tee", str(bootstrap)], input=broken.encode(), stdout=subprocess.DEVNULL, check=True) + subprocess.run(["sudo", "-n", "rm", str(active)], check=True) + failed = subprocess.run(["sudo", "-n", "env", "PATH=" + env["PATH"], "bash", str(bootstrap)], timeout=15) + assert failed.returncode == 7 and active.exists() + print("two boots and failed boot: root scripts protected; non-root handoff; malicious path inert; agent attempted: OK") + finally: + subprocess.run(["sudo", "-n", "chown", "-R", str(os.getuid()), str(root)], check=True) diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh index b806bce87..807f68a0f 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh @@ -1,38 +1,20 @@ # shellcheck shell=bash -# -# Pushes lines to the Coder agent log API -- the only way to show anything in -# the workspace UI before the agent exists, which on a first boot lasts for as -# long as the boot script runs. -# -# Callers set CODER_ACCESS_URL, CODER_AGENT_TOKEN, CODER_LOG_SOURCE_ID and -# optionally CODER_LOG_BUDGET. Sets no shell options: failing to log must -# never be fatal. - +# Pre-agent log API client; failures never prevent boot. CODER_LOG_BUDGET="${CODER_LOG_BUDGET:-524288}" CODER_LOG_STATE_DIR="${CODER_LOG_STATE_DIR:-/run/coder}" CODER_LOG_MAX_LINE=2048 CODER_LOG_FLUSH_SECS="${CODER_LOG_FLUSH_SECS:-5}" -# Inherited, so a child process that sources this library can log without -# registering the source again -- registration is per workspace build, not per -# process, and a child has no way to know whether it already happened. CODER_LOG_READY="${CODER_LOG_READY:-0}" -# Resolved once. An image without curl gets no logs: every function here then -# fails closed, which is the right trade -- logging must never be the reason a -# boot fails. _curl() { if [ -z "${CODER_CURL:-}" ]; then - command -v curl > /dev/null 2>&1 || return 1 - CODER_CURL=$(command -v curl) + CODER_CURL=$(command -v curl) || return 1 export CODER_CURL fi "$CODER_CURL" "$@" } -# Coder caps agent logs at 1 MiB per AGENT, shared across every log source. -# Overflowing does not truncate: the batch is rejected and the agent is -# flagged overflowed permanently, silently dropping all later logs from every -# source. So track our own usage and go quiet before the server says no. +# A rejected overflow permanently silences every log source on this agent. coder_log_budget_left() { local used used=$(cat "$CODER_LOG_STATE_DIR/log-budget" 2> /dev/null || echo 0) @@ -42,28 +24,22 @@ coder_log_budget_left() { coder_log_budget_add() { local used used=$(cat "$CODER_LOG_STATE_DIR/log-budget" 2> /dev/null || echo 0) - echo $((used + $1)) > "$CODER_LOG_STATE_DIR/log-budget" 2> /dev/null || true + echo $((used + $1)) > "$CODER_LOG_STATE_DIR/log-budget" } -# Idempotent: Coder swallows a duplicate id, so this can run on every boot -# with a fixed UUID. Retries on 401 because the agent record is only created -# when the provisioner job completes -- an instance can boot before then. coder_log_init() { - local display_name="$1" icon="$2" attempt=0 code - mkdir -p "$CODER_LOG_STATE_DIR" 2> /dev/null || true - + local attempt=0 code + command -v curl > /dev/null 2>&1 || return 1 + mkdir -p "$CODER_LOG_STATE_DIR" || return 1 while [ "$attempt" -lt 40 ]; do code=$( _curl -sS -o /dev/null -w '%{http_code}' -X POST \ "$CODER_ACCESS_URL/api/v2/workspaceagents/me/log-source" \ -H "Coder-Session-Token: $CODER_AGENT_TOKEN" \ -H 'Content-Type: application/json' \ - --data-binary @- << JSON || echo 000 -{"id":"$CODER_LOG_SOURCE_ID","display_name":"$display_name","icon":"$icon"} -JSON + --data-binary "$CODER_LOG_REGISTRATION" || echo 000 ) case "$code" in - # The handler returns 201 on both create and already-exists. 200 | 201) export CODER_LOG_READY=1 return 0 @@ -78,30 +54,82 @@ JSON return 1 } +# Validate UTF-8 while escaping JSON; invalid sequences become U+FFFD. +_coder_json_string() { + LC_ALL=C od -An -tu1 -v | LC_ALL=C awk ' + function replacement() { printf "%c%c%c", 239, 191, 189 } + BEGIN { printf "\"" } + { + for (i = 1; i <= NF; i++) { + n = $i + 0 + if (pending) { + if (n >= low && n <= high) { + sequence = sequence sprintf("%c", n) + pending-- + low = 128; high = 191 + if (!pending) { printf "%s", sequence; sequence = "" } + continue + } + replacement() + pending = 0; sequence = "" + } + if (n == 34 || n == 92) printf "\\%c", n + else if (n < 32 || n == 127) printf "\\u%04x", n + else if (n < 128) printf "%c", n + else { + pending = 0; low = 128; high = 191 + if (n >= 194 && n <= 223) pending = 1 + else if (n >= 224 && n <= 239) { + pending = 2 + if (n == 224) low = 160 + if (n == 237) high = 159 + } else if (n >= 240 && n <= 244) { + pending = 3 + if (n == 240) low = 144 + if (n == 244) high = 143 + } + if (pending) sequence = sprintf("%c", n) + else replacement() + } + } + } + END { if (pending) replacement(); printf "\"" } + ' +} + coder_log_json() { - local level="$1" line="$2" - [ "${#line}" -le "$CODER_LOG_MAX_LINE" ] || line="${line:0:$CODER_LOG_MAX_LINE}..." - line=$(printf '%s' "$line" | sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\r//g' -e 's/\t/ /g') - printf '{"created_at":"%s","level":"%s","output":"%s"}' \ - "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$level" "$line" + local level="$1" line="$2" encoded + # Drop oversized lines rather than splitting a UTF-8 character mid-sequence. + if [ "$(printf '%s' "$line" | LC_ALL=C wc -c)" -gt "$CODER_LOG_MAX_LINE" ]; then + line='[log line exceeded 2048 bytes]' + fi + encoded=$(printf '%s' "$line" | _coder_json_string) || return 1 + printf '{"created_at":"%s","level":%s,"output":%s}' \ + "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$(printf '%s' "$level" | _coder_json_string)" "$encoded" } coder_log_send() { - local payload size - payload=$(paste -sd, -) + local payload request size lock="$CODER_LOG_STATE_DIR/log-budget.lock" tries=0 + payload=$(paste -sd, -) || return 0 [ -n "$payload" ] || return 0 - - size=${#payload} - [ "$(coder_log_budget_left)" -gt "$size" ] || return 0 - coder_log_budget_add "$size" - + request="{\"log_source_id\":\"$CODER_LOG_SOURCE_ID\",\"logs\":[$payload]}" + size=$(printf '%s' "$request" | LC_ALL=C wc -c) + # mkdir is an atomic cross-process lock; fail closed if a writer is stuck. + while ! mkdir "$lock" 2> /dev/null; do + tries=$((tries + 1)) + [ "$tries" -lt 40 ] || return 0 + sleep 0.05 + done + if [ "$(coder_log_budget_left)" -lt "$size" ] || ! coder_log_budget_add "$size"; then + rmdir "$lock" + return 0 + fi + rmdir "$lock" _curl -sS -o /dev/null -X PATCH \ "$CODER_ACCESS_URL/api/v2/workspaceagents/me/logs" \ -H "Coder-Session-Token: $CODER_AGENT_TOKEN" \ -H 'Content-Type: application/json' \ - --data-binary @- << JSON || true -{"log_source_id":"$CODER_LOG_SOURCE_ID","logs":[$payload]} -JSON + --data-binary "$request" || true } coder_log() { @@ -111,12 +139,7 @@ coder_log() { coder_log_json "$level" "$*" | coder_log_send } -# Reads plain lines on stdin and ships them in batches. -# -# Batches are flushed by size, and also by time: a slow producer would -# otherwise sit in the buffer until 50 lines had accumulated, which reads as a -# hung workspace, and anything still buffered when the process is killed is -# simply lost. +# Flush slow producers as well as full batches. coder_log_pipe() { local level="${1:-info}" line batch="" n=0 bytes=0 obj rc now last [ "$CODER_LOG_READY" = 1 ] || { @@ -124,32 +147,24 @@ coder_log_pipe() { return 0 } last=$(date +%s) - while :; do line="" - # Not `if ! read`: `!` rewrites $? to 0, which turns every timeout into an - # end of input and ends the stream after the first quiet interval. - IFS= read -r -t "$CODER_LOG_FLUSH_SECS" line - rc=$? - - # read(1) returns >128 on timeout; any other failure is end of input, - # possibly with a last line that had no newline. + if IFS= read -r -t "$CODER_LOG_FLUSH_SECS" line; then rc=0; else rc=$?; fi if [ "$rc" -ne 0 ] && [ "$rc" -le 128 ]; then if [ -n "$line" ]; then + obj=$(coder_log_json "$level" "$line") batch="${batch:+$batch -}$(coder_log_json "$level" "$line")" +}$obj" fi break fi - if [ -n "$line" ]; then obj=$(coder_log_json "$level" "$line") batch="${batch:+$batch }$obj" n=$((n + 1)) - bytes=$((bytes + ${#obj})) + bytes=$((bytes + $(printf '%s' "$obj" | LC_ALL=C wc -c))) fi - now=$(date +%s) if [ "$n" -ge 50 ] || [ "$bytes" -ge 32768 ] \ || { [ "$n" -gt 0 ] && [ "$((now - last))" -ge "$CODER_LOG_FLUSH_SECS" ]; }; then @@ -160,12 +175,10 @@ coder_log_pipe() { last=$now fi done - [ -z "$batch" ] || printf '%s\n' "$batch" | coder_log_send return 0 } -# Tail of a failed transcript, so the UI shows the actual error. coder_log_tail() { local file="$1" lines="${2:-200}" [ -f "$file" ] || return 0 diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.test.py b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.test.py new file mode 100644 index 000000000..3e50dc2fa --- /dev/null +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.test.py @@ -0,0 +1,64 @@ +"""Run directly: python3 scripts/log.test.py (no Terraform provider needed).""" +import json +import os +from pathlib import Path +import subprocess +import tempfile + +LIBRARY = Path(__file__).with_name("log.sh") + +with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + curl = root / "curl" + curl.write_text("""#!/usr/bin/env python3 +import fcntl, json, os, sys +args = sys.argv +body = args[args.index('--data-binary') + 1] +with open(os.environ['REQUESTS'], 'a') as output: + fcntl.flock(output, fcntl.LOCK_EX) + output.write(json.dumps({'url': args[args.index('-X') + 2], 'body': body}) + '\\n') +if 'POST' in args: print('201', end='') +""") + curl.chmod(0o755) + env = os.environ | { + "PATH": str(root) + ":" + os.environ["PATH"], + "REQUESTS": str(root / "requests"), + "CODER_LOG_STATE_DIR": str(root / "state"), + "CODER_LOG_BUDGET": "4000", + "CODER_ACCESS_URL": "https://coder.example", + "CODER_AGENT_TOKEN": "token", + "CODER_LOG_SOURCE_ID": "id", + "CODER_LOG_REGISTRATION": json.dumps({"id": "id", "display_name": 'quoted " name 🐈', "icon": "/icon"}), + } + missing = subprocess.run(["bash", "--noprofile", "--norc", "-c", f'PATH={root / "empty"}; source {LIBRARY}; coder_log_init'], env=env, timeout=2) + assert missing.returncode == 1 + registration = subprocess.run( + ["bash", "--noprofile", "--norc", "-e", "-u", "-c", f"source {LIBRARY}; coder_log_init"], env=env, capture_output=True + ) + assert registration.returncode == 0, registration.stderr + env["CODER_LOG_READY"] = "1" + message = 'quote " backslash \\ tab\t CR\r unicode 🐈' + program = f'source {LIBRARY}; coder_log info "$MESSAGE"; printf "slow\\nlast" | coder_log_pipe info' + env["MESSAGE"] = message + subprocess.run(["bash", "--noprofile", "--norc", "-e", "-u", "-c", program], env=env, check=True) + requests = [json.loads(line) for line in (root / "requests").read_text().splitlines()] + assert json.loads(requests[0]["body"])["display_name"] == 'quoted " name 🐈' + logs = [item for request in requests[1:] for item in json.loads(request["body"])["logs"]] + assert [entry["output"] for entry in logs] == [message, "slow", "last"] + # Invalid leading, truncated, overlong, and surrogate encodings must not + # produce raw invalid UTF-8 in the JSON sent to Coder. + malformed = f"source {LIBRARY}; coder_log info \"$(printf 'bad\\377\\360\\237\\300\\257\\355\\240\\200')\"" + subprocess.run(["bash", "--noprofile", "--norc", "-e", "-u", "-c", malformed], env=env, check=True) + requests = [json.loads(line) for line in (root / "requests").read_text().splitlines()] + malformed_output = json.loads(requests[-1]["body"])["logs"][0]["output"] + assert malformed_output.startswith("bad\ufffd") and "\ufffd" in malformed_output + assert malformed_output.encode("utf-8").decode("utf-8") == malformed_output + before = len(requests) + workers = [subprocess.Popen(["bash", "--noprofile", "--norc", "-e", "-u", "-c", f'source {LIBRARY}; for ((i=0;i<30;i++)); do coder_log info "message-🐈-$i"; done'], env=env) for _ in range(4)] + assert all(worker.wait() == 0 for worker in workers) + requests = [json.loads(line) for line in (root / "requests").read_text().splitlines()] + assert len(requests) > before + charged = sum(len(request["body"].encode()) for request in requests[1:]) + assert charged <= 4000, charged + assert int((root / "state" / "log-budget").read_text()) == charged + print("log registration, JSON controls/Unicode/malformed UTF-8, timeout/EOF and concurrent budget: OK") diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf index f3dc3496d..bbc961397 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf @@ -3,19 +3,12 @@ terraform { } variable "flake_ref" { - description = <<-EOT - Git reference to the flake, in the form `nix` itself accepts: - `https://host/org/repo`, optionally with a `git+` prefix and a `?ref=` - branch. Without `?ref=` the remote's default branch is used. - - The configuration must be committed -- a Git flake reference only ever - sees committed files. - EOT + description = "HTTP(S) or SSH Git URL of a committed flake; optional git+ prefix and ?ref= branch. URL credentials are not supported." type = string validation { - condition = can(regex("^(git\\+)?(https?|ssh)://", var.flake_ref)) && !can(regex("^(git\\+)?https?://[^/?#]*@", var.flake_ref)) && !can(regex("[[:cntrl:]]", var.flake_ref)) - error_message = "flake_ref must be an http(s) or ssh Git URL without HTTP credentials or control characters. Use root-managed Git authentication instead of URL userinfo." + condition = can(regex("^(git\\+)?(https?|ssh)://", var.flake_ref)) && !can(regex("^(git\\+)?https?://[^/?#]*@", var.flake_ref)) && !can(regex("[[:cntrl:]]", var.flake_ref)) && !strcontains(var.flake_ref, "#") && (!strcontains(var.flake_ref, "?") || can(regex("\\?ref=[^&#?]+$", var.flake_ref))) + error_message = "flake_ref must be an http(s) or ssh Git URL with optional ?ref=, without HTTP credentials, control characters, fragments, or other query parameters." } } diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl index 6b20b85af..75a7f64e1 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl @@ -44,3 +44,11 @@ run "reject_control_characters" { } expect_failures = [var.flake_ref] } + +run "reject_unsupported_query" { + command = plan + variables { + flake_ref = "https://example.org/flake?dir=subdir" + } + expect_failures = [var.flake_ref] +} diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.test.sh b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.test.sh index 0bc4a1ba1..99aa71fb0 100755 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.test.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/scripts/lifecycle.test.sh @@ -32,14 +32,23 @@ nix_sync_checkout "file://$root/remote" main rev=$(nix_needs_rebuild) [ "$rev" = "$(git -C "$NIX_FLAKE_DIR" rev-parse HEAD)#host-one" ] nix_record_rev "$rev" -if nix_needs_rebuild > /dev/null; then echo 'unexpected rebuild' >&2; exit 1; fi +if nix_needs_rebuild > /dev/null; then + echo 'unexpected rebuild' >&2 + exit 1 +fi NIX_FLAKE_ATTR=host-two nix_needs_rebuild > /dev/null nix_record_rev "$(nix_needs_rebuild)" -if nix_needs_rebuild > /dev/null; then echo 'unexpected attribute rebuild' >&2; exit 1; fi +if nix_needs_rebuild > /dev/null; then + echo 'unexpected attribute rebuild' >&2 + exit 1 +fi printf 'ignored by Git flake\n' > "$NIX_FLAKE_DIR/untracked" -if nix_needs_rebuild > /dev/null; then echo 'untracked file caused rebuild' >&2; exit 1; fi +if nix_needs_rebuild > /dev/null; then + echo 'untracked file caused rebuild' >&2 + exit 1 +fi printf 'modified\n' > "$NIX_FLAKE_DIR/flake.nix" nix_checkout_dirty nix_needs_rebuild > /dev/null diff --git a/registry/coder-labs/templates/aws-nixos/tests/architecture.tftest.hcl b/registry/coder-labs/templates/aws-nixos/tests/architecture.tftest.hcl new file mode 100644 index 000000000..a9dbc9e7c --- /dev/null +++ b/registry/coder-labs/templates/aws-nixos/tests/architecture.tftest.hcl @@ -0,0 +1,119 @@ +mock_provider "coder" { + mock_data "coder_workspace" { + defaults = { + name = "test" + start_count = 1 + transition = "start" + } + } + mock_data "coder_workspace_owner" { + defaults = { + name = "owner" + full_name = "Owner" + email = "owner@example.org" + } + } + mock_data "coder_parameter" { + defaults = { + value = "80" + } + } +} + +mock_provider "aws" { + mock_data "aws_ami" { + defaults = { + id = "ami-example" + name = "nixos/latest" + } + } +} + +mock_provider "http" {} +mock_provider "random" {} + +override_module { + target = module.aws-region + outputs = { + value = "eu-west-3" + default_availability_zone = "eu-west-3a" + } +} + +override_module { + target = module.aws-ec2-instance-type + outputs = { + value = "t3.medium" + instances = { + "t3.medium" = { arch = "x86_64", coder_arch = "amd64" } + "t4g.medium" = { arch = "arm64", coder_arch = "arm64" } + } + } +} + +override_module { + target = module.code-server + outputs = {} +} +override_module { + target = module.jetbrains-gateway + outputs = {} +} +override_module { + target = module.git-config + outputs = {} +} + +run "x86" { + command = plan + assert { + condition = local.nix_arch == "x86_64" && coder_agent.main[0].arch == "amd64" && module.nix.flake_attr == "coder-workspace-ec2-x86_64" + error_message = "The x86 instance, agent and flake attribute must agree." + } + assert { + condition = aws_instance.dev.ami == "ami-example" && coder_metadata.workspace_info.item[0].value == aws_instance.dev.ami + error_message = "AMI metadata must show the instance's actual AMI ID." + } +} + +run "arm" { + command = plan + override_module { + target = module.aws-ec2-instance-type + outputs = { + value = "t4g.medium" + instances = { + "t3.medium" = { arch = "x86_64", coder_arch = "amd64" } + "t4g.medium" = { arch = "arm64", coder_arch = "arm64" } + } + } + } + assert { + condition = local.nix_arch == "aarch64" && coder_agent.main[0].arch == "arm64" && module.nix.flake_attr == "coder-workspace-ec2-aarch64" + error_message = "The ARM instance, agent and flake attribute must agree." + } +} + +run "stop" { + command = plan + override_data { + target = data.coder_workspace.me + values = { + name = "test" + start_count = 0 + transition = "stop" + } + } + assert { + condition = length(coder_agent.main) == 0 && aws_ec2_instance_state.dev.state == "stopped" + error_message = "Stopping must remove the agent and stop, not destroy, the EC2 instance." + } +} + +run "reject_unsupported_query" { + command = plan + variables { + flake_ref = "https://example.org/flake?dir=subdir" + } + expect_failures = [var.flake_ref] +} From 713bdc6fe7a214af4194794a3398f6344ef032f6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Phorcys=20=F0=9F=90=BE?= Date: Tue, 29 Sep 2026 10:56:23 +0000 Subject: [PATCH 24/24] refactor(aws-nixos): simplify bootstrap and loosen optional inputs Use the previous lightweight log escaping, shrink the EC2 user-data wrapper, accept optional HTTP URL credentials, and keep only absolute-path validation for bootstrap files. Restore template descriptions, preserve unknown instance architecture instead of guessing x86, and remove the standalone Python boot/log tests. --- .../coder-labs/templates/aws-nixos/README.md | 2 +- .../coder-labs/templates/aws-nixos/main.tf | 24 +++-- .../aws-nixos/modules/amazon-init/README.md | 4 +- .../aws-nixos/modules/amazon-init/main.tf | 20 +--- .../modules/amazon-init/main.tftest.hcl | 71 --------------- .../amazon-init/scripts/bootstrap.test.py | 91 ------------------- .../modules/amazon-init/scripts/log.sh | 56 +----------- .../modules/amazon-init/scripts/log.test.py | 64 ------------- .../templates/aws-nixos/modules/nix/README.md | 2 +- .../templates/aws-nixos/modules/nix/main.tf | 6 +- .../aws-nixos/modules/nix/main.tftest.hcl | 7 +- .../aws-nixos/tests/architecture.tftest.hcl | 11 +++ 12 files changed, 51 insertions(+), 307 deletions(-) delete mode 100644 registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.test.py delete mode 100644 registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.test.py diff --git a/registry/coder-labs/templates/aws-nixos/README.md b/registry/coder-labs/templates/aws-nixos/README.md index 23e293e2f..b193ff8a4 100644 --- a/registry/coder-labs/templates/aws-nixos/README.md +++ b/registry/coder-labs/templates/aws-nixos/README.md @@ -56,6 +56,6 @@ aws ec2 get-console-output --instance-id --output text Do not put secrets in Nix expressions: the Nix store is readable on the VM. Workspace facts and optional bootstrap files are not secret storage. The agent token is kept out of Nix, but EC2 user-data and Terraform state contain it; restrict access to both. Processes with instance-metadata access can read user-data. -Private repos need root Git credentials before first boot and root Nix input credentials; this template supplies neither. HTTP URL credentials in `flake_ref` are rejected. NixOS scripts need `#!/usr/bin/env bash`; downloaded IDE binaries need `programs.nix-ld`. +Private repos need Git authentication before first boot and separate credentials for private Nix inputs; this template supplies neither. HTTP credentials in `flake_ref` are allowed but exposed in Terraform state, EC2 user-data, Coder metadata, and `/etc/nixos/.git/config`. Prefer root-managed credentials. NixOS scripts need `#!/usr/bin/env bash`; downloaded IDE binaries need `programs.nix-ld`. Existing templates may retain a stored legacy `flake_attr`; update that variable explicitly before rebuilding against renamed example-flake hosts. diff --git a/registry/coder-labs/templates/aws-nixos/main.tf b/registry/coder-labs/templates/aws-nixos/main.tf index 901abbd5f..454e01a18 100644 --- a/registry/coder-labs/templates/aws-nixos/main.tf +++ b/registry/coder-labs/templates/aws-nixos/main.tf @@ -21,13 +21,20 @@ provider "aws" { } variable "flake_ref" { - description = "Git URL of the NixOS flake; optional ?ref= selects a branch. No credentials in the URL." + description = <<-EOT + Git reference to the NixOS configuration, in the form `nix` itself + accepts: `https://host/org/repo`, optionally with a `git+` prefix and a + `?ref=` branch. Without `?ref=` the remote's default branch is used. + + The configuration must be committed -- a Git flake reference only ever + sees committed files. + EOT type = string default = "https://github.com/coder/nixos-example-flake" validation { - condition = can(regex("^(git\\+)?(https?|ssh)://", var.flake_ref)) && !can(regex("^(git\\+)?https?://[^/?#]*@", var.flake_ref)) && !can(regex("[[:cntrl:]]", var.flake_ref)) && !strcontains(var.flake_ref, "#") && (!strcontains(var.flake_ref, "?") || can(regex("\\?ref=[^&#?]+$", var.flake_ref))) - error_message = "Use an http(s) or ssh Git URL without HTTP credentials, control characters, fragments, or query parameters other than ?ref=." + condition = can(regex("^(git\\+)?(https?|ssh)://", var.flake_ref)) && !can(regex("[[:cntrl:]]", var.flake_ref)) && !strcontains(var.flake_ref, "#") && (!strcontains(var.flake_ref, "?") || can(regex("\\?ref=[^&#?]+$", var.flake_ref))) + error_message = "Use an http(s) or ssh Git URL without control characters, fragments, or query parameters other than ?ref=." } } @@ -51,8 +58,13 @@ module "aws-ec2-instance-type" { source = "registry.coder.com/coder/aws-ec2-instance-type/coder" version = "~> 1.0" - default = "t3.medium" - description = "Choose enough memory for NixOS builds. The smallest sizes may run out of memory." + default = "t3.medium" + description = trimspace(<<-EOT + t3.medium is the smallest that works: the NixOS AMI configures no swap and + the Nix store shares the root volume, so a rebuild that has to compile + anything will exhaust a 1-2 GiB instance. + EOT + ) include = [ "t3", "t4g", @@ -164,7 +176,7 @@ module "git-config" { locals { instance = module.aws-ec2-instance-type.instances[module.aws-ec2-instance-type.value] - nix_arch = local.instance.arch == "arm64" ? "aarch64" : "x86_64" + nix_arch = local.instance.arch == "arm64" ? "aarch64" : local.instance.arch } module "nix" { diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md index 50f896801..7136b3607 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/README.md @@ -21,5 +21,5 @@ public `values` or `files` (0644). The root boot script receives `CODER_RUNTIME_DIR`, `CODER_WORKSPACE_FACTS`, `CODER_ACCESS_URL`, `CODER_AGENT_TOKEN`, `CODER_LOG_SOURCE_ID`, and `CODER_LOG_LIBRARY`. Early logs require outbound Coder access and `curl`. The shared 1 MiB -agent log cap is budgeted. Use trusted file directories: symlinked ancestors -can redirect writes despite lexical path checks. +agent log cap is budgeted. `files` paths are trusted admin input: do not +write into `runtime_dir` or through symlinked directories. diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf index 0081139e9..8ab823eb5 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tf @@ -69,15 +69,8 @@ variable "files" { default = {} validation { - condition = alltrue([ - for path in keys(var.files) : - startswith(path, "/") && abspath(path) == path && - !can(regex("[\\x00-\\x1f\\x7f]", path)) && - path != var.runtime_dir && !startswith(path, "${var.runtime_dir}/") && - !(startswith(var.runtime_dir, "/run/") && startswith(path, "/var${var.runtime_dir}/")) && - !(startswith(var.runtime_dir, "/var/run/") && startswith(path, "${trimprefix(var.runtime_dir, "/var")}/")) - ]) - error_message = "File paths must be absolute, canonical files outside runtime_dir (including /var/run aliases)." + condition = alltrue([for path in keys(var.files) : startswith(path, "/")]) + error_message = "File paths must be absolute." } } @@ -193,15 +186,12 @@ locals { # EC2 caps user-data at 16 KiB; compress the bootstrap before sending it. user_data = <<-SH #!/usr/bin/env bash - set -eu + set -euo pipefail runtime_dir=$(printf %s '${base64encode(var.runtime_dir)}' | base64 -d) - install -d -m 0700 -- "$runtime_dir" - chown root:root -- "$runtime_dir" - chmod 0700 -- "$runtime_dir" - base64 -d <<'CODER_PAYLOAD' | gzip -dc >"$runtime_dir/bootstrap.sh" + install -d -m 0700 -o root -g root -- "$runtime_dir" + base64 -d <<'CODER_PAYLOAD' | gzip -dc | install -m 0700 -o root -g root /dev/stdin "$runtime_dir/bootstrap.sh" ${base64gzip(local.bootstrap)} CODER_PAYLOAD - chmod 0700 -- "$runtime_dir/bootstrap.sh" exec bash "$runtime_dir/bootstrap.sh" SH } diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tftest.hcl b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tftest.hcl index 645389013..6c0dfb410 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tftest.hcl +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/main.tftest.hcl @@ -46,77 +46,6 @@ run "reject_relative_path" { expect_failures = [var.files] } -run "reject_parent_traversal" { - command = plan - variables { - agent_token = "token" - agent_init_script = "echo init" - files = { "/tmp/../sneaky" = "bad" } - } - expect_failures = [var.files] -} - -run "reject_newline_path" { - command = plan - variables { - agent_token = "token" - agent_init_script = "echo init" - files = { "/tmp/new\nline" = "bad" } - } - expect_failures = [var.files] -} - -run "reject_handoff_overwrite" { - command = plan - variables { - agent_token = "token" - agent_init_script = "echo init" - files = { "/run/coder/agent.env" = "bad" } - } - expect_failures = [var.files] -} - -run "reject_duplicate_separator" { - command = plan - variables { - agent_token = "token" - agent_init_script = "echo init" - files = { "/run//coder/agent.env" = "bad" } - } - expect_failures = [var.files] -} - -run "reject_dot_segment" { - command = plan - variables { - agent_token = "token" - agent_init_script = "echo init" - files = { "/run/./coder/agent.env" = "bad" } - } - expect_failures = [var.files] -} - -run "reject_var_run_alias" { - command = plan - variables { - agent_token = "token" - agent_init_script = "echo init" - files = { "/var/run/coder/agent.env" = "bad" } - } - expect_failures = [var.files] -} - -run "reject_reverse_var_run_alias" { - command = plan - variables { - agent_token = "token" - agent_init_script = "echo init" - runtime_dir = "/var/run/coder" - files = { "/run/coder/agent.env" = "bad" } - } - expect_failures = [var.files] -} - run "reject_noncanonical_runtime" { command = plan variables { diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.test.py b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.test.py deleted file mode 100644 index b0b179089..000000000 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/bootstrap.test.py +++ /dev/null @@ -1,91 +0,0 @@ -"""Run directly: python3 scripts/bootstrap.test.py (needs passwordless sudo).""" -import base64 -import gzip -import os -from pathlib import Path -import re -import subprocess -import tempfile - -TEMPLATE = Path(__file__).with_name("bootstrap.sh.tftpl") -LIBRARY = Path(__file__).with_name("log.sh") - - -def b64(text): - return base64.b64encode((text.encode() if isinstance(text, str) else text)).decode() - - -with tempfile.TemporaryDirectory() as directory: - root = Path(directory) - root.chmod(0o755) - fake = root / "bin" - fake.mkdir() - runtime = root / "runtime" - marker = root / "unit" - active = root / "active" - injected = root / "injected" - file_path = root / "quote'$(touch injected) 🐈" - (fake / "systemctl").write_text(f'''#!/usr/bin/env bash -case "$1" in - show) if test -e '{marker}'; then echo nobody; fi ;; - cat) test -e '{marker}' ;; - is-active) test -e '{active}' ;; - start) touch '{active}' ;; -esac -''') - (fake / "hostnamectl").write_text("#!/usr/bin/env bash\nexit 0\n") - (fake / "curl").write_text("#!/usr/bin/env bash\ncase \" $* \" in *POST*) printf 201;; esac\n") - (fake / "sleep").write_text("#!/usr/bin/env bash\nexit 0\n") - for command in fake.iterdir(): - command.chmod(0o755) - source = TEMPLATE.read_text() - substitutions = { - "ARG_PATH": b64(str(fake)), - "ARG_ACCESS_URL": b64("https://coder.example"), - "ARG_AGENT_TOKEN": b64("token"), - "ARG_RUNTIME_DIR": b64(str(runtime)), - "ARG_LOG_SOURCE_ID": "id", - "ARG_LOG_BUDGET": "4096", - "ARG_HOSTNAME": b64("test-host"), - "ARG_LOG_REGISTRATION_B64": b64('{"id":"id","display_name":"Boot","icon":"/icon"}'), - "INIT_SCRIPT": "#!/usr/bin/env bash\necho init", - "LOG_SH": LIBRARY.read_text(), - "FACTS_JSON": '{"workspace":"test"}', - "BOOT_SCRIPT": f"#!/usr/bin/env bash\ntouch '{marker}'", - } - pattern = r"%\{ for file in FILES ~\}(.*?)%\{ endfor ~\}" - loop = re.search(pattern, source, re.S) - assert loop - source = source[:loop.start()] + loop.group(1).replace("${file.path}", b64(str(file_path))).replace("${file.content}", b64(gzip.compress(b"safe text"))) + source[loop.end():] - for key, value in substitutions.items(): - source = source.replace("${" + key + "}", value) - assert not re.search(r"\$\{(?:ARG_|file\.)", source) - runtime.mkdir(mode=0o700) - bootstrap = runtime / "bootstrap.sh" - bootstrap.write_text(source) - bootstrap.chmod(0o700) - subprocess.run(["sudo", "-n", "chown", "root:root", str(runtime), str(bootstrap)], check=True) - env = os.environ | {"PATH": str(fake) + ":" + os.environ["PATH"]} - try: - for boot in range(2): - subprocess.run(["sudo", "-n", "env", "PATH=" + env["PATH"], "bash", str(bootstrap)], check=True, timeout=15) - assert file_path.read_text() == "safe text" - assert not injected.exists() - assert runtime.stat().st_uid == 0 and runtime.stat().st_mode & 0o777 == 0o711 - assert bootstrap.stat().st_uid == 0 and (runtime / "boot.sh").stat().st_uid == 0 - assert (runtime / "boot.sh").stat().st_mode & 0o077 == 0 - for handoff in ("agent.env", "init.sh"): - assert (runtime / handoff).stat().st_uid == 65534 - for attempt in (f"printf hacked >'{runtime}/boot.sh'", f"ln -sf /etc/passwd '{runtime}/boot.sh'"): - failed = subprocess.run(["sudo", "-n", "-u", "nobody", "bash", "-c", attempt], capture_output=True) - assert failed.returncode != 0, attempt - subprocess.run(["sudo", "-n", "-u", "nobody", "cat", str(runtime / "agent.env")], stdout=subprocess.DEVNULL, check=True) - # On a failed later boot the wrapper still attempts to start the agent. - broken = source.replace(f"touch '{marker}'", "exit 7") - subprocess.run(["sudo", "-n", "tee", str(bootstrap)], input=broken.encode(), stdout=subprocess.DEVNULL, check=True) - subprocess.run(["sudo", "-n", "rm", str(active)], check=True) - failed = subprocess.run(["sudo", "-n", "env", "PATH=" + env["PATH"], "bash", str(bootstrap)], timeout=15) - assert failed.returncode == 7 and active.exists() - print("two boots and failed boot: root scripts protected; non-root handoff; malicious path inert; agent attempted: OK") - finally: - subprocess.run(["sudo", "-n", "chown", "-R", str(os.getuid()), str(root)], check=True) diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh index 807f68a0f..7f54ae3c2 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh +++ b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.sh @@ -54,58 +54,12 @@ coder_log_init() { return 1 } -# Validate UTF-8 while escaping JSON; invalid sequences become U+FFFD. -_coder_json_string() { - LC_ALL=C od -An -tu1 -v | LC_ALL=C awk ' - function replacement() { printf "%c%c%c", 239, 191, 189 } - BEGIN { printf "\"" } - { - for (i = 1; i <= NF; i++) { - n = $i + 0 - if (pending) { - if (n >= low && n <= high) { - sequence = sequence sprintf("%c", n) - pending-- - low = 128; high = 191 - if (!pending) { printf "%s", sequence; sequence = "" } - continue - } - replacement() - pending = 0; sequence = "" - } - if (n == 34 || n == 92) printf "\\%c", n - else if (n < 32 || n == 127) printf "\\u%04x", n - else if (n < 128) printf "%c", n - else { - pending = 0; low = 128; high = 191 - if (n >= 194 && n <= 223) pending = 1 - else if (n >= 224 && n <= 239) { - pending = 2 - if (n == 224) low = 160 - if (n == 237) high = 159 - } else if (n >= 240 && n <= 244) { - pending = 3 - if (n == 240) low = 144 - if (n == 244) high = 143 - } - if (pending) sequence = sprintf("%c", n) - else replacement() - } - } - } - END { if (pending) replacement(); printf "\"" } - ' -} - coder_log_json() { - local level="$1" line="$2" encoded - # Drop oversized lines rather than splitting a UTF-8 character mid-sequence. - if [ "$(printf '%s' "$line" | LC_ALL=C wc -c)" -gt "$CODER_LOG_MAX_LINE" ]; then - line='[log line exceeded 2048 bytes]' - fi - encoded=$(printf '%s' "$line" | _coder_json_string) || return 1 - printf '{"created_at":"%s","level":%s,"output":%s}' \ - "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$(printf '%s' "$level" | _coder_json_string)" "$encoded" + local level="$1" line="$2" + [ "${#line}" -le "$CODER_LOG_MAX_LINE" ] || line="${line:0:$CODER_LOG_MAX_LINE}..." + line=$(printf '%s' "$line" | sed -e 's/\\/\\\\/g' -e 's/"/\\"/g' -e 's/\r//g' -e 's/\t/ /g') + printf '{"created_at":"%s","level":"%s","output":"%s"}' \ + "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$level" "$line" } coder_log_send() { diff --git a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.test.py b/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.test.py deleted file mode 100644 index 3e50dc2fa..000000000 --- a/registry/coder-labs/templates/aws-nixos/modules/amazon-init/scripts/log.test.py +++ /dev/null @@ -1,64 +0,0 @@ -"""Run directly: python3 scripts/log.test.py (no Terraform provider needed).""" -import json -import os -from pathlib import Path -import subprocess -import tempfile - -LIBRARY = Path(__file__).with_name("log.sh") - -with tempfile.TemporaryDirectory() as temporary: - root = Path(temporary) - curl = root / "curl" - curl.write_text("""#!/usr/bin/env python3 -import fcntl, json, os, sys -args = sys.argv -body = args[args.index('--data-binary') + 1] -with open(os.environ['REQUESTS'], 'a') as output: - fcntl.flock(output, fcntl.LOCK_EX) - output.write(json.dumps({'url': args[args.index('-X') + 2], 'body': body}) + '\\n') -if 'POST' in args: print('201', end='') -""") - curl.chmod(0o755) - env = os.environ | { - "PATH": str(root) + ":" + os.environ["PATH"], - "REQUESTS": str(root / "requests"), - "CODER_LOG_STATE_DIR": str(root / "state"), - "CODER_LOG_BUDGET": "4000", - "CODER_ACCESS_URL": "https://coder.example", - "CODER_AGENT_TOKEN": "token", - "CODER_LOG_SOURCE_ID": "id", - "CODER_LOG_REGISTRATION": json.dumps({"id": "id", "display_name": 'quoted " name 🐈', "icon": "/icon"}), - } - missing = subprocess.run(["bash", "--noprofile", "--norc", "-c", f'PATH={root / "empty"}; source {LIBRARY}; coder_log_init'], env=env, timeout=2) - assert missing.returncode == 1 - registration = subprocess.run( - ["bash", "--noprofile", "--norc", "-e", "-u", "-c", f"source {LIBRARY}; coder_log_init"], env=env, capture_output=True - ) - assert registration.returncode == 0, registration.stderr - env["CODER_LOG_READY"] = "1" - message = 'quote " backslash \\ tab\t CR\r unicode 🐈' - program = f'source {LIBRARY}; coder_log info "$MESSAGE"; printf "slow\\nlast" | coder_log_pipe info' - env["MESSAGE"] = message - subprocess.run(["bash", "--noprofile", "--norc", "-e", "-u", "-c", program], env=env, check=True) - requests = [json.loads(line) for line in (root / "requests").read_text().splitlines()] - assert json.loads(requests[0]["body"])["display_name"] == 'quoted " name 🐈' - logs = [item for request in requests[1:] for item in json.loads(request["body"])["logs"]] - assert [entry["output"] for entry in logs] == [message, "slow", "last"] - # Invalid leading, truncated, overlong, and surrogate encodings must not - # produce raw invalid UTF-8 in the JSON sent to Coder. - malformed = f"source {LIBRARY}; coder_log info \"$(printf 'bad\\377\\360\\237\\300\\257\\355\\240\\200')\"" - subprocess.run(["bash", "--noprofile", "--norc", "-e", "-u", "-c", malformed], env=env, check=True) - requests = [json.loads(line) for line in (root / "requests").read_text().splitlines()] - malformed_output = json.loads(requests[-1]["body"])["logs"][0]["output"] - assert malformed_output.startswith("bad\ufffd") and "\ufffd" in malformed_output - assert malformed_output.encode("utf-8").decode("utf-8") == malformed_output - before = len(requests) - workers = [subprocess.Popen(["bash", "--noprofile", "--norc", "-e", "-u", "-c", f'source {LIBRARY}; for ((i=0;i<30;i++)); do coder_log info "message-🐈-$i"; done'], env=env) for _ in range(4)] - assert all(worker.wait() == 0 for worker in workers) - requests = [json.loads(line) for line in (root / "requests").read_text().splitlines()] - assert len(requests) > before - charged = sum(len(request["body"].encode()) for request in requests[1:]) - assert charged <= 4000, charged - assert int((root / "state" / "log-budget").read_text()) == charged - print("log registration, JSON controls/Unicode/malformed UTF-8, timeout/EOF and concurrent budget: OK") diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/README.md b/registry/coder-labs/templates/aws-nixos/modules/nix/README.md index 9f9891f3c..8426f3534 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/README.md +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/README.md @@ -11,7 +11,7 @@ module "nix" { } ``` -References accept HTTP(S) or SSH Git URLs, optional `git+` and `?ref=`. Without `ref`, Git follows the default branch. `$ARCH` expands to `arch`. HTTP URL userinfo is rejected: migrate embedded passwords or tokens to root-managed authentication. The instance requires outbound Git and Nix input/substituter access, root privileges, systemd, and NixOS. +References accept HTTP(S) or SSH Git URLs, optional `git+` and `?ref=`. Without `ref`, Git follows the default branch. `$ARCH` expands to `arch`. HTTP URL userinfo is allowed but leaks into the checkout and `flake_uri` output; prefer root-managed authentication. The instance requires outbound Git and Nix input/substituter access, root privileges, systemd, and NixOS. A clean checkout fast-forwards; tracked edits and local commits remain untouched. Untracked files do not trigger builds: Git flakes ignore them. The `state_dir` lock prevents races only with callers that take it. Rebuild transcripts live in `log_dir`. First-boot failures may require AWS logs before the agent exists. diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf index bbc961397..b5463a81d 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tf @@ -3,12 +3,12 @@ terraform { } variable "flake_ref" { - description = "HTTP(S) or SSH Git URL of a committed flake; optional git+ prefix and ?ref= branch. URL credentials are not supported." + description = "HTTP(S) or SSH Git URL of a committed flake; optional git+ prefix and ?ref= branch. Embedded credentials appear in the output and checkout." type = string validation { - condition = can(regex("^(git\\+)?(https?|ssh)://", var.flake_ref)) && !can(regex("^(git\\+)?https?://[^/?#]*@", var.flake_ref)) && !can(regex("[[:cntrl:]]", var.flake_ref)) && !strcontains(var.flake_ref, "#") && (!strcontains(var.flake_ref, "?") || can(regex("\\?ref=[^&#?]+$", var.flake_ref))) - error_message = "flake_ref must be an http(s) or ssh Git URL with optional ?ref=, without HTTP credentials, control characters, fragments, or other query parameters." + condition = can(regex("^(git\\+)?(https?|ssh)://", var.flake_ref)) && !can(regex("[[:cntrl:]]", var.flake_ref)) && !strcontains(var.flake_ref, "#") && (!strcontains(var.flake_ref, "?") || can(regex("\\?ref=[^&#?]+$", var.flake_ref))) + error_message = "flake_ref must be an http(s) or ssh Git URL with optional ?ref=, without control characters, fragments, or other query parameters." } } diff --git a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl index 75a7f64e1..f093c50ca 100644 --- a/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl +++ b/registry/coder-labs/templates/aws-nixos/modules/nix/main.tftest.hcl @@ -29,12 +29,15 @@ run "default_branch" { } } -run "reject_http_userinfo" { +run "allow_http_userinfo" { command = plan variables { flake_ref = "git+https://user:token@example.org/flake?ref=main" } - expect_failures = [var.flake_ref] + assert { + condition = output.flake_uri == "https://user:token@example.org/flake?ref=main#coder-workspace-ec2-x86_64" + error_message = "Git URL userinfo should remain available for private clones." + } } run "reject_control_characters" { diff --git a/registry/coder-labs/templates/aws-nixos/tests/architecture.tftest.hcl b/registry/coder-labs/templates/aws-nixos/tests/architecture.tftest.hcl index a9dbc9e7c..62a947f47 100644 --- a/registry/coder-labs/templates/aws-nixos/tests/architecture.tftest.hcl +++ b/registry/coder-labs/templates/aws-nixos/tests/architecture.tftest.hcl @@ -117,3 +117,14 @@ run "reject_unsupported_query" { } expect_failures = [var.flake_ref] } + +run "allow_http_userinfo" { + command = plan + variables { + flake_ref = "https://user:token@example.org/flake?ref=main" + } + assert { + condition = module.nix.flake_uri == "https://user:token@example.org/flake?ref=main#coder-workspace-ec2-x86_64" + error_message = "Template must allow userinfo for private Git clones." + } +}