diff --git a/.github/actions/infra/k8s/apply/action.yml b/.github/actions/infra/k8s/apply/action.yml index d583a0e..bb8a454 100644 --- a/.github/actions/infra/k8s/apply/action.yml +++ b/.github/actions/infra/k8s/apply/action.yml @@ -87,4 +87,51 @@ runs: git config --global url."https://${AUTH}@github.com/".insteadOf "git@github.com:" - name: Apply config shell: bash - run: kubectl apply -k ${{ inputs.config-path }} + env: + VENDOR: ${{ inputs.vendor }} + CLUSTER_ID: ${{ inputs.cluster-id }} + AZURE_RESOURCE_GROUP: ${{ inputs.azure-resource-group }} + CONFIG_PATH: ${{ inputs.config-path }} + run: | + # Azure clusters are private: their API server only resolves inside the + # VNet, so kubectl cannot reach them from a hosted runner. Tunnel the + # command through the Azure control plane instead. + run_cmd() { + if [ "$VENDOR" != "azure" ]; then + eval "$*" + return + fi + + local result + result=`az aks command invoke \ + --name "$CLUSTER_ID" \ + --resource-group "$AZURE_RESOURCE_GROUP" \ + --file "$manifest_file" \ + --command "$*" \ + --output json` + + jq -r '.logs // ""' <<< "$result" + + # `az aks command invoke` exits 0 even when the remote command failed, + # so the remote exit code has to be checked explicitly. + local state + local code + state=`jq -r '.provisioningState // ""' <<< "$result"` + code=`jq -r '.exitCode // 1' <<< "$result"` + + if [ "$state" != "Succeeded" ]; then + echo "::error::Run command did not complete (state: $state): $*" >&2 + return 1 + fi + if [ "$code" != "0" ]; then + echo "::error::Command failed on cluster with exit code $code: $*" >&2 + return "$code" + fi + } + + # Pre-render kustomize so the manifest can be sent to the cluster + manifest_file=`mktemp --tmpdir=. --suffix .yaml` + kubectl kustomize "$CONFIG_PATH" > "$manifest_file" + + # Apply k8s config + run_cmd kubectl apply -f "$manifest_file" diff --git a/.github/actions/infra/k8s/rollout/action.yml b/.github/actions/infra/k8s/rollout/action.yml index f7d93d3..10bb3e5 100644 --- a/.github/actions/infra/k8s/rollout/action.yml +++ b/.github/actions/infra/k8s/rollout/action.yml @@ -83,31 +83,76 @@ runs: DEPLOYMENTS: ${{ inputs.rollout-deployments }} LABELS: ${{ inputs.rollout-labels }} IMAGE: ${{ inputs.rollout-image }} + NAMESPACE: ${{ inputs.rollout-namespace }} + VENDOR: ${{ inputs.vendor }} + CLUSTER_ID: ${{ inputs.cluster-id }} + AZURE_RESOURCE_GROUP: ${{ inputs.azure-resource-group }} run: | + # Azure clusters are private: their API server only resolves inside the + # VNet, so kubectl cannot reach them from a hosted runner. Tunnel the + # command through the Azure control plane instead. + run_cmd() { + if [ "$VENDOR" != "azure" ]; then + eval "$*" + return + fi + + local result + result=`az aks command invoke \ + --name "$CLUSTER_ID" \ + --resource-group "$AZURE_RESOURCE_GROUP" \ + --command "$*" \ + --output json` + + jq -r '.logs // ""' <<< "$result" + + # `az aks command invoke` exits 0 even when the remote command failed, + # so the remote exit code has to be checked explicitly. + local state + local code + state=`jq -r '.provisioningState // ""' <<< "$result"` + code=`jq -r '.exitCode // 1' <<< "$result"` + + if [ "$state" != "Succeeded" ]; then + echo "::error::Run command did not complete (state: $state): $*" >&2 + return 1 + fi + if [ "$code" != "0" ]; then + echo "::error::Command failed on cluster with exit code $code: $*" >&2 + return "$code" + fi + } + # Restart multiple deployments if [ -n "$DEPLOYMENTS" ]; then for deployment in ${DEPLOYMENTS//,/ }; do - kubectl rollout restart deployment \ + run_cmd kubectl rollout restart deployment \ "$deployment" \ - --namespace="${{ inputs.rollout-namespace }}" + --namespace="$NAMESPACE" done # Restart multiple labels elif [ -n "$LABELS" ]; then - kubectl rollout restart deployment \ + run_cmd kubectl rollout restart deployment \ --selector="$LABELS" \ - --namespace="${{ inputs.rollout-namespace }}" + --namespace="$NAMESPACE" # Restart deployments running the image + # jq stays on the runner, so the cluster only has to run plain kubectl elif [ -n "$IMAGE" ]; then - kubectl get deployment \ - --namespace="${{ inputs.rollout-namespace }}" -o json \ + deployments=`run_cmd kubectl get deployment --namespace="$NAMESPACE" -o json \ | jq -r --arg img "$IMAGE" '.items[] | select(any(.spec.template.spec.containers[]; .image | startswith($img + ":"))) | .metadata.name' \ - | xargs -r kubectl rollout restart deployment \ - --namespace="${{ inputs.rollout-namespace }}" + | tr '\n' ' '` + + if [ -n "${deployments// /}" ]; then + run_cmd kubectl rollout restart deployment $deployments \ + --namespace="$NAMESPACE" + else + echo "No deployments running image $IMAGE; nothing to restart." + fi # Restart everything else - kubectl rollout restart deployment \ - --namespace="${{ inputs.rollout-namespace }}" + run_cmd kubectl rollout restart deployment \ + --namespace="$NAMESPACE" fi diff --git a/Docs/infra-k8s-rollout.md b/Docs/infra-k8s-rollout.md index 7200a63..09b8e0f 100644 --- a/Docs/infra-k8s-rollout.md +++ b/Docs/infra-k8s-rollout.md @@ -44,6 +44,13 @@ Only provide the following for your chosen vendor. | `AZURE_SUBSCRIPTION_ID` | Azure subscription ID for login with an Azure service principal | Secret | Yes | | `AZURE_TENANT_ID` | Azure tenant ID for login with an Azure service principal | Secret | Yes | +Azure clusters are private: their API server only resolves inside the VNet, so a +hosted runner cannot reach it directly. `kubectl` is therefore tunnelled through +the Azure control plane with `az aks command invoke`. This requires the service +principal to hold `Microsoft.ContainerService/managedClusters/runcommand/action` +and `.../commandResults/read` (both included in *Azure Kubernetes Service Cluster +User Role*), and the cluster must not have `--disable-run-command` set. + #### DigitalOcean