From dc2e8f6f3f26e1ea14134634fa3a3b19ab65ecb8 Mon Sep 17 00:00:00 2001 From: rogeeoh Date: Fri, 21 Aug 2026 12:32:54 +0900 Subject: [PATCH] docs: make the design document English and keep a Korean translation DESIGN.md is now the English norm, with the Korean text kept alongside it as DESIGN.ko.md. Step headings carry explicit numbers so code comments can anchor to them, and the two files share section order for diffing. Go comments, CLAUDE.md and the AGENTS.md banner move to English as well. A CI job rejects a pull request that changes one design file without the other, so the translation cannot go stale unnoticed. --- .github/workflows/design-sync.yml | 30 ++ AGENTS.md | 2 +- CLAUDE.md | 78 +++-- DESIGN.ko.md | 224 +++++++++++++ DESIGN.md | 448 +++++++++++++++++-------- README.md | 4 +- cmd/kubectl-soft_drain/main.go | 37 +- internal/controller/envtest_test.go | 86 ++--- internal/controller/node_controller.go | 104 +++--- internal/controller/pod_controller.go | 19 +- internal/controller/softdrain.go | 38 +-- internal/controller/softdrain_test.go | 24 +- internal/controller/suite_test.go | 2 +- test/e2e/e2e_suite_test.go | 8 +- test/e2e/e2e_test.go | 142 ++++---- 15 files changed, 851 insertions(+), 395 deletions(-) create mode 100644 .github/workflows/design-sync.yml create mode 100644 DESIGN.ko.md diff --git a/.github/workflows/design-sync.yml b/.github/workflows/design-sync.yml new file mode 100644 index 0000000..f156fc0 --- /dev/null +++ b/.github/workflows/design-sync.yml @@ -0,0 +1,30 @@ +name: Design sync + +on: + pull_request: + paths: + - 'DESIGN.md' + - 'DESIGN.ko.md' + +jobs: + translation: + name: DESIGN.md and DESIGN.ko.md move together + runs-on: ubuntu-latest + steps: + - name: Clone the code + uses: actions/checkout@v4 + with: + fetch-depth: 0 + + # DESIGN.md is normative and DESIGN.ko.md is its translation. Letting one + # move without the other is how the translation silently goes stale. + - name: Both files must change together + run: | + changed=$(git diff --name-only ${{ github.event.pull_request.base.sha }}...HEAD) + en=$(printf '%s\n' "$changed" | grep -cx 'DESIGN.md' || true) + ko=$(printf '%s\n' "$changed" | grep -cx 'DESIGN.ko.md' || true) + if [ "$en" != "$ko" ]; then + echo "DESIGN.md=$en DESIGN.ko.md=$ko" + echo "A design change updates both files in the same PR." + exit 1 + fi diff --git a/AGENTS.md b/AGENTS.md index d34bbc6..ff01d66 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,6 +1,6 @@ # soft-drain - AI Agent Guide -> **이 문서는 kubebuilder가 생성한 범용 안내서다. 설계 규범은 DESIGN.md, 작업 규칙은 CLAUDE.md이며 충돌 시 그쪽이 우선한다. 이 프로젝트는 CRD를 만들지 않는다 — `api/`, `config/crd` 관련 내용은 해당 없다.** +> **This is the generic guide kubebuilder generated. The design norm is DESIGN.md and the working rules are CLAUDE.md; where they conflict, those win. This project creates no CRDs — anything about `api/` or `config/crd` does not apply.** ## Project Structure diff --git a/CLAUDE.md b/CLAUDE.md index 82cbe40..8b7affd 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1,39 +1,69 @@ -# soft-drain 작업 규칙 +# soft-drain working rules -이 문서는 작업 규칙만 담는다. 설계는 전부 DESIGN.md에 있다. +This file holds working rules only. The design lives entirely in DESIGN.md. ## SSOT -- **DESIGN.md가 유일한 설계 규범이다.** 코드와 다르면 코드가 틀린 것이다. 설계 자체가 틀렸다고 판단되면 DESIGN.md를 고치는 게 아니라 사용자에게 보고한다. DESIGN.md 수정은 사용자 승인이 있을 때만 한다. -- AGENTS.md는 kubebuilder가 생성한 범용 안내서다. DESIGN.md나 이 문서와 충돌하면 무시한다. 특히 CRD/api 디렉토리 관련 내용은 이 프로젝트에 해당 없다 — `api/`와 `config/crd`가 없는 것이 의도된 상태다. +- **DESIGN.md is the only design norm.** If the code disagrees with it, the code is wrong. If + the design itself looks wrong, report it to the user rather than editing DESIGN.md. DESIGN.md + is modified only with the user's approval. +- **DESIGN.ko.md is a translation, not a norm.** A design change touches both files in the same + commit, and CI rejects a PR that moves one without the other. Where they disagree, DESIGN.md + is right. Keep the section order and numbering identical so the two can be diffed against + each other. +- AGENTS.md is the generic guide kubebuilder generated. Ignore it wherever it conflicts with + DESIGN.md or this file. Its CRD and `api/` directory material does not apply here — the + absence of `api/` and `config/crd` is intended. -## 금지 +## Prohibited -- **CRD를 만들지 않는다. 제안도 하지 않는다.** 노드 라벨이 유일한 API다. -- **문서화된 K8s 동작을 클러스터에서 재검증하지 않는다.** ReplicaSet adoption, pod-deletion-cost 정렬 등은 이미 검증됐다. -- **실 클러스터에 접근하지 않는다.** 테스트는 kind 클러스터(`soft-drain-test-e2e`)만 쓴다. 사용자의 kubeconfig 컨텍스트를 실 클러스터로 바꾸지 않는다. -- 리뷰 에이전트는 **읽기 전용**이다. 코드를 고치지 말고 발견만 보고한다. 확신이 없으면 스스로 결정하지 말고 보고서에 올린다. +- **No CRDs, and do not propose one.** Node labels are the only API. +- **Do not re-verify documented Kubernetes behaviour against a cluster.** ReplicaSet adoption, + pod-deletion-cost ordering and the like are settled. +- **Do not touch a real cluster.** Tests use the kind cluster (`soft-drain-test-e2e`) only. + Never point the user's kubeconfig context at a real cluster. +- Review agents are **read-only**. Report findings, do not fix code. When unsure, put it in the + report instead of deciding alone. -## 테스트 3층 +## The three test layers -| 층 | 명령 | 검증 대상 | +| Layer | Command | What it verifies | |---|---|---| -| 유닛 | `go test ./internal/...` | 순수 판정 함수. 클러스터 없음 | -| envtest | `make test` | 우리 컨트롤러가 API 서버에 **쓰는 것** | -| e2e | `make test-e2e` | kind 멀티노드에서 전체 루프의 수렴 | +| unit | `go test ./internal/...` | pure decision functions, no cluster | +| envtest | `make test` | what our controller **writes** to the API server | +| e2e | `make test-e2e` | the whole loop converging on a multi-node kind cluster | -envtest에는 kube-controller-manager와 scheduler가 없다. ReplicaSet 입양·삭제·스케줄링은 일어나지 않으므로 거기서 검증하려 들지 않는다. 그건 e2e에서만 보이고, e2e에서도 RS 동작 자체가 아니라 우리 컨트롤러의 결과(Pod이 옮겨지고 Complete가 붙는가)를 본다. +envtest has no kube-controller-manager and no scheduler. ReplicaSet adoption, deletion and +scheduling do not happen there, so do not try to verify them there. They are visible in e2e +only, and even there what is checked is our controller's outcome — the Pod moved, Complete was +set — not ReplicaSet behaviour itself. -타이밍 경합도 테스트한다. 다만 층이 다르다 — 경합의 교차 순서를 밖에서 제어할 수 없는 e2e에서 경합을 "발생"시키려 하면 안 일어난 채 통과(거짓 안심)하거나 가끔 실패(flaky)한다. 경합은 envtest·유닛에서 교차를 손으로 배열해 결정론적으로 검증하고(예: 판정↔삭제 사이의 넘기기, 넘긴 뒤 타깃 생존), e2e는 결정론적으로 유도되는 시나리오를 담는다: 정상 경로, 다중 Deployment 무중단 감시, 취소 2종(라벨 제거·uncordon), 자원 부족 Pending, 롤아웃 겹침 3종(stale 대체 조기 회수 포함), Deployment 삭제, 스케일업·스케일 0, 전 워커 drain 교착·해소, 선점 cordon 소유권, 비대상 Pod 불가침, 연쇄 재-drain, 착지 노드 drain 시 조기 회수, tolerate 착지-삭제 반복, 컨트롤러 자기 자신 drain, kubectl 플러그인 경로, Complete 후 uncordon 복귀. +Timing races are tested too, but at a different layer. e2e cannot control the interleaving from +outside, so trying to *produce* a race there either passes because it never happened (false +comfort) or fails now and then (flaky). Races are arranged by hand and verified +deterministically in envtest and unit tests — a hand-over between the decision and the +deletion, a target surviving a hand-over. e2e carries the scenarios that can be induced +deterministically: the happy path, uninterrupted watching of several Deployments, both +cancellations (label removal, uncordon), Pending on insufficient resources, three +rollout-overlap cases (including early reclamation of a stale replacement), Deployment +deletion, scale-up and scale-to-zero, the all-workers-drained deadlock and its release, +preemptive cordon ownership, non-target Pods left untouched, cascading re-drain, early +reclamation when the landing node is drained, the tolerate land-and-delete loop, the controller +draining itself, the kubectl plugin path, and the return to uncordon after Complete. -## 리뷰 절차 +## Review procedure -- 구현은 메인 세션이 단독으로 한다. 리뷰 에이전트는 체크포인트(컨트롤러 하나 완성, 커밋 직전)마다 새로 소집한다. -- 리뷰어 소집 프롬프트: "CLAUDE.md 규칙 하에 DESIGN.md 대비 이 코드를 리뷰하라." -- 발견은 REVIEW.md에 쌓고 사용자와 **하나씩** 검토한 뒤 반영한다. 리뷰 라운드는 체크포인트당 최대 2회. +- Implementation is done by the main session alone. Review agents are convened fresh at each + checkpoint: one controller finished, or just before a commit. +- Reviewer prompt: "Review this code against DESIGN.md under the rules in CLAUDE.md." +- Findings accumulate in REVIEW.md and are gone through with the user **one at a time** before + being applied. At most two review rounds per checkpoint. -## 문서·코드 스타일 +## Documentation and code style -- 한국어 문서는 장황하게 쓰지 않고, 번역체를 쓰지 않고, 히스토리("과거에 이랬지만")를 서술하지 않는다. -- 커밋 메시지와 테스트 스펙명 등 CI에 드러나는 텍스트는 영어로 쓴다. -- 코드 주석은 코드가 보여줄 수 없는 제약만 적는다. 로그는 K8s 컨벤션(대문자 시작, 마침표 없음, 과거형)을 따른다. +- Documents are not verbose, do not read like translations, and do not narrate history ("this + used to be…"). That applies to DESIGN.ko.md as much as to the English files: it is written as + Korean, not as a word-for-word rendering. +- Everything that surfaces in CI — commit messages, test spec names — is in English. +- Code comments record only the constraints the code cannot show. Logs follow the Kubernetes + convention: leading capital, no trailing period, past tense. diff --git a/DESIGN.ko.md b/DESIGN.ko.md new file mode 100644 index 0000000..01c0c25 --- /dev/null +++ b/DESIGN.ko.md @@ -0,0 +1,224 @@ +# soft-drain + +> 이 문서는 [DESIGN.md](DESIGN.md)의 번역본이다. 규범은 영어본이고, 둘이 어긋나면 영어본이 맞다. + +## 왜 만드는가 + +노드를 빼려면 `kubectl drain`을 쓰는데, replicas가 1인 워크로드는 Pod이 먼저 죽고 새로 뜨는 동안 장애가 난다. 2벌을 띄워 HA를 만들자니 비용이 두 배고, PDB를 걸면 drain이 429로 튕길 뿐 Pod이 옮겨지지는 않는다. + +soft-drain은 **새 Pod을 먼저 띄우고 Ready가 된 다음 옛 Pod을 없앤다.** 용량이 비는 순간이 없다. + +## 원리 + +ReplicaSet에는 두 가지 성질이 있다. + +1. selector에 맞고 주인 없는 Pod을 보면 자기 자식으로 데려간다. +2. 자식이 replicas보다 많으면 하나를 지우는데, `pod-deletion-cost`가 낮은 쪽을 먼저 지운다. + +soft-drain은 이 둘을 이어 붙인다. 옮길 Pod과 똑같은 Pod을 하나 더 만들되 `pod-template-hash` 라벨은 빼둔다. 그러면 selector에 안 걸려서 ReplicaSet이 데려가지 않는다. 그 Pod이 Ready가 되면 hash를 붙인다. ReplicaSet이 데려가면서 자식이 하나 늘고, 늘어난 만큼 하나를 지운다. `pod-deletion-cost`가 음수인 옛 Pod이 지워진다. + +**Pod을 옮기는 일은 ReplicaSet이 한다. soft-drain은 재료만 놓아준다.** + +`pod-template-hash`를 처음부터 붙이면 안 된다. ReplicaSet이 만들자마자 데려가는데 그때 새 Pod은 아직 Pending이고, 삭제 순서에서 Pending과 NotReady가 `pod-deletion-cost`보다 앞이라 방금 만든 Pod이 먼저 죽는다. + +`pod-template-hash`는 ReplicaSet selector에는 들어가고 Service selector에는 안 들어간다. 그래서 대체 Pod은 Ready가 되는 즉시 Endpoints에 올라가고 입양 전후로 빠지지 않는다. 엔드포인트 갱신이 Pod당 한 번뿐이다. + +**보장하는 것은 노출이지 어느 Pod이 지워지는가가 아니다.** `pod-deletion-cost`는 삭제 정렬의 4순위 힌트다. 넘기는 순간 무관한 Pod이 NotReady면 ReplicaSet은 그쪽을 지우고 우리 타깃은 남는다. 그래도 Ready인 Pod을 먼저 늘린 다음 초과분이 지워지므로 노출은 `N` 밑으로 내려가지 않고, 남은 타깃은 다음 라운드에 다시 시도된다. + +## 쓰는 법 + +``` +kubectl label node node-01 soft-drain.com/drain=true # 시작 +kubectl get nodes -l soft-drain.com/state=Complete # 완료 확인 +kubectl label node node-01 soft-drain.com/drain- # 취소 +kubectl uncordon node-01 # 이것도 취소다 (state=Cancelled 로 남는다) +``` + +끝난 노드는 cordon된 채로 남는다. 그 다음에 drain을 하든 노드를 리부팅하든 soft-drain이 상관할 일이 아니다. 정비가 끝나 drain 라벨을 걷으면 우리가 걸었던 cordon도 함께 걷힌다 — 라벨 제거가 곧 노드 반환이다. 사람이 미리 걸어둔 cordon이면 그대로 둔다. + +### kubectl 플러그인 + +`kubectl soft-drain`은 위 네 줄의 포장이다. **쓰는 것은 drain 라벨 하나뿐이고 나머지는 읽기다** — 서버 쪽 표면은 늘지 않는다. 문법은 `git stash`형이다 — 맨몸+노드가 주 동작이고, 나머지는 서브커맨드다. `status`, `release`, `version`은 예약어다. + +``` +kubectl soft-drain node-01 [node-02 ...] # 라벨을 붙이고 전부 Complete될 때까지 진행을 보여준다 (--wait=false, --timeout) +kubectl soft-drain status # 현황판: 관여 중인 노드·남은 타깃·대체 Pod +kubectl soft-drain status node-01 -o json # 특정 노드, 기계용 (json|yaml) +kubectl soft-drain release node-01 [...] # 라벨을 걷고 복원을 기다린다 +``` + +release는 진행 중이면 취소가 되고 Complete면 관리 종료가 된다 — 실체는 같은 라벨 제거고, 결말도 같다: 우리가 남긴 것을 전부 걷는다(우리가 걸었던 cordon 포함). 그래서 release는 정비가 끝난 뒤의 동사다 — 리부팅 전에 하면 비워 둔 노드가 도로 열린다. `kubectl uncordon`도 취소지만 라벨과 Cancelled 래치가 남는 점이 다르다 — release는 전부 걷는다. `--timeout`이 터지면 Pending 대체 Pod과 스케줄러 메시지를 보여주고 0이 아닌 코드로 나간다 — "막혔을 때 보는 법"의 자동화다. 중간에 끊어도 라벨은 남으므로 drain은 계속된다. `state=Cancelled`인 노드에 다시 drain을 걸면 라벨을 걷어 복원시킨 뒤 다시 붙인다. + +현황판에 "언제부터"는 없다 — 컨트롤러가 무기억이라 시작 시각을 어디에도 기록하지 않는다. `-o`의 몫은 플러그인만 계산할 수 있는 집계(타깃·대체 Pod 상태·스케줄러 메시지)다. 노드명 목록이 필요한 기계는 라벨 조회가 정석이다: `kubectl get nodes -l soft-drain.com/state=Complete -o name`. + +`kubectl drain`의 `--ignore-daemonsets`, `--delete-emptydir-data`, `--force`는 없다. eviction 전제의 개념이라 여기 해당이 없다. + +## 컨트롤러가 하는 일 + +``` +1. drain 라벨이 붙은 노드를 cordon한다 +2. 그 노드의 Deployment Pod(타깃)에 pod-deletion-cost 를 음수로 박는다 +3. 타깃마다 대체 Pod이 하나씩 있도록 맞춘다 +4. 대체 Pod이 Ready가 되면 pod-template-hash 를 붙여 ReplicaSet이 데려가게 한다 +5. 타깃이 없어질 때까지 반복한다 +6. 타깃이 없으면 완료 표시를 한다 +``` + +**어떤 상태도 기억하지 않는다.** 매번 클러스터를 다시 보고 전부 다시 판정하므로 중간에 죽어도 다음 라운드가 이어서 한다. + +**읽기는 API 서버에서 직접 한다.** watch는 다시 볼 때를 알려주는 데만 쓴다. 캐시가 뒤처진 상태를 설계가 감당하기 시작하면 조건이 급격히 복잡해진다. 부하가 문제가 되면 그때 캐시를 붙인다. + +### 1. 노드 마킹 + +`drain` 라벨이 있으면 cordon한다. 우리가 실제로 값을 바꿨을 때만 `cordoned-by-controller` 어노테이션을 단다. 사람이 미리 걸어둔 cordon을 나중에 우리가 푸는 일을 막기 위해서다. + +**cordon은 준비 작업이 아니라 이 반복이 끝나는 이유다.** cordon을 걸어두면 그 노드에는 Pod이 새로 안 뜬다. 빼야 할 Pod이 늘어날 일이 없으니 하나씩 빼다 보면 언젠가 바닥이 난다. 예외는 `unschedulable`을 tolerate하는 워크로드뿐인데, 그건 빼도 그 자리에 다시 앉아서 줄지를 않는다. 그래서 거기서 멈춘다(3번). + +`drain` 라벨이 사라지면 되돌린다 — 우리 값이 박힌 `pod-deletion-cost`를 걷고, `cordoned-by-controller`가 있으면 uncordon하고, `state` 라벨을 지운다. + +**누가 uncordon하면 관여를 접는다 — 진행 중이든 끝난 뒤든.** `state`가 `InProgress`나 `Complete`인데 노드가 `unschedulable`이 아니면 그렇게 된 것이다 — 두 상태 모두 cordon을 확인한 뒤에만 붙기 때문에 이 조합이 곧 증거다. 진행 중이라면 cordon은 종료를 보장하던 전제라서 전제가 사라진 채 계속할 수 없고, 끝난 뒤라면 사람이 우리 cordon을 풀고 노드를 다시 쓰기로 결정한 것이다. 어느 쪽이든 다시 cordon해서 사람과 싸우지 않는다. cost를 걷고 `cordoned-by-controller`를 지우고 `state=Cancelled`를 붙인 뒤 손을 뗀다. 어노테이션을 지우는 이유: 기록된 cordon은 사람 손에 이미 풀렸으므로, 이후 사람이 새로 건 cordon을 라벨 제거 시점의 복원이 우리 것으로 오인해 풀면 안 된다. 대체 Pod은 회수 경로가 걷는다. `Cancelled`는 래치다 — 라벨을 걷으면 지워지고, 다시 하려면 라벨을 걷었다가 다시 붙인다. + +`Complete`를 접지 않고 두면 그 노드가 삭제 자석이 된다. 방금 비워져 가장 한가한 노드가 열렸으니 스케줄러는 다른 drain의 대체 Pod을 정확히 거기 앉히고, 착지 검사는 앉는 족족 지운다. 클러스터가 작을수록 모든 drain이 그 노드로 빨려 들어간다. + +### 2. 타깃 표시 + +타깃은 **그 노드 위에서 owner가 ReplicaSet이고 그 ReplicaSet의 owner가 Deployment인 Pod**이다. phase가 `Failed`나 `Succeeded`인 Pod은 빼고 센다 — ReplicaSet도 active로 세지 않는 Pod이라 대체는 이미 딴 곳에 만들어져 있고, 노드에 남은 시체가 완료 판정만 막는다. + +`controller.kubernetes.io/pod-deletion-cost = -2147483648`을 쓴다. 타깃에만 쓰고, 시점은 대체 Pod을 만들기 전이다 — 넘기기까지 미룰 이유가 없고, 그 사이 무관한 스케일다운이 나도 드레인 대상이 먼저 죽는 쪽이 낫다. + +대체 Pod에는 쓰지 않는다. 입양 전에는 ReplicaSet이 쳐다보지 않는 Pod이라 값이 무의미하고, 입양 후에 남으면 그 Pod이 다음 스케일다운마다 1순위로 죽는다. 타깃은 지워지면서 값도 같이 사라지므로 걷을 것이 없다. + +원래 값이 있었어도 덮어쓰고 복원하지 않는다. 되돌릴 때는 값이 정확히 `-2147483648`인 것만 지운다. 그 값을 쓰는 게 우리뿐이라 값이 이것이면 우리가 붙인 것이다. + +### 3. 대체 Pod 맞추기 + +만들기만 하는 단계가 아니다. **있어야 할 집합과 있는 집합을 맞춘다.** + +``` +있어야 할 것 = deletionTimestamp 가 없는 타깃마다 하나 +있는 것 = soft-drain.com/replaces = <타깃 UID> 이면서 + controller ownerRef 없고 + phase 가 Failed / Succeeded 가 아니고 + deletionTimestamp 도 없는 Pod + +모자라면 만들고, 남으면 지운다 +``` + +**타깃이 사라지면 대체 Pod도 사라진다.** 취소, 롤아웃으로 인한 ReplicaSet prune, Deployment 삭제, `replicas` 축소, 타깃의 eviction이 전부 이 한 줄에 걸린다. 따로 처리할 것이 없다. + +**terminating 타깃은 만들기에서 뺀다.** ReplicaSet은 `deletionTimestamp`가 찍힌 Pod을 active에서 빼므로 이미 스스로 대체를 만들고 있고, 노드가 cordon이라 그 Pod은 다른 노드에 뜬다. 자리는 우리가 아무것도 안 해도 비워진다. + +**죽은 대체 Pod은 있는 것으로 세지 않고, 지운다.** 노드 압력 eviction이나 kubelet admission 거부로 `Failed`가 된 Pod은 Ready가 될 수도 입양될 수도 없다. 살아 있는 것으로 세면 그 타깃이 영원히 멈춘다. Pod에는 재시작이 없어서(phase `Failed`는 터미널이고 `restartPolicy`는 컨테이너 얘기다) 복구는 새 Pod뿐인데, 세지 않고 지우지도 않으면 만들 때마다 시체가 쌓인다. 원인이 지속되면 만들고-죽고-지우기를 반복하다가 원인이 풀리는 순간 수렴한다. + +**drain 중인 노드에 앉은 대체 Pod도 세지 않고, 지운다.** cordon보다 스케줄이 먼저여서, 앉은 뒤에 그 노드에 drain이 걸리는 경우가 생긴다. Ready를 기다릴 이유가 없다 — Ready가 되어도 넘기면 비우려는 노드에 타깃이 하나 더 생길 뿐이라 결말은 삭제뿐이고, 그동안 자리만 먹는다. hash가 없어 어느 ReplicaSet의 자식도 아니므로 지워도 노출은 줄지 않고, 같은 라운드가 새로 만들면 스케줄러가 cordon을 피해 앉힌다. 지울 때 Warning Event를 남긴다. + +`node.kubernetes.io/unschedulable`을 tolerate하는 워크로드는 새로 만든 Pod이 또 drain 노드에 앉을 수 있고, 그러면 만들고 지우기를 반복한다. cordon을 무시하도록 만든 워크로드이므로 우리가 `nodeAffinity`를 주입해 그 의도를 뒤집지 않는다. 반복되는 Warning Event가 옮길 수 없다는 사실을 보여준다. + +**타깃의 ReplicaSet이 Deployment의 현재 템플릿이 아니면 만들지 않고, 있으면 지운다.** 롤아웃이 그 타깃을 이미 대체하는 중이다 — 새 버전을 다른 노드에 올리고 Ready 후 타깃을 지우는, 우리가 하려던 일 그대로다. 우리 대체 Pod은 낙선한 버전을 짓는 잉여이고, Healthy(D)가 롤아웃 내내 넘기기를 막으므로 입양에 도달할 수도 없다. `pod-deletion-cost`는 타깃에 남아 있으므로 old ReplicaSet의 스케일다운이 drain 노드의 타깃부터 지운다. 지울 때 Normal Event를 남긴다. 판정은 Deployment의 `spec.template`과 타깃 ReplicaSet의 템플릿을 `pod-template-hash` 라벨만 빼고 비교한다(Deployment 컨트롤러의 EqualIgnoreHash와 같다) — 이미지 변경이든 `rollout restart`든, 템플릿이 바뀌는 모든 경우가 같은 길로 잡힌다. 단 `spec.paused`면 템플릿이 달라도 이 규칙을 적용하지 않는다 — 롤아웃이 실제로 움직이지 않아 "대체 중"이라는 전제가 깨진다. 대체는 평소처럼 유지되고 넘기기만 Healthy(D)에 걸려 미뤄지다가, 재개되는 순간 이 규칙이 잡는다. + +**두 삭제가 preconditions에 거부되면 라운드를 접는다.** 거부는 판정과 삭제 사이에 Pod이 변했다는 뜻이다. 낡은 명단으로 넘기기까지 진행하지 않고 짧은 requeue로 라운드를 끝내, 다음 라운드가 새 상태에서 처음부터 판정한다. 그래서 넘기기는 "drain 중인 노드에 앉은 대체"를 다시 검사할 필요가 없다. + +**이미 넘긴 것도 세지 않는다.** 넘기면서 `soft-drain.com/replaces` 라벨을 떼기 때문에 애초에 후보가 아니다. 그래서 넘겼는데 타깃이 살아남은 경우가 자연히 복구된다 — 넘기는 순간 `replicas`가 올라가면 초과분이 증설분에 흡수되어 아무것도 안 지워지는데, 다음 라운드가 "타깃은 그대로인데 대신할 Pod이 없다"를 보고 하나 더 만든다. + +**만드는 쪽과 지우는 쪽 양쪽에서 깨어나야 한다.** 노드에서 출발하는 순회만 있으면, ReplicaSet이 prune될 때 타깃 Pod도 같이 사라져서 순회할 대상이 없어지고 대체 Pod을 쳐다볼 일이 없어진다. 그래서 대체 Pod 자체를 키로 하는 경로가 따로 있어야 한다. 판정은 위 한 줄로 같다. + +대체 Pod은 이렇게 만든다. + +```yaml +metadata: + generateName: aaa-5449d4d8c8- # 타깃의 ReplicaSet 이름 + "-" + labels: + app: aaa # rs.spec.template.metadata.labels 에서 + soft-drain.com/replaces: 3f2a... # 타깃 Pod의 UID + # pod-template-hash 는 뺀다 +spec: +``` + +스펙은 **살아 있는 Pod이 아니라 `rs.spec.template`에서** 가져온다. 살아 있는 Pod을 베끼면 `nodeName`이 따라오고, webhook이 이미 넣어둔 사이드카 위에 하나가 더 들어간다. + +`rs.spec.template.metadata.labels`에는 `pod-template-hash`가 **이미 들어 있다.** 복사한 뒤 명시적으로 제거한다. 이 문서에서 가장 중요한 한 줄이다. + +생성이 거부되면 Warning Event를 낸다. ResourceQuota 초과나 admission webhook 거부가 여기 걸리는데, 이 경우만 Pod 오브젝트가 안 생겨서 밖에서 볼 흔적이 없다. 거부돼도 멈추지 않고 다음 라운드에 다시 시도한다. + +### 4. 넘기기 + +대체 Pod이 Ready가 되면 patch 하나로 `pod-template-hash`를 붙이고 `soft-drain.com/replaces`를 뗀다. + +붙일 hash는 **타깃 Pod의 ownerRef가 가리키는 ReplicaSet**에서 읽는다. Deployment를 거쳐 현재 ReplicaSet을 찾는 경로는 쓰지 않는다 — 대체 Pod은 타깃의 ReplicaSet 템플릿으로 만들어졌고, 롤아웃 중이면 그게 현재 ReplicaSet이 아닐 수 있다. + +넘기기 전에 하나를 본다. drain 중인 노드에 앉은 대체는 3단계가 이미 지웠으므로 여기 오지 않는다. + +**사용자 Deployment가 Healthy한가.** + +``` +Healthy(D) ≡ D.status.observedGeneration >= D.metadata.generation + ∧ D.status.replicas == D.status.updatedReplicas + ∧ D.status.availableReplicas >= D.spec.replicas +``` + +Healthy가 아니면 미룬다. 사용자가 `N` 미만이면 넘겨도 초과분이 없어 아무것도 지워지지 않고, 롤아웃 중이면 넘겨받을 ReplicaSet이 하나로 정해지지 않아 노출이 rollout 설정보다 더 내려갈 수 있다. + +`replicas == updatedReplicas` 항이 "Pod을 가진 ReplicaSet이 하나뿐"을 판정한다. 나머지 두 항만으로는 롤아웃을 못 잡는다 — `maxUnavailable: 0`이면 `availableReplicas >= N`이 롤아웃 내내 유지되는데, 무중단을 원하는 사용자가 정확히 그 설정을 쓴다. `spec.paused`도 이 항에 걸린다. + +판정은 Deployment마다 따로 하고 준비된 것부터 넘긴다. 묶으면 제일 느린 하나가 나머지를 인질로 잡는다. + +### 5. 완료 + +노드 위에 타깃이 하나도 없으면 `state=Complete`를 붙인 뒤 Event를 낸다. `cordoned-by-controller` 어노테이션은 그대로 둔다 — cordon은 여전히 우리가 건 것이고, drain 라벨이 걷힐 때 함께 걷힌다. `Complete`는 래치다 — cordon이 유지되는 동안에는 drain 라벨이 걷힐 때까지 관여하지 않는다. cordon된 노드에 새로 앉을 수 있는 건 `unschedulable`을 tolerate하는 Pod뿐인데, 그건 어차피 옮기지 못하는 부류다. 래치가 없으면 리부팅을 기다리는 노드가 도로 열려 Pod이 몰린 채로 리부팅하게 된다 — 노드를 여는 순간은 사람이 라벨을 걷을 때(반환)와 uncordon할 때(취소)뿐이어야 한다. + +사람이 uncordon하면 래치는 `Cancelled`로 접힌다(1번). 착지 금지도 함께 풀린다 — 리부팅하러 갈 노드라서 막았던 것인데, uncordon은 리부팅 안 간다는 선언이다. uncordon과 감지 사이의 짧은 창에서는 착지한 대체 Pod이 지워질 수 있지만, 다음 라운드가 새로 만든다. + +**완료 판정에는 terminating 타깃도 센다.** `deletionTimestamp`가 찍혀도 grace period 동안 계속 돈다. 여기서 빼면 아직 작업이 돌고 있는 노드에 `Complete`가 붙고, 그걸 보고 노드를 리부팅한 사람이 그 작업을 죽인다. 만들기에서는 빼고 완료 판정에서는 세는 이유가 이것이다. + +## 메타데이터 + +| 대상 | 키 | 값 | 쓰는 쪽 | +|---|---|---|---| +| 노드 | `soft-drain.com/drain` (라벨) | `"true"` | 사람 | +| 노드 | `soft-drain.com/state` (라벨) | `InProgress` / `Complete` / `Cancelled` | 컨트롤러 | +| 노드 | `soft-drain.com/cordoned-by-controller` (어노테이션) | `"true"` | 컨트롤러 | +| 타깃 Pod | `controller.kubernetes.io/pod-deletion-cost` (어노테이션) | `-2147483648` | 컨트롤러 | +| 대체 Pod | `soft-drain.com/replaces` (라벨) | 타깃 Pod의 UID | 컨트롤러 | + +`soft-drain.com/replaces`를 쓰는 것은 우리뿐이다. 이 라벨이 없는 Pod은 만들지도 지우지도 않는다. + +## 지켜야 할 것 + +1. 대체 Pod은 `pod-template-hash` 없이 만든다. +2. 대체 Pod의 스펙은 살아 있는 Pod이 아니라 `rs.spec.template`에서 가져온다. +3. `pod-deletion-cost`를 먼저 쓰고 `pod-template-hash`를 나중에 붙인다. +4. `soft-drain.com/replaces` 라벨이 있는 Pod만 지운다. +5. controller ownerRef가 있는 Pod은 지우지 않는다. +6. 대체 Pod을 지울 때는 읽었던 UID와 resourceVersion을 preconditions로 건다. 판정과 삭제 사이에 hash가 붙어 ReplicaSet이 데려간 Pod이면 삭제가 거부되고, 다음 라운드가 다시 판정한다. + +## 안 하는 것 + +- Deployment 소속 Pod만 옮긴다. StatefulSet, DaemonSet, Job, 직접 만든 Pod은 그대로 둔다. **`Complete`는 "내 몫이 끝났다"이지 "노드가 비었다"가 아니다.** +- 노드 위 대상 Pod을 한꺼번에 옮긴다. 자원이 모자라면 Pending으로 기다린다. +- 옮길 수 없는 워크로드를 미리 걸러내지 않는다. Pending으로 남고 사람이 보면 된다. +- PDB를 조회하지 않는다. 지우는 주체가 사용자 ReplicaSet이라 eviction API를 타지 않는다. +- 사용자 Deployment의 `spec`을 수정하지 않는다. 사용자 Pod에는 어노테이션 하나만 쓴다. +- 노드를 drain하거나 끄지 않는다. + +## 막혔을 때 보는 법 + +노드가 `InProgress`에서 안 움직이면 대체 Pod을 본다. + +```bash +kubectl get pods -A -l soft-drain.com/replaces +kubectl describe pod +``` + +스케줄러가 `PodScheduled=False`의 message에 이유를 그대로 써 둔다 — `0/12 nodes are available: 5 Insufficient cpu, 7 node(s) didn't match pod anti-affinity rules` 같은 식이다. 컨트롤러가 따로 진단을 만들지 않는 이유다. + +대체 Pod이 하나도 안 보이면 생성이 거부됐거나(ResourceQuota, admission webhook), 롤아웃이 이주를 대신 수행 중이라 만들지 않는 것이다. 어느 쪽이든 `kubectl describe node <노드>`의 Event에 남는다. + +## 알려진 한계 + +- **여유 자원이 없으면 진행하지 못한다.** 자리는 옛 Pod이 죽어야 나고 옛 Pod은 새 Pod이 Ready여야 죽으므로, 여유가 0이면 스스로 풀리지 않는다. RWO 볼륨과 로컬 PV도 같은 구조다. +- **배치 규칙이 한 자리를 못 내주면 자원이 남아돌아도 진행하지 못한다.** 노드당 하나로 제한하는 required `podAntiAffinity`를 걸어두고 후보 노드를 전부 채운 경우가 대표적이다. 이건 soft-drain만의 제약이 아니라 "먼저 띄우고 나중에 지운다"는 방식 전체의 산술이다 — 같은 워크로드는 `maxSurge: 1` 롤아웃도 똑같이 막힌다. 그래서 그런 사용자는 이미 `maxUnavailable: 1`로 운영하며 롤아웃마다 `N` 밑으로 내려가는 것을 감수하고 있다. 노드를 뺄 때도 `kubectl drain`을 쓰면 된다. +- **`node.kubernetes.io/unschedulable`을 tolerate하는 워크로드는 옮기지 못할 수 있다.** 대체 Pod이 drain 노드에 앉을 때마다 지우고 다시 만들기를 반복하고, 다른 노드에 앉는 운이 따라야 끝난다. +- **입양 전 대체 Pod은 controller가 없어 PDB 집계를 흔든다.** 같은 라벨을 갖고 Ready라 `currentHealthy`에는 들어가는데 `expectedCount`에는 안 들어가서, 그동안 `disruptionsAllowed`가 1 늘어난다. PDB가 지키는 바닥 아래로 내려가지는 않는다. 같은 이유로 사용자 PDB에 `UnmanagedPods` Warning이 쌓인다. +- **주인 없는 대체 Pod이 있는 노드는 Cluster Autoscaler가 축소하지 못한다.** 넘기기가 멈춘 상태로 오래 가면 그 노드가 컨솔리데이션에서 계속 빠진다. +- **사용자 Pod의 `pod-deletion-cost` 원래 값은 복원하지 않는다.** +- **`pod-deletion-cost`가 필요하므로 Kubernetes 1.22 이상이어야 한다.** diff --git a/DESIGN.md b/DESIGN.md index dcf0288..59d1170 100644 --- a/DESIGN.md +++ b/DESIGN.md @@ -1,152 +1,277 @@ # soft-drain -## 왜 만드는가 +> Korean translation: [DESIGN.ko.md](DESIGN.ko.md). This file is normative. -노드를 빼려면 `kubectl drain`을 쓰는데, replicas가 1인 워크로드는 Pod이 먼저 죽고 새로 뜨는 동안 장애가 난다. 2벌을 띄워 HA를 만들자니 비용이 두 배고, PDB를 걸면 drain이 429로 튕길 뿐 Pod이 옮겨지지는 않는다. +## Why -soft-drain은 **새 Pod을 먼저 띄우고 Ready가 된 다음 옛 Pod을 없앤다.** 용량이 비는 순간이 없다. +Taking a node out means `kubectl drain`, and a workload with one replica goes down while it +runs: the Pod is killed first, and the new one only starts after. Running a second copy for +HA doubles the bill, and a PDB only makes the drain bounce with 429s — the Pod still does not +move. -## 원리 +soft-drain **brings the new Pod up, waits for it to be Ready, and only then lets the old one +go.** There is no moment where capacity is missing. -ReplicaSet에는 두 가지 성질이 있다. +## How it works -1. selector에 맞고 주인 없는 Pod을 보면 자기 자식으로 데려간다. -2. 자식이 replicas보다 많으면 하나를 지우는데, `pod-deletion-cost`가 낮은 쪽을 먼저 지운다. +A ReplicaSet has two properties. -soft-drain은 이 둘을 이어 붙인다. 옮길 Pod과 똑같은 Pod을 하나 더 만들되 `pod-template-hash` 라벨은 빼둔다. 그러면 selector에 안 걸려서 ReplicaSet이 데려가지 않는다. 그 Pod이 Ready가 되면 hash를 붙인다. ReplicaSet이 데려가면서 자식이 하나 늘고, 늘어난 만큼 하나를 지운다. `pod-deletion-cost`가 음수인 옛 Pod이 지워진다. +1. It adopts an ownerless Pod that matches its selector. +2. When it has more children than `replicas`, it deletes one — the lowest `pod-deletion-cost` + first. -**Pod을 옮기는 일은 ReplicaSet이 한다. soft-drain은 재료만 놓아준다.** +soft-drain chains the two. It creates a second Pod identical to the one to be moved, but +without the `pod-template-hash` label. Without the hash the Pod does not match the selector, +so the ReplicaSet leaves it alone. Once that Pod is Ready, the hash is added. The ReplicaSet +adopts it, is now one child over `replicas`, and deletes one: the old Pod, whose +`pod-deletion-cost` is negative. -`pod-template-hash`를 처음부터 붙이면 안 된다. ReplicaSet이 만들자마자 데려가는데 그때 새 Pod은 아직 Pending이고, 삭제 순서에서 Pending과 NotReady가 `pod-deletion-cost`보다 앞이라 방금 만든 Pod이 먼저 죽는다. +**The ReplicaSet is what moves the Pod. soft-drain only lays out the materials.** -`pod-template-hash`는 ReplicaSet selector에는 들어가고 Service selector에는 안 들어간다. 그래서 대체 Pod은 Ready가 되는 즉시 Endpoints에 올라가고 입양 전후로 빠지지 않는다. 엔드포인트 갱신이 Pod당 한 번뿐이다. +The hash must not be there from the start. The ReplicaSet would adopt the new Pod the moment +it appears, while it is still Pending — and Pending and NotReady rank ahead of +`pod-deletion-cost` in the deletion order, so the Pod just created would be the first to die. -**보장하는 것은 노출이지 어느 Pod이 지워지는가가 아니다.** `pod-deletion-cost`는 삭제 정렬의 4순위 힌트다. 넘기는 순간 무관한 Pod이 NotReady면 ReplicaSet은 그쪽을 지우고 우리 타깃은 남는다. 그래도 Ready인 Pod을 먼저 늘린 다음 초과분이 지워지므로 노출은 `N` 밑으로 내려가지 않고, 남은 타깃은 다음 라운드에 다시 시도된다. +`pod-template-hash` belongs to the ReplicaSet selector, not to the Service selector. A +replacement therefore enters Endpoints as soon as it is Ready and never drops out across +adoption. One endpoint update per Pod, no more. -## 쓰는 법 +**What is guaranteed is exposure, not which Pod gets deleted.** `pod-deletion-cost` is the +fourth-ranked hint in the deletion ordering. If an unrelated Pod is NotReady at the moment of +hand-over, the ReplicaSet deletes that one and our target survives. Exposure still never falls +below `N`, because a Ready Pod is added before the surplus is removed, and the surviving target +is tried again in the next round. + +## Usage ``` -kubectl label node node-01 soft-drain.com/drain=true # 시작 -kubectl get nodes -l soft-drain.com/state=Complete # 완료 확인 -kubectl label node node-01 soft-drain.com/drain- # 취소 -kubectl uncordon node-01 # 이것도 취소다 (state=Cancelled 로 남는다) +kubectl label node node-01 soft-drain.com/drain=true # start +kubectl get nodes -l soft-drain.com/state=Complete # check for completion +kubectl label node node-01 soft-drain.com/drain- # cancel +kubectl uncordon node-01 # also a cancel (leaves state=Cancelled) ``` -끝난 노드는 cordon된 채로 남는다. 그 다음에 drain을 하든 노드를 리부팅하든 soft-drain이 상관할 일이 아니다. 정비가 끝나 drain 라벨을 걷으면 우리가 걸었던 cordon도 함께 걷힌다 — 라벨 제거가 곧 노드 반환이다. 사람이 미리 걸어둔 cordon이면 그대로 둔다. +A finished node stays cordoned. What happens next — draining it, rebooting it — is not +soft-drain's business. When maintenance is over and the drain label comes off, the cordon we +placed comes off with it: removing the label is how the node is handed back. A cordon a human +placed beforehand is left alone. -### kubectl 플러그인 +### The kubectl plugin -`kubectl soft-drain`은 위 네 줄의 포장이다. **쓰는 것은 drain 라벨 하나뿐이고 나머지는 읽기다** — 서버 쪽 표면은 늘지 않는다. 문법은 `git stash`형이다 — 맨몸+노드가 주 동작이고, 나머지는 서브커맨드다. `status`, `release`, `version`은 예약어다. +`kubectl soft-drain` wraps the four lines above. **The only thing it writes is the drain label; +everything else is a read** — the server-side surface does not grow. The grammar follows +`git stash`: bare invocation plus nodes is the main verb, the rest are subcommands. `status`, +`release` and `version` are reserved words. ``` -kubectl soft-drain node-01 [node-02 ...] # 라벨을 붙이고 전부 Complete될 때까지 진행을 보여준다 (--wait=false, --timeout) -kubectl soft-drain status # 현황판: 관여 중인 노드·남은 타깃·대체 Pod -kubectl soft-drain status node-01 -o json # 특정 노드, 기계용 (json|yaml) -kubectl soft-drain release node-01 [...] # 라벨을 걷고 복원을 기다린다 +kubectl soft-drain node-01 [node-02 ...] # labels, then shows progress until all are Complete (--wait=false, --timeout) +kubectl soft-drain status # dashboard: managed nodes, remaining targets, replacements +kubectl soft-drain status node-01 -o json # one node, for machines (json|yaml) +kubectl soft-drain release node-01 [...] # removes the label and waits for the restore ``` -release는 진행 중이면 취소가 되고 Complete면 관리 종료가 된다 — 실체는 같은 라벨 제거고, 결말도 같다: 우리가 남긴 것을 전부 걷는다(우리가 걸었던 cordon 포함). 그래서 release는 정비가 끝난 뒤의 동사다 — 리부팅 전에 하면 비워 둔 노드가 도로 열린다. `kubectl uncordon`도 취소지만 라벨과 Cancelled 래치가 남는 점이 다르다 — release는 전부 걷는다. `--timeout`이 터지면 Pending 대체 Pod과 스케줄러 메시지를 보여주고 0이 아닌 코드로 나간다 — "막혔을 때 보는 법"의 자동화다. 중간에 끊어도 라벨은 남으므로 drain은 계속된다. `state=Cancelled`인 노드에 다시 drain을 걸면 라벨을 걷어 복원시킨 뒤 다시 붙인다. +`release` cancels a drain in progress and ends management of a Complete one — the same label +removal underneath, with the same ending: everything we left behind is taken back, including +the cordon we placed. That makes `release` the verb for after maintenance; run it before the +reboot and the node you just emptied opens up again. `kubectl uncordon` also cancels, but +leaves the label and the `Cancelled` latch behind — `release` takes back all of it. On +`--timeout` it prints the Pending replacements and the scheduler's message and exits non-zero: +the automation of "reading a stuck node" below. Interrupting it changes nothing, since the +label stays and the drain continues. Draining a node that is `state=Cancelled` removes the +label to restore it first, then puts it back. -현황판에 "언제부터"는 없다 — 컨트롤러가 무기억이라 시작 시각을 어디에도 기록하지 않는다. `-o`의 몫은 플러그인만 계산할 수 있는 집계(타깃·대체 Pod 상태·스케줄러 메시지)다. 노드명 목록이 필요한 기계는 라벨 조회가 정석이다: `kubectl get nodes -l soft-drain.com/state=Complete -o name`. +The dashboard has no "since when" — the controller is memoryless and records the start time +nowhere. What `-o` is for is the aggregation only the plugin can compute: targets, replacement +status, scheduler messages. A machine that needs a list of node names should query labels, as +usual: `kubectl get nodes -l soft-drain.com/state=Complete -o name`. -`kubectl drain`의 `--ignore-daemonsets`, `--delete-emptydir-data`, `--force`는 없다. eviction 전제의 개념이라 여기 해당이 없다. +There is no `--ignore-daemonsets`, `--delete-emptydir-data` or `--force`. Those are concepts +that presuppose eviction, and none applies here. -## 컨트롤러가 하는 일 +## What the controller does ``` -1. drain 라벨이 붙은 노드를 cordon한다 -2. 그 노드의 Deployment Pod(타깃)에 pod-deletion-cost 를 음수로 박는다 -3. 타깃마다 대체 Pod이 하나씩 있도록 맞춘다 -4. 대체 Pod이 Ready가 되면 pod-template-hash 를 붙여 ReplicaSet이 데려가게 한다 -5. 타깃이 없어질 때까지 반복한다 -6. 타깃이 없으면 완료 표시를 한다 +1. cordon the node carrying the drain label +2. write a negative pod-deletion-cost on that node's Deployment Pods (the targets) +3. keep one replacement Pod per target +4. once a replacement is Ready, attach pod-template-hash so the ReplicaSet takes it +5. repeat until no targets are left +6. with no targets left, mark it complete ``` -**어떤 상태도 기억하지 않는다.** 매번 클러스터를 다시 보고 전부 다시 판정하므로 중간에 죽어도 다음 라운드가 이어서 한다. +**No state is remembered.** Every round re-reads the cluster and re-derives everything, so a +controller that dies mid-way is simply continued by the next round. -**읽기는 API 서버에서 직접 한다.** watch는 다시 볼 때를 알려주는 데만 쓴다. 캐시가 뒤처진 상태를 설계가 감당하기 시작하면 조건이 급격히 복잡해진다. 부하가 문제가 되면 그때 캐시를 붙인다. +**Reads go straight to the API server.** Watches are used only to know when to look again. Once +the design starts absorbing stale-cache states the conditions get complicated fast. If load +ever becomes the problem, a cache can be added then. -### 1. 노드 마킹 +### Step 1. Marking the node -`drain` 라벨이 있으면 cordon한다. 우리가 실제로 값을 바꿨을 때만 `cordoned-by-controller` 어노테이션을 단다. 사람이 미리 걸어둔 cordon을 나중에 우리가 푸는 일을 막기 위해서다. +A node with the `drain` label gets cordoned. The `cordoned-by-controller` annotation is written +only when we actually changed the value. That is what stops us from later lifting a cordon a +human placed beforehand. -**cordon은 준비 작업이 아니라 이 반복이 끝나는 이유다.** cordon을 걸어두면 그 노드에는 Pod이 새로 안 뜬다. 빼야 할 Pod이 늘어날 일이 없으니 하나씩 빼다 보면 언젠가 바닥이 난다. 예외는 `unschedulable`을 tolerate하는 워크로드뿐인데, 그건 빼도 그 자리에 다시 앉아서 줄지를 않는다. 그래서 거기서 멈춘다(4번). +**The cordon is not preparation; it is the reason this loop terminates.** A cordoned node takes +no new Pods. The set of Pods to remove cannot grow, so removing them one at a time eventually +empties it. The one exception is a workload that tolerates `unschedulable`: removing such a Pod +just seats another one in the same place and the count never falls. That is where it stops +(step 3). -`drain` 라벨이 사라지면 되돌린다 — 우리 값이 박힌 `pod-deletion-cost`를 걷고, `cordoned-by-controller`가 있으면 uncordon하고, `state` 라벨을 지운다. +When the `drain` label disappears, everything is reverted — the `pod-deletion-cost` values we +wrote are removed, the node is uncordoned if `cordoned-by-controller` is present, and the +`state` label is deleted. -**누가 uncordon하면 관여를 접는다 — 진행 중이든 끝난 뒤든.** `state`가 `InProgress`나 `Complete`인데 노드가 `unschedulable`이 아니면 그렇게 된 것이다 — 두 상태 모두 cordon을 확인한 뒤에만 붙기 때문에 이 조합이 곧 증거다. 진행 중이라면 cordon은 종료를 보장하던 전제라서 전제가 사라진 채 계속할 수 없고, 끝난 뒤라면 사람이 우리 cordon을 풀고 노드를 다시 쓰기로 결정한 것이다. 어느 쪽이든 다시 cordon해서 사람과 싸우지 않는다. cost를 걷고 `cordoned-by-controller`를 지우고 `state=Cancelled`를 붙인 뒤 손을 뗀다. 어노테이션을 지우는 이유: 기록된 cordon은 사람 손에 이미 풀렸으므로, 이후 사람이 새로 건 cordon을 라벨 제거 시점의 복원이 우리 것으로 오인해 풀면 안 된다. 대체 Pod은 회수 경로가 걷는다. `Cancelled`는 래치다 — 라벨을 걷으면 지워지고, 다시 하려면 라벨을 걷었다가 다시 붙인다. +**If someone uncordons the node, we let go — mid-drain or after.** A node whose `state` is +`InProgress` or `Complete` but which is no longer `unschedulable` got there that way: both +states are only ever set after confirming the cordon, so the combination is the evidence. +Mid-drain, the cordon was the premise that guaranteed termination, and we cannot continue +without it; after completion, a human has lifted our cordon and decided to use the node again. +Either way we do not re-cordon and fight them. The costs are removed, `cordoned-by-controller` +is deleted, `state=Cancelled` is set, and we stop. The annotation goes because the cordon it +recorded is already gone by a human's hand: a cordon they place later must not be mistaken for +ours and lifted by the restore that runs on label removal. Replacement Pods are collected by +the reclamation path. `Cancelled` is a latch — removing the label clears it, and starting over +means removing the label and adding it again. -`Complete`를 접지 않고 두면 그 노드가 삭제 자석이 된다. 방금 비워져 가장 한가한 노드가 열렸으니 스케줄러는 다른 drain의 대체 Pod을 정확히 거기 앉히고, 착지 검사는 앉는 족족 지운다. 클러스터가 작을수록 모든 drain이 그 노드로 빨려 들어간다. +Leaving `Complete` unfinished turns the node into a deletion magnet. The node with the most +room, just emptied, is now open, so the scheduler seats another drain's replacements exactly +there, and the landing check deletes them as fast as they arrive. The smaller the cluster, the +more every drain is pulled into that one node. -### 2. 타깃 표시 +### Step 2. Marking targets -타깃은 **그 노드 위에서 owner가 ReplicaSet이고 그 ReplicaSet의 owner가 Deployment인 Pod**이다. phase가 `Failed`나 `Succeeded`인 Pod은 빼고 센다 — ReplicaSet도 active로 세지 않는 Pod이라 대체는 이미 딴 곳에 만들어져 있고, 노드에 남은 시체가 완료 판정만 막는다. +A target is **a Pod on that node whose owner is a ReplicaSet whose owner is a Deployment**. +Pods in phase `Failed` or `Succeeded` are not counted — the ReplicaSet does not count them as +active either, so their replacements already exist elsewhere, and the corpses left on the node +would only block the completion check. -`controller.kubernetes.io/pod-deletion-cost = -2147483648`을 쓴다. 타깃에만 쓰고, 시점은 대체 Pod을 만들기 전이다 — 넘기기까지 미룰 이유가 없고, 그 사이 무관한 스케일다운이 나도 드레인 대상이 먼저 죽는 쪽이 낫다. +The annotation used is `controller.kubernetes.io/pod-deletion-cost = -2147483648`. It is +written on targets only, and it is written before the replacement is created: there is no +reason to wait until hand-over, and if an unrelated scale-down happens in between, the drained +node's Pods are the better ones to lose. -대체 Pod에는 쓰지 않는다. 입양 전에는 ReplicaSet이 쳐다보지 않는 Pod이라 값이 무의미하고, 입양 후에 남으면 그 Pod이 다음 스케일다운마다 1순위로 죽는다. 타깃은 지워지면서 값도 같이 사라지므로 걷을 것이 없다. +It is not written on replacements. Before adoption the ReplicaSet does not look at that Pod, so +the value means nothing; if it survives adoption, that Pod becomes the first to die in every +later scale-down. Targets carry the value away with them when they are deleted, so there is +nothing to clean up. -원래 값이 있었어도 덮어쓰고 복원하지 않는다. 되돌릴 때는 값이 정확히 `-2147483648`인 것만 지운다. 그 값을 쓰는 게 우리뿐이라 값이 이것이면 우리가 붙인 것이다. +An existing value is overwritten and not restored. On revert, only values that are exactly +`-2147483648` are removed. Nothing else writes that value, so a value of exactly that is one of +ours. -### 3. 대체 Pod 맞추기 +### Step 3. Matching replacements -만들기만 하는 단계가 아니다. **있어야 할 집합과 있는 집합을 맞춘다.** +This step does not only create. **It reconciles the set that should exist against the set that +does.** ``` -있어야 할 것 = deletionTimestamp 가 없는 타깃마다 하나 -있는 것 = soft-drain.com/replaces = <타깃 UID> 이면서 - controller ownerRef 없고 - phase 가 Failed / Succeeded 가 아니고 - deletionTimestamp 도 없는 Pod +should exist = one per target without a deletionTimestamp +does exist = Pods with soft-drain.com/replaces = + that have no controller ownerRef, + are not in phase Failed / Succeeded, + and have no deletionTimestamp -모자라면 만들고, 남으면 지운다 +too few, create; too many, delete ``` -**타깃이 사라지면 대체 Pod도 사라진다.** 취소, 롤아웃으로 인한 ReplicaSet prune, Deployment 삭제, `replicas` 축소, 타깃의 eviction이 전부 이 한 줄에 걸린다. 따로 처리할 것이 없다. - -**terminating 타깃은 만들기에서 뺀다.** ReplicaSet은 `deletionTimestamp`가 찍힌 Pod을 active에서 빼므로 이미 스스로 대체를 만들고 있고, 노드가 cordon이라 그 Pod은 다른 노드에 뜬다. 자리는 우리가 아무것도 안 해도 비워진다. - -**죽은 대체 Pod은 있는 것으로 세지 않고, 지운다.** 노드 압력 eviction이나 kubelet admission 거부로 `Failed`가 된 Pod은 Ready가 될 수도 입양될 수도 없다. 살아 있는 것으로 세면 그 타깃이 영원히 멈춘다. Pod에는 재시작이 없어서(phase `Failed`는 터미널이고 `restartPolicy`는 컨테이너 얘기다) 복구는 새 Pod뿐인데, 세지 않고 지우지도 않으면 만들 때마다 시체가 쌓인다. 원인이 지속되면 만들고-죽고-지우기를 반복하다가 원인이 풀리는 순간 수렴한다. - -**drain 중인 노드에 앉은 대체 Pod도 세지 않고, 지운다.** cordon보다 스케줄이 먼저여서, 앉은 뒤에 그 노드에 drain이 걸리는 경우가 생긴다. Ready를 기다릴 이유가 없다 — Ready가 되어도 넘기면 비우려는 노드에 타깃이 하나 더 생길 뿐이라 결말은 삭제뿐이고, 그동안 자리만 먹는다. hash가 없어 어느 ReplicaSet의 자식도 아니므로 지워도 노출은 줄지 않고, 같은 라운드가 새로 만들면 스케줄러가 cordon을 피해 앉힌다. 지울 때 Warning Event를 남긴다. - -`node.kubernetes.io/unschedulable`을 tolerate하는 워크로드는 새로 만든 Pod이 또 drain 노드에 앉을 수 있고, 그러면 만들고 지우기를 반복한다. cordon을 무시하도록 만든 워크로드이므로 우리가 `nodeAffinity`를 주입해 그 의도를 뒤집지 않는다. 반복되는 Warning Event가 옮길 수 없다는 사실을 보여준다. - -**타깃의 ReplicaSet이 Deployment의 현재 템플릿이 아니면 만들지 않고, 있으면 지운다.** 롤아웃이 그 타깃을 이미 대체하는 중이다 — 새 버전을 다른 노드에 올리고 Ready 후 타깃을 지우는, 우리가 하려던 일 그대로다. 우리 대체 Pod은 낙선한 버전을 짓는 잉여이고, Healthy(D)가 롤아웃 내내 넘기기를 막으므로 입양에 도달할 수도 없다. `pod-deletion-cost`는 타깃에 남아 있으므로 old ReplicaSet의 스케일다운이 drain 노드의 타깃부터 지운다. 지울 때 Normal Event를 남긴다. 판정은 Deployment의 `spec.template`과 타깃 ReplicaSet의 템플릿을 `pod-template-hash` 라벨만 빼고 비교한다(Deployment 컨트롤러의 EqualIgnoreHash와 같다) — 이미지 변경이든 `rollout restart`든, 템플릿이 바뀌는 모든 경우가 같은 길로 잡힌다. 단 `spec.paused`면 템플릿이 달라도 이 규칙을 적용하지 않는다 — 롤아웃이 실제로 움직이지 않아 "대체 중"이라는 전제가 깨진다. 대체는 평소처럼 유지되고 넘기기만 Healthy(D)에 걸려 미뤄지다가, 재개되는 순간 이 규칙이 잡는다. - -**두 삭제가 preconditions에 거부되면 라운드를 접는다.** 거부는 판정과 삭제 사이에 Pod이 변했다는 뜻이다. 낡은 명단으로 넘기기까지 진행하지 않고 짧은 requeue로 라운드를 끝내, 다음 라운드가 새 상태에서 처음부터 판정한다. 그래서 넘기기는 "drain 중인 노드에 앉은 대체"를 다시 검사할 필요가 없다. - -**이미 넘긴 것도 세지 않는다.** 넘기면서 `soft-drain.com/replaces` 라벨을 떼기 때문에 애초에 후보가 아니다. 그래서 넘겼는데 타깃이 살아남은 경우가 자연히 복구된다 — 넘기는 순간 `replicas`가 올라가면 초과분이 증설분에 흡수되어 아무것도 안 지워지는데, 다음 라운드가 "타깃은 그대로인데 대신할 Pod이 없다"를 보고 하나 더 만든다. - -**만드는 쪽과 지우는 쪽 양쪽에서 깨어나야 한다.** 노드에서 출발하는 순회만 있으면, ReplicaSet이 prune될 때 타깃 Pod도 같이 사라져서 순회할 대상이 없어지고 대체 Pod을 쳐다볼 일이 없어진다. 그래서 대체 Pod 자체를 키로 하는 경로가 따로 있어야 한다. 판정은 위 한 줄로 같다. - -대체 Pod은 이렇게 만든다. +**When a target goes away, so does its replacement.** Cancellation, a ReplicaSet pruned by a +rollout, a deleted Deployment, a lowered `replicas`, an evicted target — all of it is covered +by that single line. Nothing needs handling on its own. + +**Terminating targets are excluded from creation.** A ReplicaSet drops Pods with a +`deletionTimestamp` from its active count, so it is already making its own replacement, and +since the node is cordoned that Pod lands elsewhere. The slot is freed without us doing +anything. + +**Dead replacements are not counted, and are deleted.** A Pod that went `Failed` through node +pressure eviction or a kubelet admission rejection can neither become Ready nor be adopted. +Counting it as alive stalls that target forever. There is no restart for a Pod — phase `Failed` +is terminal and `restartPolicy` is about containers — so recovery means a new Pod; and if the +dead one is neither counted nor deleted, corpses pile up with every attempt. While the cause +persists this cycles through create-die-delete, and it converges the moment the cause clears. + +**Replacements seated on a draining node are not counted either, and are deleted.** +Scheduling can precede the cordon, so a Pod can land on a node that only afterwards gets the +drain label. There is no reason to wait for Ready: even Ready, handing it over would just add +one more target to the node we are emptying, so deletion is the only ending, and meanwhile it +occupies a slot. It has no hash, so it is no ReplicaSet's child, and deleting it does not lower +exposure; the same round creates a new one and the scheduler seats it away from the cordon. A +Warning Event is emitted on deletion. + +A workload that tolerates `node.kubernetes.io/unschedulable` may have its new Pod land on a +draining node again, and then create-and-delete repeats. It is a workload built to ignore +cordons, so we do not invert that intent by injecting `nodeAffinity`. The repeating Warning +Events are what show that it cannot be moved. + +**If the target's ReplicaSet is not the Deployment's current template, nothing is created, and +anything present is deleted.** A rollout is already replacing that target — bringing the new +version up on another node and deleting the target once it is Ready, which is exactly what we +were about to do. Our replacement would build the losing version for nothing, and Healthy(D) +blocks hand-over for the whole rollout, so it could never reach adoption anyway. +`pod-deletion-cost` stays on the target, so the old ReplicaSet's scale-down deletes the drained +node's targets first. A Normal Event is emitted on deletion. The check compares the +Deployment's `spec.template` against the target ReplicaSet's template, ignoring only the +`pod-template-hash` label (the same EqualIgnoreHash the Deployment controller uses) — an image +change, a `rollout restart`, every case where the template changes is caught the same way. The +exception is `spec.paused`: with a paused Deployment the rule does not apply even if the +templates differ, because the rollout is not actually moving and the premise of "already being +replaced" breaks. Replacements are maintained as usual and only hand-over is held back by +Healthy(D), until the rollout resumes and this rule catches it. + +**If either deletion is rejected by its preconditions, the round ends.** A rejection means the +Pod changed between the decision and the deletion. Rather than carry a stale list into +hand-over, the round ends with a short requeue and the next round decides everything again from +the current state. That is why hand-over does not need to re-check for replacements seated on a +draining node. + +**Pods already handed over are not counted.** The `soft-drain.com/replaces` label is removed as +part of the hand-over, so they are not candidates to begin with. This is what makes the +"handed over but the target survived" case recover on its own: if `replicas` goes up at the +moment of hand-over, the surplus is absorbed by the increase and nothing is deleted — and the +next round sees "the target is still there and nothing stands in for it" and creates one more. + +**Both the creating and the deleting side must be able to wake us.** With only the traversal +that starts from the node, a pruned ReplicaSet takes the target Pods with it, leaving nothing +to traverse and no reason to ever look at the replacements again. So there is a separate path +keyed on the replacement Pod itself. The decision is the same single rule above. + +A replacement is built like this. ```yaml metadata: - generateName: aaa-5449d4d8c8- # 타깃의 ReplicaSet 이름 + "-" + generateName: aaa-5449d4d8c8- # the target's ReplicaSet name + "-" labels: - app: aaa # rs.spec.template.metadata.labels 에서 - soft-drain.com/replaces: 3f2a... # 타깃 Pod의 UID - # pod-template-hash 는 뺀다 -spec: + app: aaa # from rs.spec.template.metadata.labels + soft-drain.com/replaces: 3f2a... # the target Pod's UID + # pod-template-hash is removed +spec: ``` -스펙은 **살아 있는 Pod이 아니라 `rs.spec.template`에서** 가져온다. 살아 있는 Pod을 베끼면 `nodeName`이 따라오고, webhook이 이미 넣어둔 사이드카 위에 하나가 더 들어간다. +The spec comes **from `rs.spec.template`, not from the living Pod.** Copying the living Pod +brings `nodeName` along, and stacks another sidecar on top of the one a webhook already +injected. -`rs.spec.template.metadata.labels`에는 `pod-template-hash`가 **이미 들어 있다.** 복사한 뒤 명시적으로 제거한다. 이 문서에서 가장 중요한 한 줄이다. +`rs.spec.template.metadata.labels` **already contains** `pod-template-hash`. It is removed +explicitly after the copy. This is the single most important line in this document. -생성이 거부되면 Warning Event를 낸다. ResourceQuota 초과나 admission webhook 거부가 여기 걸리는데, 이 경우만 Pod 오브젝트가 안 생겨서 밖에서 볼 흔적이 없다. 거부돼도 멈추지 않고 다음 라운드에 다시 시도한다. +A rejected creation produces a Warning Event. A ResourceQuota overrun or an admission webhook +rejection lands here, and this is the one case where no Pod object exists, so there is no trace +to see from outside. A rejection does not stop anything; the next round tries again. -### 4. 넘기기 +### Step 4. The hand-over -대체 Pod이 Ready가 되면 patch 하나로 `pod-template-hash`를 붙이고 `soft-drain.com/replaces`를 뗀다. +Once a replacement is Ready, a single patch adds `pod-template-hash` and removes +`soft-drain.com/replaces`. -붙일 hash는 **타깃 Pod의 ownerRef가 가리키는 ReplicaSet**에서 읽는다. Deployment를 거쳐 현재 ReplicaSet을 찾는 경로는 쓰지 않는다 — 대체 Pod은 타깃의 ReplicaSet 템플릿으로 만들어졌고, 롤아웃 중이면 그게 현재 ReplicaSet이 아닐 수 있다. +The hash is read **from the ReplicaSet the target Pod's ownerRef points at**. The route through +the Deployment to the current ReplicaSet is not used — the replacement was built from the +target's ReplicaSet template, and during a rollout that may not be the current one. -넘기기 전에 하나를 본다. drain 중인 노드에 앉은 대체는 3단계가 이미 지웠으므로 여기 오지 않는다. +One thing is checked before handing over. Replacements seated on a draining node were already +deleted in step 3, so they never get here. -**사용자 Deployment가 Healthy한가.** +**Is the user's Deployment healthy.** ``` Healthy(D) ≡ D.status.observedGeneration >= D.metadata.generation @@ -154,69 +279,114 @@ Healthy(D) ≡ D.status.observedGeneration >= D.metadata.generation ∧ D.status.availableReplicas >= D.spec.replicas ``` -Healthy가 아니면 미룬다. 사용자가 `N` 미만이면 넘겨도 초과분이 없어 아무것도 지워지지 않고, 롤아웃 중이면 넘겨받을 ReplicaSet이 하나로 정해지지 않아 노출이 rollout 설정보다 더 내려갈 수 있다. +If it is not healthy, hold. Below `N` the hand-over creates no surplus and nothing gets +deleted; mid-rollout there is no single ReplicaSet to hand over to, and exposure could fall +further than the rollout settings allow. -`replicas == updatedReplicas` 항이 "Pod을 가진 ReplicaSet이 하나뿐"을 판정한다. 나머지 두 항만으로는 롤아웃을 못 잡는다 — `maxUnavailable: 0`이면 `availableReplicas >= N`이 롤아웃 내내 유지되는데, 무중단을 원하는 사용자가 정확히 그 설정을 쓴다. `spec.paused`도 이 항에 걸린다. +The `replicas == updatedReplicas` term is what decides "only one ReplicaSet has Pods". The +other two terms cannot catch a rollout on their own: with `maxUnavailable: 0`, +`availableReplicas >= N` holds for the entire rollout — and that is exactly the setting a user +who wants zero downtime uses. `spec.paused` is caught by this term as well. -판정은 Deployment마다 따로 하고 준비된 것부터 넘긴다. 묶으면 제일 느린 하나가 나머지를 인질로 잡는다. +The check is per Deployment, and whichever is ready hands over first. Batching them lets the +slowest one hold the rest hostage. -### 5. 완료 +### Step 5. Completion -노드 위에 타깃이 하나도 없으면 `state=Complete`를 붙인 뒤 Event를 낸다. `cordoned-by-controller` 어노테이션은 그대로 둔다 — cordon은 여전히 우리가 건 것이고, drain 라벨이 걷힐 때 함께 걷힌다. `Complete`는 래치다 — cordon이 유지되는 동안에는 drain 라벨이 걷힐 때까지 관여하지 않는다. cordon된 노드에 새로 앉을 수 있는 건 `unschedulable`을 tolerate하는 Pod뿐인데, 그건 어차피 옮기지 못하는 부류다. 래치가 없으면 리부팅을 기다리는 노드가 도로 열려 Pod이 몰린 채로 리부팅하게 된다 — 노드를 여는 순간은 사람이 라벨을 걷을 때(반환)와 uncordon할 때(취소)뿐이어야 한다. +With no targets left on the node, `state=Complete` is set and an Event is emitted. The +`cordoned-by-controller` annotation stays — the cordon is still ours, and it comes off when the +drain label does. `Complete` is a latch: while the cordon holds, we do not act again until the +drain label is removed. The only Pods that can newly land on a cordoned node are the ones that +tolerate `unschedulable`, and those are the kind we cannot move anyway. Without the latch a +node waiting for its reboot would reopen and be rebooted with Pods crowded back onto it — the +only moments a node opens should be a human removing the label (hand-back) and a human +uncordoning it (cancel). -사람이 uncordon하면 래치는 `Cancelled`로 접힌다(1번). 착지 금지도 함께 풀린다 — 리부팅하러 갈 노드라서 막았던 것인데, uncordon은 리부팅 안 간다는 선언이다. uncordon과 감지 사이의 짧은 창에서는 착지한 대체 Pod이 지워질 수 있지만, 다음 라운드가 새로 만든다. +A human uncordoning folds the latch into `Cancelled` (step 1). The landing ban lifts with it — +it existed because the node was headed for a reboot, and an uncordon declares that it is not. +In the short window between the uncordon and our noticing it, a landed replacement can still be +deleted, but the next round creates a new one. -**완료 판정에는 terminating 타깃도 센다.** `deletionTimestamp`가 찍혀도 grace period 동안 계속 돈다. 여기서 빼면 아직 작업이 돌고 있는 노드에 `Complete`가 붙고, 그걸 보고 노드를 리부팅한 사람이 그 작업을 죽인다. 만들기에서는 빼고 완료 판정에서는 세는 이유가 이것이다. +**Terminating targets do count toward completion.** A `deletionTimestamp` does not stop the +work; it keeps running for the grace period. Excluding them would put `Complete` on a node +whose work is still running, and the human who reboots the node on that signal kills it. That +is the reason they are excluded from creation but counted for completion. -## 메타데이터 +## Metadata -| 대상 | 키 | 값 | 쓰는 쪽 | +| Object | Key | Value | Written by | |---|---|---|---| -| 노드 | `soft-drain.com/drain` (라벨) | `"true"` | 사람 | -| 노드 | `soft-drain.com/state` (라벨) | `InProgress` / `Complete` / `Cancelled` | 컨트롤러 | -| 노드 | `soft-drain.com/cordoned-by-controller` (어노테이션) | `"true"` | 컨트롤러 | -| 타깃 Pod | `controller.kubernetes.io/pod-deletion-cost` (어노테이션) | `-2147483648` | 컨트롤러 | -| 대체 Pod | `soft-drain.com/replaces` (라벨) | 타깃 Pod의 UID | 컨트롤러 | - -`soft-drain.com/replaces`를 쓰는 것은 우리뿐이다. 이 라벨이 없는 Pod은 만들지도 지우지도 않는다. - -## 지켜야 할 것 - -1. 대체 Pod은 `pod-template-hash` 없이 만든다. -2. 대체 Pod의 스펙은 살아 있는 Pod이 아니라 `rs.spec.template`에서 가져온다. -3. `pod-deletion-cost`를 먼저 쓰고 `pod-template-hash`를 나중에 붙인다. -4. `soft-drain.com/replaces` 라벨이 있는 Pod만 지운다. -5. controller ownerRef가 있는 Pod은 지우지 않는다. -6. 대체 Pod을 지울 때는 읽었던 UID와 resourceVersion을 preconditions로 건다. 판정과 삭제 사이에 hash가 붙어 ReplicaSet이 데려간 Pod이면 삭제가 거부되고, 다음 라운드가 다시 판정한다. - -## 안 하는 것 - -- Deployment 소속 Pod만 옮긴다. StatefulSet, DaemonSet, Job, 직접 만든 Pod은 그대로 둔다. **`Complete`는 "내 몫이 끝났다"이지 "노드가 비었다"가 아니다.** -- 노드 위 대상 Pod을 한꺼번에 옮긴다. 자원이 모자라면 Pending으로 기다린다. -- 옮길 수 없는 워크로드를 미리 걸러내지 않는다. Pending으로 남고 사람이 보면 된다. -- PDB를 조회하지 않는다. 지우는 주체가 사용자 ReplicaSet이라 eviction API를 타지 않는다. -- 사용자 Deployment의 `spec`을 수정하지 않는다. 사용자 Pod에는 어노테이션 하나만 쓴다. -- 노드를 drain하거나 끄지 않는다. - -## 막혔을 때 보는 법 - -노드가 `InProgress`에서 안 움직이면 대체 Pod을 본다. +| Node | `soft-drain.com/drain` (label) | `"true"` | human | +| Node | `soft-drain.com/state` (label) | `InProgress` / `Complete` / `Cancelled` | controller | +| Node | `soft-drain.com/cordoned-by-controller` (annotation) | `"true"` | controller | +| Target Pod | `controller.kubernetes.io/pod-deletion-cost` (annotation) | `-2147483648` | controller | +| Replacement Pod | `soft-drain.com/replaces` (label) | the target Pod's UID | controller | + +Nothing but soft-drain writes `soft-drain.com/replaces`. A Pod without that label is neither +created nor deleted by us. + +## Invariants + +1. A replacement is created without `pod-template-hash`. +2. A replacement's spec comes from `rs.spec.template`, not from the living Pod. +3. `pod-deletion-cost` is written first, `pod-template-hash` attached later. +4. Only Pods carrying the `soft-drain.com/replaces` label are deleted. +5. Pods with a controller ownerRef are never deleted. +6. Deleting a replacement uses the UID and resourceVersion we read as preconditions. If the + hash was attached in between and a ReplicaSet took the Pod, the deletion is rejected and the + next round decides again. + +## Non-goals + +- Only Pods belonging to a Deployment are moved. StatefulSets, DaemonSets, Jobs and + hand-made Pods are left as they are. **`Complete` means "my part is done", not "the node is + empty".** +- Every eligible Pod on the node is moved at once. If resources run short, they wait as + Pending. +- Workloads that cannot be moved are not filtered out ahead of time. They stay Pending and a + human can look. +- PDBs are not consulted. The deletion is performed by the user's ReplicaSet, so it never goes + through the eviction API. +- The user's Deployment `spec` is never modified. On user Pods we write exactly one annotation. +- Nodes are neither drained nor shut down. + +## Reading a stuck node + +When a node sits at `InProgress`, look at the replacements. ```bash kubectl get pods -A -l soft-drain.com/replaces -kubectl describe pod +kubectl describe pod ``` -스케줄러가 `PodScheduled=False`의 message에 이유를 그대로 써 둔다 — `0/12 nodes are available: 5 Insufficient cpu, 7 node(s) didn't match pod anti-affinity rules` 같은 식이다. 컨트롤러가 따로 진단을 만들지 않는 이유다. - -대체 Pod이 하나도 안 보이면 생성이 거부됐거나(ResourceQuota, admission webhook), 롤아웃이 이주를 대신 수행 중이라 만들지 않는 것이다. 어느 쪽이든 `kubectl describe node <노드>`의 Event에 남는다. - -## 알려진 한계 - -- **여유 자원이 없으면 진행하지 못한다.** 자리는 옛 Pod이 죽어야 나고 옛 Pod은 새 Pod이 Ready여야 죽으므로, 여유가 0이면 스스로 풀리지 않는다. RWO 볼륨과 로컬 PV도 같은 구조다. -- **배치 규칙이 한 자리를 못 내주면 자원이 남아돌아도 진행하지 못한다.** 노드당 하나로 제한하는 required `podAntiAffinity`를 걸어두고 후보 노드를 전부 채운 경우가 대표적이다. 이건 soft-drain만의 제약이 아니라 "먼저 띄우고 나중에 지운다"는 방식 전체의 산술이다 — 같은 워크로드는 `maxSurge: 1` 롤아웃도 똑같이 막힌다. 그래서 그런 사용자는 이미 `maxUnavailable: 1`로 운영하며 롤아웃마다 `N` 밑으로 내려가는 것을 감수하고 있다. 노드를 뺄 때도 `kubectl drain`을 쓰면 된다. -- **`node.kubernetes.io/unschedulable`을 tolerate하는 워크로드는 옮기지 못할 수 있다.** 대체 Pod이 drain 노드에 앉을 때마다 지우고 다시 만들기를 반복하고, 다른 노드에 앉는 운이 따라야 끝난다. -- **입양 전 대체 Pod은 controller가 없어 PDB 집계를 흔든다.** 같은 라벨을 갖고 Ready라 `currentHealthy`에는 들어가는데 `expectedCount`에는 안 들어가서, 그동안 `disruptionsAllowed`가 1 늘어난다. PDB가 지키는 바닥 아래로 내려가지는 않는다. 같은 이유로 사용자 PDB에 `UnmanagedPods` Warning이 쌓인다. -- **주인 없는 대체 Pod이 있는 노드는 Cluster Autoscaler가 축소하지 못한다.** 넘기기가 멈춘 상태로 오래 가면 그 노드가 컨솔리데이션에서 계속 빠진다. -- **사용자 Pod의 `pod-deletion-cost` 원래 값은 복원하지 않는다.** -- **`pod-deletion-cost`가 필요하므로 Kubernetes 1.22 이상이어야 한다.** +The scheduler writes the reason verbatim into the message of `PodScheduled=False` — something +like `0/12 nodes are available: 5 Insufficient cpu, 7 node(s) didn't match pod anti-affinity +rules`. That is why the controller does not produce a diagnosis of its own. + +If no replacement is there at all, either the creation was rejected (ResourceQuota, admission +webhook) or a rollout is performing the migration instead, so none is created. Either way it is +in the Events of `kubectl describe node `. + +## Known limitations + +- **With no spare capacity there is no progress.** A slot opens only when an old Pod dies, and + an old Pod dies only once a new one is Ready, so at zero headroom nothing resolves itself. + RWO volumes and local PVs have the same shape. +- **If placement rules cannot yield a single slot, there is no progress even with resources to + spare.** The typical case is a required `podAntiAffinity` limiting one per node with every + candidate node already filled. This is not a soft-drain constraint but the arithmetic of + "bring it up first, delete it later" in general — the same workload also blocks a + `maxSurge: 1` rollout. Which is why such users already run with `maxUnavailable: 1` and accept + dropping below `N` on every rollout. They can use `kubectl drain` to take a node out too. +- **A workload tolerating `node.kubernetes.io/unschedulable` may never move.** Every time a + replacement lands on the draining node it is deleted and recreated, and it only ends if one + happens to land elsewhere. +- **Before adoption a replacement has no controller, which skews PDB accounting.** It carries + the same labels and is Ready, so it counts toward `currentHealthy` but not toward + `expectedCount`, which raises `disruptionsAllowed` by one for that period. It never goes below + the floor the PDB protects. For the same reason `UnmanagedPods` Warnings accumulate on the + user's PDB. +- **The Cluster Autoscaler cannot scale down a node holding an ownerless replacement.** If + hand-over stays stalled for long, that node keeps being excluded from consolidation. +- **The original `pod-deletion-cost` value on a user Pod is not restored.** +- **`pod-deletion-cost` is required, so Kubernetes 1.22 or newer is needed.** diff --git a/README.md b/README.md index f4fc4d4..57ef772 100644 --- a/README.md +++ b/README.md @@ -217,5 +217,5 @@ make test-e2e # kind cluster, ~15 min ## Design -The full design rationale lives in [DESIGN.md](DESIGN.md) (Korean; English translation -planned). It is the source of truth for how the controller behaves. +The full design rationale lives in [DESIGN.md](DESIGN.md) — the source of truth for how the +controller behaves. A Korean translation is kept in [DESIGN.ko.md](DESIGN.ko.md). diff --git a/cmd/kubectl-soft_drain/main.go b/cmd/kubectl-soft_drain/main.go index 5f4fde5..7249087 100644 --- a/cmd/kubectl-soft_drain/main.go +++ b/cmd/kubectl-soft_drain/main.go @@ -1,5 +1,5 @@ -// kubectl-soft_drain은 kubectl 플러그인이다 (kubectl soft-drain 으로 호출). -// 쓰는 것은 drain 라벨 하나뿐이고 나머지는 읽기다. +// kubectl-soft_drain is a kubectl plugin (invoked as kubectl soft-drain). +// The only thing it writes is the drain label; everything else is a read. package main import ( @@ -27,7 +27,7 @@ import ( const pollInterval = 2 * time.Second -// version은 빌드 시 -ldflags "-X main.version=..."로 주입된다. +// version is injected at build time with -ldflags "-X main.version=...". var version = "dev" func main() { @@ -35,8 +35,8 @@ func main() { if err == nil { return } - // Ctrl-C는 실패가 아니라 분리다 — 라벨은 남고 작업은 계속된다. - // 종료코드 130(128+SIGINT)으로 "완주 안 됨"만 관례대로 알린다. + // Ctrl-C is a detach, not a failure — the label stays and the work continues. + // Exit code 130 (128+SIGINT) reports only "did not run to completion", as is conventional. if errors.Is(err, context.Canceled) { fmt.Fprintln(os.Stderr, "interrupted; the operation keeps running on the cluster (labels remain)") fmt.Fprintln(os.Stderr, "check with: kubectl soft-drain status") @@ -52,8 +52,9 @@ func run() error { timeout := flags.Duration("timeout", 0, "give up waiting after this duration (0 = wait forever; the operation itself keeps running)") output := flags.StringP("output", "o", "", "status output format: json or yaml") - // nil인 필드는 플래그로 등록되지 않는다. 연결 선택(--kubeconfig, --context)만 - // 남긴다 — 이 플러그인에 네임스페이스나 고급 연결 플래그는 의미가 없다. + // A nil field is not registered as a flag. Only the connection selectors + // (--kubeconfig, --context) are kept — namespaces and advanced connection flags mean + // nothing to this plugin. cfg := genericclioptions.NewConfigFlags(true) cfg.CacheDir = nil cfg.ClusterName = nil @@ -96,7 +97,7 @@ the label and restores the node fully. flags.Usage() return nil } - // version은 클러스터 없이도 답해야 한다 — 클라이언트 생성보다 먼저 본다. + // version must answer without a cluster — checked before the client is built. if args[0] == "version" { fmt.Printf("kubectl-soft_drain %s\n", version) return nil @@ -136,7 +137,7 @@ the label and restores the node fully. } err2 = runDrain(ctx, cs, args, *waitDone) } - // 어느 지점에서 끊겼든, 신호가 원인이면 실패가 아니라 분리다. + // Wherever it was interrupted, a signal means a detach, not a failure. if err2 != nil && sigCtx.Err() != nil { return context.Canceled } @@ -164,8 +165,8 @@ type nodeStatus struct { Replacements []replacementStatus `json:"replacements"` } -// runStatus는 soft-drain이 관여 중인 노드의 현황이다. 전부 읽기다. -// 노드명 목록만 필요한 기계는 라벨 조회가 정석이다: +// runStatus reports on the nodes soft-drain is managing. Reads only. +// A machine that just needs node names should query labels, as usual: // kubectl get nodes -l soft-drain.com/state=Complete -o name func runStatus(ctx context.Context, cs kubernetes.Interface, filter []string, output string) error { if output != "" && output != "json" && output != "yaml" { @@ -294,7 +295,7 @@ func runDrain(ctx context.Context, cs kubernetes.Interface, nodes []string, wait fmt.Printf("node/%s is already drained (state=Complete)\n", node) continue case sd.StateCancelled: - // Cancelled는 래치다 — 라벨을 걷어 복원시킨 뒤에만 다시 걸 수 있다. + // Cancelled is a latch — the label must be removed to restore before it can be set again. fmt.Printf("node/%s has a cancelled drain; clearing it first\n", node) if err := setDrainLabel(ctx, cs, node, false); err != nil { return err @@ -402,8 +403,8 @@ func watchDrains(ctx context.Context, cs kubernetes.Interface, nodes []string) e return nil } -// diagnose는 "막혔을 때 보는 법"의 자동화다 — Pending 대체 Pod의 스케줄러 -// 메시지를 그대로 보여준다. +// diagnose automates "reading a stuck node" — it shows the scheduler's message for +// Pending replacements verbatim. func diagnose(cs kubernetes.Interface, node string, seenTargets map[types.UID]string) { ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() @@ -426,8 +427,8 @@ func diagnose(cs kubernetes.Interface, node string, seenTargets map[types.UID]st } } -// runRelease는 라벨을 걷는다. 진행 중이면 취소가 되고, Complete면 관리 종료가 -// 된다 — 실체는 같은 동작이다. +// runRelease removes the label. On a drain in progress that is a cancel, on a Complete +// node it ends management — underneath it is the same action. func runRelease(ctx context.Context, cs kubernetes.Interface, nodes []string, waitDone bool) error { var pending []string for _, node := range nodes { @@ -489,8 +490,8 @@ func waitForState(ctx context.Context, cs kubernetes.Interface, node, want strin } } -// markedPods는 노드 위에서 컨트롤러가 cost를 박은 Pod을 준다. 값이 -// -2147483648이면 우리가 붙인 것이다 — 그 값을 쓰는 게 우리뿐이다. +// markedPods returns the Pods on the node the controller wrote a cost on. A value of +// -2147483648 is ours — nothing else writes it. func markedPods(ctx context.Context, cs kubernetes.Interface, node string) ([]corev1.Pod, error) { list, err := cs.CoreV1().Pods("").List(ctx, metav1.ListOptions{ FieldSelector: "spec.nodeName=" + node, diff --git a/internal/controller/envtest_test.go b/internal/controller/envtest_test.go index 3bc7cb0..b7d8ad3 100644 --- a/internal/controller/envtest_test.go +++ b/internal/controller/envtest_test.go @@ -16,10 +16,10 @@ limitations under the License. package controller -// envtest에는 kube-controller-manager, scheduler, kubelet이 없다. -// 여기서는 우리 컨트롤러가 API 서버에 쓰는 것만 검증한다 (CLAUDE.md 테스트 3층). -// - ReplicaSet 입양·삭제·스케줄링은 일어나지 않으므로 오브젝트를 전부 손으로 만든다. -// - Pod 삭제는 kubelet이 없어 terminating에 머무니 deletionTimestamp로 단언한다. +// envtest has no kube-controller-manager, scheduler or kubelet. +// Only what our controller writes to the API server is verified here (CLAUDE.md, the three test layers). +// - ReplicaSet adoption, deletion and scheduling never happen, so every object is built by hand. +// - Pod deletion stays terminating without a kubelet, so it is asserted through deletionTimestamp. import ( "context" @@ -69,8 +69,8 @@ func uniq(prefix string) string { const testHash = "abc1234" -// conflictOnDelete는 판정과 삭제 사이에 Pod이 변해 preconditions가 거부되는 -// 교차를 흉내 낸다. +// conflictOnDelete simulates the interleaving where the Pod changes between the +// decision and the deletion, so the preconditions reject it. type conflictOnDelete struct{ client.Client } func (c *conflictOnDelete) Delete(ctx context.Context, obj client.Object, opts ...client.DeleteOption) error { @@ -78,7 +78,7 @@ func (c *conflictOnDelete) Delete(ctx context.Context, obj client.Object, opts . fmt.Errorf("the object has been modified")) } -// 롤아웃 스펙들이 템플릿을 이 이미지로 바꿔 세대를 밀어낸다 +// the rollout specs push the generation forward by changing the template to this image const rolledImage = "nginx:1.16" type fixture struct { @@ -108,7 +108,7 @@ func simpleContainer() []corev1.Container { return []corev1.Container{{Name: "app", Image: "nginx:1.15"}} } -// createWorkload는 Deployment → ReplicaSet → 타깃 Pod 사슬을 손으로 만든다. +// createWorkload builds the Deployment → ReplicaSet → target Pod chain by hand. func createWorkload(name, nodeName string) (*appsv1.Deployment, *appsv1.ReplicaSet, *corev1.Pod) { appLabels := map[string]string{"app": name} deploy := &appsv1.Deployment{ @@ -173,8 +173,8 @@ func setupFixture() *fixture { } } -// createReplacement는 스케줄까지 끝난 대체 Pod을 흉내 낸다. envtest에는 -// 스케줄러가 없어 nodeName을 생성 시점에 박는다. +// createReplacement simulates a replacement that is already scheduled. envtest has no +// scheduler, so nodeName is written at creation. func createReplacement(f *fixture, nodeName string, ready bool) *corev1.Pod { repl := &corev1.Pod{ ObjectMeta: metav1.ObjectMeta{ @@ -263,7 +263,7 @@ var _ = Describe("NodeReconciler", func() { _, err := f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) - // patch 하나로 hash가 붙고 replaces가 떨어져 소유가 ReplicaSet으로 넘어간다 + // one patch attaches the hash and removes replaces, passing ownership to the ReplicaSet got := getPod(repl.Name) Expect(got.Labels[LabelPodTemplateHash]).To(Equal(testHash)) Expect(got.Labels).NotTo(HaveKey(LabelReplaces)) @@ -273,7 +273,7 @@ var _ = Describe("NodeReconciler", func() { f := setupFixture() other := createNode(uniq("other-node"), nil) repl := createReplacement(f, other.Name, true) - // 롤아웃 중: replicas != updatedReplicas + // mid-rollout: replicas != updatedReplicas fresh := &appsv1.Deployment{} Expect(k8sClient.Get(ctx, client.ObjectKeyFromObject(f.deploy), fresh)).To(Succeed()) fresh.Status = appsv1.DeploymentStatus{ @@ -306,7 +306,7 @@ var _ = Describe("NodeReconciler", func() { Expect(got.Labels).NotTo(HaveKey(LabelPodTemplateHash)) Expect(f.rec.has("ReplacementOnDrainingNode")).To(BeTrue()) - // terminating은 있는 것으로 세지 않으므로 지운 라운드가 바로 새로 만든다 + // terminating does not count as existing, so the round that deleted it creates a new one at once var live int for _, p := range listReplacements(f.target.UID) { if p.DeletionTimestamp == nil { @@ -321,12 +321,12 @@ var _ = Describe("NodeReconciler", func() { other := createNode(uniq("other-node"), nil) repl := createReplacement(f, other.Name, false) - // 앉은 노드가 멀쩡하면 Ready를 기다린다 + // if the node it landed on is fine, wait for Ready _, err := f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) Expect(getPod(repl.Name).DeletionTimestamp).To(BeNil()) - // 그 노드에 drain이 걸리면 Ready를 기다릴 이유가 없다 + // once that node is draining there is no reason to wait for Ready patch := mergePatch(map[string]any{ "metadata": map[string]any{"labels": map[string]any{LabelDrain: "true"}}, }) @@ -343,7 +343,7 @@ var _ = Describe("NodeReconciler", func() { other := createNode(uniq("other-node"), nil) repl := createReplacement(f, other.Name, false) - // 이미지가 바뀌면 타깃의 RS는 현재 세대가 아니다 + // with the image changed the target's RS is not the current generation fresh := &appsv1.Deployment{} Expect(k8sClient.Get(ctx, client.ObjectKeyFromObject(f.deploy), fresh)).To(Succeed()) fresh.Spec.Template.Spec.Containers[0].Image = rolledImage @@ -354,7 +354,7 @@ var _ = Describe("NodeReconciler", func() { Expect(getPod(repl.Name).DeletionTimestamp).NotTo(BeNil()) Expect(f.rec.has("ReplacementSuperseded")).To(BeTrue()) - // stale한 동안은 다시 만들지 않는다 — 이주는 롤아웃의 몫이다 + // while stale, nothing is created again — the migration belongs to the rollout _, err = f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) var live int @@ -365,7 +365,7 @@ var _ = Describe("NodeReconciler", func() { } Expect(live).To(Equal(0)) - // 세대가 돌아오면 재개된다 + // it resumes once the generation comes back Expect(k8sClient.Get(ctx, client.ObjectKeyFromObject(f.deploy), fresh)).To(Succeed()) fresh.Spec.Template.Spec.Containers[0].Image = "nginx:1.15" Expect(k8sClient.Update(ctx, fresh)).To(Succeed()) @@ -394,7 +394,7 @@ var _ = Describe("NodeReconciler", func() { _, err := f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) - // paused면 롤아웃이 실제로 움직이지 않는다 — 대체는 평소처럼 유지된다 + // paused means the rollout does not actually move — replacements are maintained as usual Expect(getPod(repl.Name).DeletionTimestamp).To(BeNil()) Expect(f.rec.has("ReplacementSuperseded")).To(BeFalse()) }) @@ -408,7 +408,7 @@ var _ = Describe("NodeReconciler", func() { res, err := f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) - // 삭제가 거부된 낡은 명단이 넘기기로 흘러가지 않는다 — 라운드를 접고 재판정한다 + // a stale list whose deletion was rejected never reaches hand-over — the round ends and decides again got := getPod(repl.Name) Expect(got.DeletionTimestamp).To(BeNil()) Expect(got.Labels).NotTo(HaveKey(LabelPodTemplateHash)) @@ -429,8 +429,8 @@ var _ = Describe("NodeReconciler", func() { _, err := f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) - // Ready였어도 입양이 아니라 삭제다 — Healthy(D)가 롤아웃 내내 넘기기를 막으므로 - // 이 대체는 입양에 도달할 수 없는 Pod이다 + // deletion, not adoption, even if it was Ready — Healthy(D) blocks hand-over for the + // whole rollout, so this replacement can never reach adoption got := getPod(repl.Name) Expect(got.DeletionTimestamp).NotTo(BeNil()) Expect(got.Labels).NotTo(HaveKey(LabelPodTemplateHash)) @@ -442,10 +442,10 @@ var _ = Describe("NodeReconciler", func() { other := createNode(uniq("other-node"), nil) repl := createReplacement(f, other.Name, true) - // 삭제자(PodReconciler)가 읽어둔 시점의 복사본 + // the copy as the deleter (PodReconciler) read it stale := repl.DeepCopy() - // 그 사이 넘기기가 patch 하나로 hash를 붙이고 replaces를 뗀다 + // meanwhile the hand-over attaches the hash and removes replaces in one patch patch := mergePatch(map[string]any{ "metadata": map[string]any{"labels": map[string]any{ LabelPodTemplateHash: testHash, @@ -454,7 +454,7 @@ var _ = Describe("NodeReconciler", func() { }) Expect(k8sClient.Patch(ctx, repl, patch)).To(Succeed()) - // stale 복사본으로 삭제 시도 — resourceVersion precondition이 거부한다 + // deleting with the stale copy — the resourceVersion precondition rejects it deleted, err := deleteReplacement(ctx, k8sClient, stale) Expect(err).NotTo(HaveOccurred()) Expect(deleted).To(BeFalse()) @@ -467,15 +467,15 @@ var _ = Describe("NodeReconciler", func() { repl := createReplacement(f, other.Name, true) setDeployHealthy(f.deploy) - // 1라운드: 넘기기까지 간다 + // round 1: goes through hand-over _, err := f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) handed := getPod(repl.Name) Expect(handed.Labels[LabelPodTemplateHash]).To(Equal(testHash)) - // envtest에는 RS 컨트롤러가 없어 타깃이 지워지지 않는다 — 넘기는 순간 - // replicas가 올라 초과분이 증설분에 흡수된 것과 같은 상태다. - // 2라운드: "타깃은 그대로인데 대신할 Pod이 없다"를 보고 하나 더 만든다. + // envtest has no RS controller, so the target is not deleted — the same state as + // replicas going up at the moment of hand-over and the surplus being absorbed. + // round 2: sees "the target is still there and nothing stands in for it" and creates one more. _, err = f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) @@ -487,7 +487,7 @@ var _ = Describe("NodeReconciler", func() { } Expect(live).To(HaveLen(1)) Expect(live[0].Name).NotTo(Equal(repl.Name)) - // 넘긴 Pod은 건드리지 않았다 + // the handed-over Pod was left untouched Expect(getPod(repl.Name).Labels).NotTo(HaveKey(LabelReplaces)) }) @@ -502,7 +502,7 @@ var _ = Describe("NodeReconciler", func() { got := getNode(node.Name) Expect(got.Spec.Unschedulable).To(BeTrue()) Expect(got.Labels[LabelState]).To(Equal(StateComplete)) - // cordon은 여전히 우리 것 — 어노테이션은 라벨이 걷힐 때 함께 걷힌다 + // the cordon is still ours — the annotation comes off when the label does Expect(got.Annotations).To(HaveKeyWithValue(AnnotationCordoned, "true")) Expect(rec.has("DrainComplete")).To(BeTrue()) }) @@ -512,7 +512,7 @@ var _ = Describe("NodeReconciler", func() { r := &NodeReconciler{Client: k8sClient, Reader: k8sClient, Recorder: rec} node := createNode(uniq("done-node"), map[string]string{LabelDrain: "true"}) - // 빈 노드라 첫 라운드에 cordon과 Complete까지 간다 + // an empty node reaches cordon and Complete in the first round _, err := r.Reconcile(ctx, nodeReq(node)) Expect(err).NotTo(HaveOccurred()) got := getNode(node.Name) @@ -536,7 +536,7 @@ var _ = Describe("NodeReconciler", func() { _, err := f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) - // 사람이 uncordon — Cancelled로 접히며 어노테이션이 지워진다 + // a human uncordons — it folds into Cancelled and the annotation is deleted node := getNode(f.node.Name) node.Spec.Unschedulable = false Expect(k8sClient.Update(ctx, node)).To(Succeed()) @@ -544,7 +544,7 @@ var _ = Describe("NodeReconciler", func() { Expect(err).NotTo(HaveOccurred()) Expect(getNode(f.node.Name).Labels[LabelState]).To(Equal(StateCancelled)) - // 사람이 다른 이유로 다시 cordon한 뒤 라벨을 걷는다 + // a human re-cordons for another reason, then the label is removed node = getNode(f.node.Name) node.Spec.Unschedulable = true Expect(k8sClient.Update(ctx, node)).To(Succeed()) @@ -556,7 +556,7 @@ var _ = Describe("NodeReconciler", func() { Expect(err).NotTo(HaveOccurred()) got := getNode(f.node.Name) - // 우리 기록이 아닌 cordon은 걷지 않는다 + // a cordon that is not our record is not lifted Expect(got.Spec.Unschedulable).To(BeTrue()) Expect(got.Labels).NotTo(HaveKey(LabelState)) Expect(got.Annotations).NotTo(HaveKey(AnnotationCordoned)) @@ -564,13 +564,13 @@ var _ = Describe("NodeReconciler", func() { It("Complete latches: later targets are left alone", func() { f := setupFixture() - // Complete 상태를 만든다: 타깃을 먼저 지우고(0 grace로 즉시) reconcile + // reach Complete: delete the target first (0 grace, immediate) and reconcile Expect(k8sClient.Delete(ctx, f.target, client.GracePeriodSeconds(0))).To(Succeed()) _, err := f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) Expect(getNode(f.node.Name).Labels[LabelState]).To(Equal(StateComplete)) - // tolerate 워크로드가 뒤늦게 앉은 상황 + // a tolerating workload seated late _, _, late := createWorkload(uniq("late"), f.node.Name) _, err = f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) @@ -585,7 +585,7 @@ var _ = Describe("NodeReconciler", func() { _, err := f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) - // 사람이 uncordon + // a human uncordons node := getNode(f.node.Name) node.Spec.Unschedulable = false Expect(k8sClient.Update(ctx, node)).To(Succeed()) @@ -600,7 +600,7 @@ var _ = Describe("NodeReconciler", func() { Expect(getPod(f.target.Name).Annotations).NotTo(HaveKey(AnnotationPodDeletionCost)) Expect(f.rec.has("DrainCancelled")).To(BeTrue()) - // 래치: 다시 돌려도 cordon하지 않는다 + // latch: another round does not cordon again _, err = f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) Expect(getNode(f.node.Name).Spec.Unschedulable).To(BeFalse()) @@ -613,7 +613,7 @@ var _ = Describe("NodeReconciler", func() { Expect(err).NotTo(HaveOccurred()) Expect(getNode(f.node.Name).Labels[LabelState]).To(Equal(StateComplete)) - // 사람이 우리 cordon을 풀고 노드를 다시 쓰기로 함 + // a human lifts our cordon and decides to use the node again node := getNode(f.node.Name) node.Spec.Unschedulable = false Expect(k8sClient.Update(ctx, node)).To(Succeed()) @@ -626,10 +626,10 @@ var _ = Describe("NodeReconciler", func() { Expect(got.Spec.Unschedulable).To(BeFalse()) Expect(f.rec.has("DrainCancelled")).To(BeTrue()) - // Cancelled로 접혔으므로 착지 검사도 이 노드를 막지 않는다 + // folded into Cancelled, so the landing check does not block this node either Expect(drainActive(got)).To(BeFalse()) - // 래치: 다시 돌려도 cordon하지 않는다 + // latch: another round does not cordon again _, err = f.r.Reconcile(ctx, nodeReq(f.node)) Expect(err).NotTo(HaveOccurred()) Expect(getNode(f.node.Name).Spec.Unschedulable).To(BeFalse()) @@ -690,7 +690,7 @@ var _ = Describe("PodReconciler", func() { _, err := newReconciler().Reconcile(ctx, podReq(repl)) Expect(err).NotTo(HaveOccurred()) - // 스케줄 전 Pod은 kubelet 확인이 필요 없어 즉시 사라진다 + // an unscheduled Pod needs no kubelet confirmation, so it disappears immediately gone := k8sClient.Get(ctx, types.NamespacedName{Namespace: "default", Name: repl.Name}, &corev1.Pod{}) Expect(apierrors.IsNotFound(gone)).To(BeTrue()) }) diff --git a/internal/controller/node_controller.go b/internal/controller/node_controller.go index 73a98e0..77e03c5 100644 --- a/internal/controller/node_controller.go +++ b/internal/controller/node_controller.go @@ -35,15 +35,15 @@ import ( "sigs.k8s.io/controller-runtime/pkg/reconcile" ) -// NodeReconciler는 drain 라벨이 붙은 노드를 비운다 (DESIGN.md "컨트롤러가 하는 일"). +// NodeReconciler empties nodes carrying the drain label (DESIGN.md "What the controller does"). type NodeReconciler struct { client.Client - // Reader는 API 서버를 직접 읽는다. 캐시는 다시 볼 때를 알려주는 데만 쓴다. + // Reader reads straight from the API server. The cache only tells us when to look again. Reader client.Reader Recorder events.EventRecorder } -// target은 그 노드 위에서 owner가 ReplicaSet이고 그 ReplicaSet의 owner가 Deployment인 Pod이다. +// target is a Pod on the node whose owner is a ReplicaSet whose owner is a Deployment. type target struct { pod *corev1.Pod rs *appsv1.ReplicaSet @@ -66,8 +66,8 @@ func (r *NodeReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl. return r.drain(ctx, node) } -// restore는 drain 라벨이 사라진 노드에서 우리가 남긴 것을 걷는다. -// 대체 Pod은 PodReconciler의 판정(타깃이 drain 노드에 없으면 지운다)이 걷는다. +// restore takes back what we left on a node whose drain label is gone. +// Replacements are collected by PodReconciler's rule: delete unless the target is on a draining node. func (r *NodeReconciler) restore(ctx context.Context, node *corev1.Node) error { if node.Labels[LabelState] == "" && node.Annotations[AnnotationCordoned] == "" { return nil @@ -94,7 +94,7 @@ func (r *NodeReconciler) restore(ctx context.Context, node *corev1.Node) error { return nil } -// sweepDeletionCost는 노드 위 Pod에서 우리가 박은 pod-deletion-cost를 걷는다. +// sweepDeletionCost removes the pod-deletion-cost we wrote on the node's Pods. func (r *NodeReconciler) sweepDeletionCost(ctx context.Context, nodeName string) error { pods := &corev1.PodList{} if err := r.Reader.List(ctx, pods, client.MatchingFields{"spec.nodeName": nodeName}); err != nil { @@ -102,7 +102,7 @@ func (r *NodeReconciler) sweepDeletionCost(ctx context.Context, nodeName string) } for i := range pods.Items { pod := &pods.Items[i] - // 값이 정확히 우리 것일 때만 지운다 + // delete it only when the value is exactly ours if pod.Annotations[AnnotationPodDeletionCost] != PodDeletionCost { continue } @@ -116,10 +116,10 @@ func (r *NodeReconciler) sweepDeletionCost(ctx context.Context, nodeName string) return nil } -// cancel은 drain 중 uncordon된 노드에서 손을 뗀다. cordon은 이 방식의 전제라 -// 전제가 사라지면 계속할 의미가 없다. 다시 cordon해서 사람과 싸우지 않는다. -// 어노테이션도 함께 지운다 — 기록된 cordon은 사람 손에 이미 풀렸으므로, 이후 -// 사람이 새로 건 cordon을 restore가 우리 것으로 오인해 풀면 안 된다. +// cancel lets go of a node uncordoned mid-drain. The cordon is the premise of this +// approach, and without it there is no point continuing. We do not re-cordon and fight +// the human. The annotation goes too — the cordon it recorded is already gone by their +// hand, and a cordon they place later must not be mistaken for ours by restore. func (r *NodeReconciler) cancel(ctx context.Context, node *corev1.Node) error { if err := r.sweepDeletionCost(ctx, node.Name); err != nil { return err @@ -142,26 +142,26 @@ func (r *NodeReconciler) cancel(ctx context.Context, node *corev1.Node) error { func (r *NodeReconciler) drain(ctx context.Context, node *corev1.Node) (ctrl.Result, error) { log := logf.FromContext(ctx) - // Cancelled는 래치다. drain 라벨이 걷힐 때까지 관여하지 않는다. + // Cancelled is a latch. Nothing happens until the drain label is removed. if node.Labels[LabelState] == StateCancelled { return ctrl.Result{}, nil } - // InProgress와 Complete는 cordon을 확인한 뒤에만 붙는다. 그런데 unschedulable이 - // 아니라면 누군가 uncordon한 것이다 — 진행 중이면 종료를 보장하던 전제가 - // 사라졌고, 끝난 뒤면 사람이 우리 cordon을 풀고 노드를 다시 쓰기로 한 - // 것이다. 어느 쪽이든 관여를 접는다. + // InProgress and Complete are only ever set after confirming the cordon. If the node + // is no longer unschedulable, someone uncordoned it — mid-drain the premise that + // guaranteed termination is gone, and after completion a human lifted our cordon and + // decided to use the node again. Either way we let go. state := node.Labels[LabelState] if (state == StateInProgress || state == StateComplete) && !node.Spec.Unschedulable { return ctrl.Result{}, r.cancel(ctx, node) } - // Complete는 래치다. cordon이 유지되는 동안 관여하지 않는다. + // Complete is a latch. While the cordon holds, we do not act. if state == StateComplete { return ctrl.Result{}, nil } - // 1. 노드 마킹 — 우리가 실제로 값을 바꿨을 때만 어노테이션을 단다 + // 1. marking the node — the annotation is written only when we actually changed the value if !node.Spec.Unschedulable { patch := mergePatch(map[string]any{ "spec": map[string]any{"unschedulable": true}, @@ -178,7 +178,7 @@ func (r *NodeReconciler) drain(ctx context.Context, node *corev1.Node) (ctrl.Res return ctrl.Result{}, err } - // 6. 완료 — terminating 타깃도 센다 + // 6. completion — terminating targets count too if len(targets) == 0 { return ctrl.Result{}, r.complete(ctx, node) } @@ -205,8 +205,8 @@ func (r *NodeReconciler) drain(ctx context.Context, node *corev1.Node) (ctrl.Res return ctrl.Result{}, err } if retry { - // 판정과 삭제 사이에 세상이 바뀌었다. 낡은 명단으로 넘기기까지 가지 않고 - // 라운드를 접는다. 다음 라운드가 새 상태에서 처음부터 판정한다. + // The world changed between the decision and the deletion. Rather than carry a stale + // list into hand-over, end the round. The next round decides again from scratch. return ctrl.Result{RequeueAfter: time.Second}, nil } if err := r.handOver(ctx, node, targets, replsByUID); err != nil { @@ -216,8 +216,8 @@ func (r *NodeReconciler) drain(ctx context.Context, node *corev1.Node) (ctrl.Res return ctrl.Result{RequeueAfter: 15 * time.Second}, nil } -// complete는 Complete를 붙인다. cordoned-by-controller 어노테이션은 그대로 둔다 — -// cordon은 여전히 우리가 건 것이고, drain 라벨이 걷힐 때 restore가 함께 걷는다. +// complete sets Complete. The cordoned-by-controller annotation stays — +// the cordon is still ours, and restore takes it back when the drain label goes. func (r *NodeReconciler) complete(ctx context.Context, node *corev1.Node) error { if node.Labels[LabelState] == StateComplete { return nil @@ -236,7 +236,7 @@ func (r *NodeReconciler) complete(ctx context.Context, node *corev1.Node) error return nil } -// markTargets는 타깃에 pod-deletion-cost를 박는다 (DESIGN.md 2단계). +// markTargets writes pod-deletion-cost on the targets (DESIGN.md step 2). func (r *NodeReconciler) markTargets(ctx context.Context, targets []target) error { for _, t := range targets { if t.pod.Annotations[AnnotationPodDeletionCost] == PodDeletionCost { @@ -252,24 +252,24 @@ func (r *NodeReconciler) markTargets(ctx context.Context, targets []target) erro return nil } -// reconcileReplacements는 있어야 할 집합과 있는 집합을 맞춘다 (DESIGN.md 3단계). -// 모자라면 만들고, 같은 타깃에 남으면 지운다. drain 중인 노드에 앉았거나 -// 롤아웃에 밀린 대체도 여기서 지운다 — 어느 쪽도 입양에 도달할 수 없다. -// 그 삭제가 preconditions에 거부되면 retry를 돌려 라운드를 접는다 — 낡은 -// 명단이 넘기기까지 흘러가지 않게 하고, 다음 라운드가 새 상태에서 판정한다. +// reconcileReplacements matches the set that should exist against the set that does +// (DESIGN.md step 3). Too few, create; a surplus on the same target, delete. Replacements +// seated on a draining node or superseded by a rollout are deleted here too — neither can +// reach adoption. If such a deletion is rejected by its preconditions, a retry ends the +// round, keeping a stale list out of hand-over so the next round decides from fresh state. func (r *NodeReconciler) reconcileReplacements(ctx context.Context, node *corev1.Node, targets []target, replsByUID map[string][]*corev1.Pod) (retry bool, err error) { log := logf.FromContext(ctx) drainingNodes := map[string]bool{node.Name: true} deploys := map[types.NamespacedName]*appsv1.Deployment{} for _, t := range targets { - // terminating 타깃은 ReplicaSet이 이미 스스로 대체를 만들고 있다. - // 남아 있던 대체 Pod은 회수 경로(PodReconciler)가 같은 판정으로 지운다. + // A terminating target already has the ReplicaSet making its own replacement. + // Any replacement left over is deleted by the reclamation path (PodReconciler), same rule. if t.pod.DeletionTimestamp != nil { continue } uid := string(t.pod.UID) - // drain 중인 노드에 앉은 대체는 Ready 여부와 무관하게 지운다. + // A replacement seated on a draining node is deleted regardless of readiness. var kept []*corev1.Pod for _, repl := range replsByUID[uid] { landed, err := r.nodeDraining(ctx, drainingNodes, repl.Spec.NodeName) @@ -296,8 +296,8 @@ func (r *NodeReconciler) reconcileReplacements(ctx context.Context, node *corev1 } replsByUID[uid] = kept - // 타깃의 이주를 롤아웃이 대신 수행 중이면 대체를 지우고, - // 세대가 돌아올 때까지 만들지 않는다. + // If a rollout is performing the target's migration instead, delete the replacement + // and create none until the generation comes back. superseded, err := r.supersededByRollout(ctx, deploys, t.rs) if err != nil { return false, err @@ -336,8 +336,8 @@ func (r *NodeReconciler) reconcileReplacements(ctx context.Context, node *corev1 log.Info("Created replacement Pod", "pod", repl.Namespace+"/"+repl.Name, "target", t.pod.Namespace+"/"+t.pod.Name, "node", node.Name) case len(kept) > 1: - // 제일 오래된 하나만 남긴다. 트림은 이 라운드의 넘기기가 방금 지운 - // Pod을 다시 보지 않게 하기 위한 것이다. + // Keep only the oldest one. The trim exists so this round's hand-over does not + // look again at a Pod it just deleted. for _, extra := range kept[1:] { if _, err := deleteReplacement(ctx, r.Client, extra); err != nil { return false, err @@ -349,8 +349,8 @@ func (r *NodeReconciler) reconcileReplacements(ctx context.Context, node *corev1 return false, nil } -// handOver는 Ready인 대체 Pod에 hash를 붙여 ReplicaSet이 데려가게 한다 (DESIGN.md 4단계). -// Deployment마다 따로 판정하고 준비된 것부터 넘긴다. +// handOver attaches the hash to a Ready replacement so the ReplicaSet takes it (DESIGN.md step 4). +// The decision is per Deployment, and whichever is ready hands over first. func (r *NodeReconciler) handOver(ctx context.Context, node *corev1.Node, targets []target, replsByUID map[string][]*corev1.Pod) error { log := logf.FromContext(ctx) healthyDeploys := map[types.NamespacedName]bool{} @@ -358,7 +358,7 @@ func (r *NodeReconciler) handOver(ctx context.Context, node *corev1.Node, target if t.pod.DeletionTimestamp != nil { continue } - // drain 중인 노드에 앉은 대체는 reconcileReplacements가 이미 지웠다 + // replacements seated on a draining node were already deleted by reconcileReplacements existing := replsByUID[string(t.pod.UID)] if len(existing) == 0 { continue @@ -376,15 +376,15 @@ func (r *NodeReconciler) handOver(ctx context.Context, node *corev1.Node, target continue } - // hash는 타깃의 ReplicaSet에서 읽는다. 롤아웃 중이면 Deployment의 현재 RS가 아닐 수 있다. + // The hash is read from the target's ReplicaSet; during a rollout that may not be the current RS. hash := t.rs.Labels[LabelPodTemplateHash] if hash == "" { - // Deployment가 만든 RS에는 항상 hash 라벨이 있다. 여기 오면 비정상이므로 흔적을 남긴다. + // An RS created by a Deployment always has the hash label. Reaching here is abnormal, so leave a trace. log.Info("Skipped handover because ReplicaSet has no pod-template-hash label", "replicaset", t.rs.Namespace+"/"+t.rs.Name, "target", t.pod.Namespace+"/"+t.pod.Name) continue } - // patch 하나로 hash를 붙이고 replaces를 뗀다. 이 순간 소유가 ReplicaSet으로 넘어간다. + // One patch attaches the hash and removes replaces. Ownership passes to the ReplicaSet here. patch := mergePatch(map[string]any{ "metadata": map[string]any{"labels": map[string]any{ LabelPodTemplateHash: hash, @@ -470,18 +470,18 @@ func (r *NodeReconciler) nodeDraining(ctx context.Context, cache map[string]bool } return false, err } - // Cancelled 노드는 열린 보통 노드라 착지해도 된다. Complete 노드는 사람이 - // 리부팅하러 갈 노드라 여전히 막는다. + // A Cancelled node is an ordinary open node, so landing there is fine. A Complete node + // is one a human is about to reboot, so it stays banned. cache[name] = drainActive(n) return cache[name], nil } -// supersededByRollout는 타깃의 이주를 롤아웃이 대신 수행 중인지 본다 (DESIGN.md 3단계). -// paused면 템플릿이 달라도 롤아웃이 실제로 움직이지 않으므로 해당하지 않는다. +// supersededByRollout reports whether a rollout is performing the target's migration (DESIGN.md step 3). +// A paused rollout does not actually move, so it does not apply even if the templates differ. func (r *NodeReconciler) supersededByRollout(ctx context.Context, cache map[types.NamespacedName]*appsv1.Deployment, rs *appsv1.ReplicaSet) (bool, error) { ref := metav1.GetControllerOf(rs) if ref == nil { - // collectTargets가 ownedByDeployment를 통과한 RS만 넘기므로 도달하지 않는다 + // unreachable: collectTargets only passes RSes that cleared ownedByDeployment return false, nil } key := types.NamespacedName{Namespace: rs.Namespace, Name: ref.Name} @@ -490,7 +490,7 @@ func (r *NodeReconciler) supersededByRollout(ctx context.Context, cache map[type d = &appsv1.Deployment{} if err := r.Reader.Get(ctx, key, d); err != nil { if apierrors.IsNotFound(err) { - // Deployment가 사라지면 타깃도 곧 사라진다. 회수는 그 경로가 한다 + // If the Deployment is gone the targets go soon after; that path does the reclamation d = nil } else { return false, err @@ -507,7 +507,7 @@ func (r *NodeReconciler) supersededByRollout(ctx context.Context, cache map[type func (r *NodeReconciler) deployHealthy(ctx context.Context, cache map[types.NamespacedName]bool, rs *appsv1.ReplicaSet) (bool, error) { ref := metav1.GetControllerOf(rs) if ref == nil { - // collectTargets가 ownedByDeployment를 통과한 RS만 넘기므로 도달하지 않는다 + // unreachable: collectTargets only passes RSes that cleared ownedByDeployment return false, nil } key := types.NamespacedName{Namespace: rs.Namespace, Name: ref.Name} @@ -540,8 +540,8 @@ func (r *NodeReconciler) SetupWithManager(mgr ctrl.Manager) error { Complete(r) } -// nodesForPod은 Pod 이벤트를 깨어날 노드로 바꾼다. 대체 Pod은 다른 노드에 떠 있어서 -// 어느 drain의 것인지 Pod만으로는 알 수 없으므로 drain 중인 노드 전부를 깨운다. +// nodesForPod turns a Pod event into the nodes to wake. A replacement lives on another +// node, and the Pod alone does not say which drain it belongs to, so every draining node wakes. func (r *NodeReconciler) nodesForPod(ctx context.Context, obj client.Object) []reconcile.Request { pod, ok := obj.(*corev1.Pod) if !ok { @@ -551,7 +551,7 @@ func (r *NodeReconciler) nodesForPod(ctx context.Context, obj client.Object) []r if pod.Labels[LabelReplaces] != "" { nodes := &corev1.NodeList{} if err := r.List(ctx, nodes, client.MatchingLabels{LabelDrain: "true"}); err != nil { - // 이 경로가 실패하면 진행이 주기적 requeue에만 의존하므로 흔적을 남긴다 + // If this path fails, progress depends on the periodic requeue alone, so leave a trace log.Error(err, "Failed to list draining nodes for replacement Pod event", "pod", pod.Namespace+"/"+pod.Name) return nil diff --git a/internal/controller/pod_controller.go b/internal/controller/pod_controller.go index 4506097..ae7a4c2 100644 --- a/internal/controller/pod_controller.go +++ b/internal/controller/pod_controller.go @@ -31,12 +31,13 @@ import ( "sigs.k8s.io/controller-runtime/pkg/predicate" ) -// PodReconciler는 대체 Pod을 키로 하는 회수 경로다 (DESIGN.md 3단계). -// ReplicaSet prune으로 타깃이 사라지면 노드 순회로는 대체 Pod을 쳐다볼 일이 없어서 -// 대체 Pod 자체에서 출발하는 경로가 따로 있어야 한다. 판정은 같다. +// PodReconciler is the reclamation path keyed on the replacement Pod (DESIGN.md step 3). +// When a pruned ReplicaSet takes the targets with it, the node traversal never looks at +// the replacements again, so a path starting from the replacement itself is required. +// The decision is the same. type PodReconciler struct { client.Client - // Reader는 API 서버를 직접 읽는다. + // Reader reads straight from the API server. Reader client.Reader } @@ -53,7 +54,7 @@ func (r *PodReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.R if uid == "" || pod.DeletionTimestamp != nil { return ctrl.Result{}, nil } - // controller ownerRef가 있는 Pod은 지우지 않는다 + // Pods with a controller ownerRef are never deleted if metav1.GetControllerOf(pod) != nil { return ctrl.Result{}, nil } @@ -63,7 +64,7 @@ func (r *PodReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.R return ctrl.Result{}, err } if needed { - // 취소나 타깃 eviction은 이 Pod의 이벤트를 만들지 않으므로 주기적으로 다시 판정한다 + // Cancellation and target eviction produce no event for this Pod, so re-decide periodically return ctrl.Result{RequeueAfter: 30 * time.Second}, nil } deleted, err := deleteReplacement(ctx, r.Client, pod) @@ -76,9 +77,9 @@ func (r *PodReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.R return ctrl.Result{}, nil } -// replacementNeeded는 타깃이 살아서 drain 중인 노드에 있는지 본다. +// replacementNeeded reports whether the target is alive and on a draining node. func (r *PodReconciler) replacementNeeded(ctx context.Context, repl *corev1.Pod, uid types.UID) (bool, error) { - // 죽은 대체 Pod은 Ready가 될 수도 입양될 수도 없다 + // A dead replacement can neither become Ready nor be adopted if repl.Status.Phase == corev1.PodFailed || repl.Status.Phase == corev1.PodSucceeded { return false, nil } @@ -109,7 +110,7 @@ func (r *PodReconciler) replacementNeeded(ctx context.Context, repl *corev1.Pod, } return false, err } - // Cancelled 노드의 drain은 끝난 것이다 — 대체 Pod을 유지할 이유가 없다 + // A Cancelled node's drain is over — no reason to keep the replacement return drainActive(node), nil } diff --git a/internal/controller/softdrain.go b/internal/controller/softdrain.go index 67db948..38f4018 100644 --- a/internal/controller/softdrain.go +++ b/internal/controller/softdrain.go @@ -39,7 +39,7 @@ const ( AnnotationCordoned = "soft-drain.com/cordoned-by-controller" AnnotationPodDeletionCost = "controller.kubernetes.io/pod-deletion-cost" - // int32 최솟값. 이 값을 쓰는 게 우리뿐이라 값이 이것이면 우리가 붙인 것이다. + // The int32 minimum. Nothing else writes this value, so a Pod carrying it is ours. PodDeletionCost = "-2147483648" LabelPodTemplateHash = appsv1.DefaultDeploymentUniqueLabelKey @@ -53,13 +53,13 @@ func draining(node *corev1.Node) bool { return node.Labels[LabelDrain] == "true" } -// drainActive는 그 노드의 drain이 실제로 진행 중인지 본다. Cancelled 노드는 -// uncordon된 보통 노드다 — 대체 Pod을 유지할 이유도, 착지를 막을 이유도 없다. +// drainActive reports whether the node's drain is actually running. A Cancelled node +// is an ordinary uncordoned node — no reason to keep replacements or to ban landings. func drainActive(node *corev1.Node) bool { return draining(node) && node.Labels[LabelState] != StateCancelled } -// replicaSetRef는 Pod의 controller가 apps/v1 ReplicaSet일 때 그 참조를 준다. +// replicaSetRef returns the reference when the Pod's controller is an apps/v1 ReplicaSet. func replicaSetRef(pod *corev1.Pod) *metav1.OwnerReference { ref := metav1.GetControllerOf(pod) if ref == nil || ref.Kind != "ReplicaSet" || ref.APIVersion != "apps/v1" { @@ -73,7 +73,7 @@ func ownedByDeployment(rs *appsv1.ReplicaSet) bool { return ref != nil && ref.Kind == "Deployment" && ref.APIVersion == "apps/v1" } -// validReplacement는 DESIGN.md 3단계의 "있는 것" 판정이다. +// validReplacement is the "does exist" test of DESIGN.md step 3. func validReplacement(pod *corev1.Pod) bool { return pod.Labels[LabelReplaces] != "" && metav1.GetControllerOf(pod) == nil && @@ -91,16 +91,16 @@ func podReady(pod *corev1.Pod) bool { return false } -// deploymentHealthy는 DESIGN.md 4단계의 Healthy(D) 판정이다. -// replicas == updatedReplicas 항이 "Pod을 가진 ReplicaSet이 하나뿐"을 잡는다. +// deploymentHealthy is the Healthy(D) test of DESIGN.md step 4. +// The replicas == updatedReplicas term catches "only one ReplicaSet has Pods". func deploymentHealthy(d *appsv1.Deployment) bool { return d.Status.ObservedGeneration >= d.Generation && d.Status.Replicas == d.Status.UpdatedReplicas && d.Status.AvailableReplicas >= ptr.Deref(d.Spec.Replicas, 1) } -// templatesEqualIgnoreHash는 Deployment 컨트롤러의 EqualIgnoreHash와 같은 비교다. -// pod-template-hash 라벨만 빼고 같으면 rs는 그 Deployment의 현재 세대다. +// templatesEqualIgnoreHash is the same comparison as the Deployment controller's EqualIgnoreHash. +// Equal but for the pod-template-hash label means rs is that Deployment's current generation. func templatesEqualIgnoreHash(a, b *corev1.PodTemplateSpec) bool { a2, b2 := a.DeepCopy(), b.DeepCopy() delete(a2.Labels, LabelPodTemplateHash) @@ -108,8 +108,8 @@ func templatesEqualIgnoreHash(a, b *corev1.PodTemplateSpec) bool { return apiequality.Semantic.DeepEqual(a2, b2) } -// buildReplacement는 rs.spec.template에서 대체 Pod을 만든다. -// 살아 있는 Pod을 베끼면 nodeName과 webhook이 넣은 사이드카가 따라온다. +// buildReplacement builds a replacement Pod from rs.spec.template. +// Copying the living Pod would bring nodeName and webhook-injected sidecars along. func buildReplacement(rs *appsv1.ReplicaSet, targetUID types.UID) *corev1.Pod { tpl := rs.Spec.Template.DeepCopy() @@ -117,13 +117,13 @@ func buildReplacement(rs *appsv1.ReplicaSet, targetUID types.UID) *corev1.Pod { if labels == nil { labels = map[string]string{} } - // rs.spec.template.metadata.labels에는 hash가 이미 들어 있다. - // hash가 있으면 ReplicaSet이 Pending인 Pod을 데려가서 그 Pod부터 지운다. + // rs.spec.template.metadata.labels already carries the hash. + // With the hash the ReplicaSet adopts a Pending Pod and deletes that one first. delete(labels, LabelPodTemplateHash) labels[LabelReplaces] = string(targetUID) - // pod-deletion-cost는 타깃에만 쓴다. 대체 Pod에 남으면 입양 후 - // 다음 스케일다운마다 그 Pod이 1순위로 죽는다. + // pod-deletion-cost is for targets only. Left on a replacement, that Pod dies first + // in every scale-down after adoption. return &corev1.Pod{ ObjectMeta: metav1.ObjectMeta{ GenerateName: rs.Name + "-", @@ -147,14 +147,14 @@ func sortPodsByAge(pods []*corev1.Pod) { func mergePatch(doc map[string]any) client.Patch { raw, err := json.Marshal(doc) if err != nil { - panic(err) // map[string]any 리터럴만 들어오므로 도달 불가 + panic(err) // unreachable: only map[string]any literals are passed in } return client.RawPatch(types.MergePatchType, raw) } -// deleteReplacement는 읽었던 UID와 resourceVersion을 preconditions로 걸어 지운다. -// 판정과 삭제 사이에 hash가 붙어 ReplicaSet이 데려간 Pod이면 resourceVersion이 -// 바뀌어 삭제가 거부된다(deleted=false). 다음 라운드가 다시 판정한다. +// deleteReplacement deletes with the UID and resourceVersion we read as preconditions. +// If the hash was attached in between and a ReplicaSet took the Pod, the resourceVersion +// changed and the deletion is rejected (deleted=false). The next round decides again. func deleteReplacement(ctx context.Context, c client.Writer, pod *corev1.Pod) (deleted bool, err error) { err = c.Delete(ctx, pod, client.Preconditions{UID: &pod.UID, ResourceVersion: &pod.ResourceVersion}) switch { diff --git a/internal/controller/softdrain_test.go b/internal/controller/softdrain_test.go index beeba32..fb6f378 100644 --- a/internal/controller/softdrain_test.go +++ b/internal/controller/softdrain_test.go @@ -71,14 +71,14 @@ func TestBuildReplacement(t *testing.T) { if pod.Labels[LabelReplaces] != "3f2a-uid" { t.Errorf("replaces label = %q", pod.Labels[LabelReplaces]) } - // cost가 대체 Pod에 남으면 입양 후 다음 스케일다운마다 그 Pod이 1순위로 죽는다 + // cost left on a replacement makes that Pod die first in every scale-down after adoption if _, ok := pod.Annotations[AnnotationPodDeletionCost]; ok { t.Error("replacement must not carry pod-deletion-cost") } if len(pod.Spec.Containers) != 1 || pod.Spec.Containers[0].Image != "nginx:1.15" { t.Errorf("spec not copied from rs template: %+v", pod.Spec) } - // 원본 템플릿이 오염되면 다음 라운드가 hash 없는 템플릿으로 판정한다 + // a polluted source template makes the next round decide on a template without the hash if rs.Spec.Template.Labels[LabelPodTemplateHash] != fixtureHash { t.Error("buildReplacement must not mutate rs.Spec.Template") } @@ -114,10 +114,10 @@ func TestDeploymentHealthy(t *testing.T) { }{ {"healthy", func(d *appsv1.Deployment) {}, true}, {"stale observedGeneration", func(d *appsv1.Deployment) { d.Status.ObservedGeneration = 2 }, false}, - // maxUnavailable: 0 롤아웃은 이 항으로만 잡힌다 + // a maxUnavailable: 0 rollout is caught by this term alone {"rollout in progress", func(d *appsv1.Deployment) { d.Status.Replicas = 3 }, false}, {"below desired availability", func(d *appsv1.Deployment) { d.Status.AvailableReplicas = 1 }, false}, - // >= 경계 — ==로 잘못 조이는 회귀를 잡는다 + // the >= boundary — catches a regression that wrongly tightens it to == {"newer observedGeneration", func(d *appsv1.Deployment) { d.Status.ObservedGeneration = 4 }, true}, {"more available than desired", func(d *appsv1.Deployment) { d.Status.AvailableReplicas = 3 }, true}, {"nil replicas defaults to 1", func(d *appsv1.Deployment) { @@ -161,7 +161,7 @@ func TestValidReplacement(t *testing.T) { APIVersion: "apps/v1", Kind: "ReplicaSet", Name: "rs", Controller: ptr.To(true), }} }, false}, - // controller가 아닌 ownerRef는 입양이 아니다 — 판정 기준은 controller ownerRef뿐 + // a non-controller ownerRef is not adoption — only a controller ownerRef counts {"non-controller ownerRef", func(p *corev1.Pod) { p.OwnerReferences = []metav1.OwnerReference{{ APIVersion: "apps/v1", Kind: "ReplicaSet", Name: "rs", Controller: ptr.To(false), @@ -194,7 +194,7 @@ func TestPodReady(t *testing.T) { if !podReady(pod) { t.Error("PodReady=True must be ready") } - // 노드가 NotReady로 빠지면 kubelet이 실제로 Unknown을 만든다 + // when a node goes NotReady the kubelet really does produce Unknown pod.Status.Conditions[1].Status = corev1.ConditionUnknown if podReady(pod) { t.Error("PodReady=Unknown must not be ready") @@ -289,7 +289,7 @@ func TestMergePatch(t *testing.T) { if err != nil { t.Fatal(err) } - // nil이 JSON null로 직렬화되어야 merge patch의 키 삭제가 동작한다 + // nil must serialize to JSON null for the merge patch to delete the key want := `{"metadata":{"labels":{"keep":"v","remove":null}}}` if string(data) != want { t.Errorf("patch = %s, want %s", data, want) @@ -308,9 +308,9 @@ func TestDrainActive(t *testing.T) { {"no labels", nil, false}, {"draining", map[string]string{LabelDrain: "true"}, true}, {"in progress", map[string]string{LabelDrain: "true", LabelState: StateInProgress}, true}, - // Complete 노드는 사람이 리부팅하러 갈 노드라 여전히 막는다 + // a Complete node is one a human is about to reboot, so it stays banned {"complete", map[string]string{LabelDrain: "true", LabelState: StateComplete}, true}, - // Cancelled 노드는 uncordon된 보통 노드다 + // a Cancelled node is an ordinary uncordoned node {"cancelled", map[string]string{LabelDrain: "true", LabelState: StateCancelled}, false}, {"wrong label value", map[string]string{LabelDrain: "false"}, false}, } @@ -352,12 +352,12 @@ func TestTemplatesEqualIgnoreHash(t *testing.T) { mutate func(rsTpl *corev1.PodTemplateSpec) want bool }{ - // rs 템플릿에는 hash가 붙어 있어도 같은 세대다 + // the same generation even though the rs template carries the hash {"same generation", func(*corev1.PodTemplateSpec) {}, true}, {"image changed", func(tpl *corev1.PodTemplateSpec) { tpl.Spec.Containers[0].Image = "nginx:1.16" }, false}, - // rollout restart는 템플릿 어노테이션(restartedAt)만 바꾼다 + // rollout restart changes only the template annotation (restartedAt) {"rollout restart", func(tpl *corev1.PodTemplateSpec) { tpl.Annotations = map[string]string{"kubectl.kubernetes.io/restartedAt": "2026-08-16T00:00:00Z"} }, false}, @@ -373,7 +373,7 @@ func TestTemplatesEqualIgnoreHash(t *testing.T) { if got := templatesEqualIgnoreHash(deployTpl(), rsTpl); got != tt.want { t.Errorf("templatesEqualIgnoreHash = %v, want %v", got, tt.want) } - // 비교가 원본을 오염시키면 이후 라운드가 다른 판정을 한다 + // if the comparison pollutes the source, later rounds decide differently if rsTpl.Labels[LabelPodTemplateHash] != fixtureHash { t.Error("comparison must not mutate its inputs") } diff --git a/internal/controller/suite_test.go b/internal/controller/suite_test.go index 0b86ea7..c15a2d4 100644 --- a/internal/controller/suite_test.go +++ b/internal/controller/suite_test.go @@ -48,7 +48,7 @@ var ( ) func TestControllers(t *testing.T) { - // envtest 바이너리가 없으면 유닛 테스트만 돈다. envtest는 make test로. + // Without the envtest binaries only the unit tests run. Use make test for envtest. if os.Getenv("KUBEBUILDER_ASSETS") == "" && getFirstFoundEnvTestBinaryDir() == "" { t.Skip("envtest binaries not found; run via make test") } diff --git a/test/e2e/e2e_suite_test.go b/test/e2e/e2e_suite_test.go index 65664e0..5db59d1 100644 --- a/test/e2e/e2e_suite_test.go +++ b/test/e2e/e2e_suite_test.go @@ -53,16 +53,16 @@ func TestE2E(t *testing.T) { } var _ = BeforeSuite(func() { - // 실 클러스터 보호 1: 스위트 전용 kubeconfig를 만들어 프로세스 전체에 고정한다. - // 사용자 kubeconfig의 current-context에 의존하면, 실행 도중 다른 셸에서 - // 컨텍스트를 바꿨을 때 deploy/undeploy가 그 클러스터로 나간다. + // Real-cluster guard 1: build a suite-owned kubeconfig and pin it for the whole process. + // Relying on the user kubeconfig's current-context means a context switched from another + // shell mid-run would send deploy/undeploy to that cluster. kubeconfig := filepath.Join(GinkgoT().TempDir(), "kubeconfig") _, err := utils.Run(exec.Command("kind", "export", "kubeconfig", "--name", kindClusterName(), "--kubeconfig", kubeconfig)) Expect(err).NotTo(HaveOccurred(), "Failed to export the kind cluster kubeconfig") os.Setenv("KUBECONFIG", kubeconfig) - // 실 클러스터 보호 2: 고정한 kubeconfig가 정말 전용 kind 클러스터를 가리키는지 확인 + // Real-cluster guard 2: verify the pinned kubeconfig really points at the dedicated kind cluster out, err := utils.Run(exec.Command("kubectl", "config", "current-context")) Expect(err).NotTo(HaveOccurred(), "Failed to read current kubectl context") Expect(strings.TrimSpace(out)).To(Equal("kind-"+kindClusterName()), diff --git a/test/e2e/e2e_test.go b/test/e2e/e2e_test.go index 57d65c0..d3bfd73 100644 --- a/test/e2e/e2e_test.go +++ b/test/e2e/e2e_test.go @@ -19,9 +19,9 @@ limitations under the License. package e2e -// e2e는 kcm과 스케줄러가 실제로 도는 유일한 층이다. 여기서 처음으로 -// ReplicaSet 입양과 초과분 삭제까지 포함한 전체 루프의 수렴을 본다. -// 타이밍 경합 없이 결정론적으로 유도할 수 있는 시나리오만 담는다. +// e2e is the only layer where the kcm and the scheduler actually run. This is where the +// full loop — ReplicaSet adoption and surplus deletion included — is seen converging. +// Only scenarios that can be induced deterministically, without timing races, live here. import ( "fmt" @@ -59,14 +59,14 @@ func applyYAML(y string) { type workload struct { name string - replicas int // 0이면 1 - pin string // 비어 있지 않으면 그 노드에 nodeSelector로 고정 - recreate bool // Recreate 전략 (롤아웃이 옛 Pod을 즉시 지운다) + replicas int // 0 means 1 + pin string // non-empty pins to that node with a nodeSelector + recreate bool // Recreate strategy (the rollout deletes the old Pod at once) tolerate bool // node.kubernetes.io/unschedulable toleration - antiAffinity bool // required podAntiAffinity(hostname) — 노드당 하나로 확산 - readyDelay int // readiness initialDelaySeconds (0이면 1) - probeFile string // 비어 있지 않으면 hostPath 파일 exec probe — 파일이 있는 노드에서만 Ready - pvc string // 비어 있지 않으면 이 PVC를 마운트 + antiAffinity bool // required podAntiAffinity(hostname) — spread one per node + readyDelay int // readiness initialDelaySeconds (0 means 1) + probeFile string // non-empty adds a hostPath file exec probe — Ready only on the node holding the file + pvc string // non-empty mounts this PVC } func deployYAML(w workload) string { @@ -282,7 +282,7 @@ func nodeUnschedulable(node string) string { return strings.TrimSpace(out) } -// podsOf는 원본이든 대체든 그 앱의 Pod 전부를 준다. +// podsOf returns every Pod of that app, original or replacement. func podsOf(app string) []string { out := mustKubectl("get", "pods", "-n", "default", "-l", "app="+app, "-o", `jsonpath={range .items[*]}{.metadata.name}{"\n"}{end}`) @@ -305,8 +305,8 @@ func podCost(name string) string { return strings.TrimSpace(out) } -// replacementPods는 그 앱의 대체 Pod만 센다. 클러스터 전역으로 세면 -// 다른 스펙이나 이전 실행이 남긴 것에 오염된다. +// replacementPods counts only that app's replacements. Counting cluster-wide would be +// polluted by leftovers from another spec or an earlier run. func replacementPods(app string) []string { out := mustKubectl("get", "pods", "-n", "default", "-l", "soft-drain.com/replaces,app="+app, "-o", `jsonpath={range .items[*]}{.metadata.name}{"\n"}{end}`) @@ -335,7 +335,7 @@ func allWorkers() []string { return workers } -// pickWorker는 스케줄 가능한 워커 노드 하나를 고른다. +// pickWorker picks one schedulable worker node. func pickWorker() string { out := mustKubectl("get", "nodes", "-o", `jsonpath={range .items[*]}{.metadata.name} {.spec.unschedulable}{"\n"}{end}`) @@ -349,8 +349,8 @@ func pickWorker() string { return "" } -// cleanup은 best-effort다. 마지막 스펙의 DeferCleanup은 AfterAll(컨트롤러 -// undeploy) 뒤에 돌 수 있어 컨트롤러의 restore에 기대지 않고 직접 걷는다. +// cleanup is best-effort. The last spec's DeferCleanup can run after AfterAll (which +// undeploys the controller), so it takes things back itself instead of relying on restore. func cleanupDrainNode(node string) { _, _ = kubectl("label", "node", node, "soft-drain.com/drain-") _, _ = kubectl("label", "node", node, "soft-drain.com/state-") @@ -360,8 +360,8 @@ func cleanupDrainNode(node string) { func cleanupApp(app string) { _, _ = kubectl("delete", "deployment", "-n", "default", app, "--ignore-not-found", "--wait=false") - // 주인 없는 대체 Pod은 Deployment 삭제로 걷히지 않는다. 컨트롤러가 이미 - // 내려간 뒤에도 남지 않도록 직접 지운다. + // An ownerless replacement is not collected by deleting the Deployment. Delete it here + // so nothing is left behind once the controller is gone. _, _ = kubectl("delete", "pods", "-n", "default", "-l", "app="+app, "--ignore-not-found", "--wait=false") } @@ -370,8 +370,8 @@ func cleanupDrain(node, app string) { cleanupApp(app) } -// deployPackedOnNode는 다른 워커를 잠시 cordon해서 워크로드 전체를 그 노드에 -// 앉힌다. nodeSelector 고정과 달리 대체 Pod은 자유롭게 다른 노드로 갈 수 있다. +// deployPackedOnNode cordons the other workers briefly so the whole workload lands on +// that node. Unlike a nodeSelector pin, replacements are free to go to another node. func deployPackedOnNode(w workload, node string) { var others []string for _, o := range allWorkers() { @@ -393,8 +393,8 @@ func deployPackedOnNode(w workload, node string) { const controllerDeploy = "deploy/soft-drain-controller-manager" -// startPinnedDrain은 노드에 고정된 워크로드를 만들고 drain을 걸어 -// "InProgress + Pending 대체 Pod 1개" 상태까지 끌고 간다. +// startPinnedDrain creates a workload pinned to a node and starts a drain, driving it to +// the state "InProgress with one Pending replacement". func startPinnedDrain(app, worker string, w workload) (origPod string) { applyYAML(deployYAML(w)) mustKubectl("rollout", "status", "-n", "default", "deploy/"+app, "--timeout=180s") @@ -417,8 +417,8 @@ type availabilityMonitor struct { violations []string } -// watchAvailability는 각 Deployment의 availableReplicas와 Service Endpoints를 -// 주기적으로 샘플링해 의도한 개수 밑으로 떨어진 순간을 기록한다. +// watchAvailability samples each Deployment's availableReplicas and the Service Endpoints +// periodically, recording any moment either falls below the intended count. func watchAvailability(apps map[string]int) *availabilityMonitor { m := &availabilityMonitor{stop: make(chan struct{}), done: make(chan struct{})} go func() { @@ -512,7 +512,7 @@ var _ = Describe("soft-drain", Ordered, func() { Eventually(func() string { return nodeStateLabel(srcNode) }, 3*time.Minute, 3*time.Second).Should(Equal("Complete")) - // 노드는 cordon된 채로 남는다 + // the node stays cordoned Expect(nodeUnschedulable(srcNode)).To(Equal("true")) By("verifying the workload moved") @@ -522,7 +522,7 @@ var _ = Describe("soft-drain", Ordered, func() { g.Expect(pods[0]).NotTo(Equal(origPod)) g.Expect(nodeOfPod(pods[0])).NotTo(Equal(srcNode)) g.Expect(podPhase(pods[0])).To(Equal("Running")) - // ReplicaSet에 입양됐다: hash는 있고 replaces는 없다 + // adopted by the ReplicaSet: the hash is there and replaces is gone labels := mustKubectl("get", "pod", "-n", "default", pods[0], "-o", "jsonpath={.metadata.labels}") g.Expect(labels).To(ContainSubstring("pod-template-hash")) g.Expect(labels).NotTo(ContainSubstring("soft-drain.com/replaces")) @@ -572,7 +572,7 @@ var _ = Describe("soft-drain", Ordered, func() { const app = "sd-pending" worker := pickWorker() DeferCleanup(func() { cleanupDrain(worker, app) }) - // 모든 워커를 막을 수는 없으니 노드 고정으로 "갈 곳 없음"을 만든다 + // not every worker can be blocked, so a node pin manufactures "nowhere to go" origPod := startPinnedDrain(app, worker, workload{name: app, pin: worker}) Expect(podCost(origPod)).To(Equal("-2147483648")) @@ -588,7 +588,7 @@ var _ = Describe("soft-drain", Ordered, func() { g.Expect(podCost(origPod)).To(BeEmpty()) }, 2*time.Minute, 3*time.Second).Should(Succeed()) - // 원래 Pod은 건드리지 않았다 + // the original Pod was left untouched Expect(podsOf(app)).To(Equal([]string{origPod})) }) @@ -612,7 +612,7 @@ var _ = Describe("soft-drain", Ordered, func() { Eventually(func() []string { return replacementPods(app) }, 2*time.Minute, 3*time.Second).Should(BeEmpty()) - // 원래 Pod은 건드리지 않았다 + // the original Pod was left untouched Expect(podsOf(app)).To(Equal([]string{origPod})) }) @@ -651,8 +651,8 @@ var _ = Describe("soft-drain", Ordered, func() { DeferCleanup(func() { cleanupDrain(worker, app) }) startPinnedDrain(app, worker, workload{name: app, pin: worker, recreate: true}) - // Recreate가 옛 Pod(타깃)을 즉시 지운다. 새 Pod은 cordon된 노드에 - // 고정돼 미스케줄 Pending이므로 타깃이 아니다. + // Recreate deletes the old Pod (the target) at once. The new Pod is pinned to the + // cordoned node and stays mis-scheduled Pending, so it is not a target. By("triggering a Recreate rollout mid-drain") mustKubectl("patch", "deployment", "-n", "default", app, "--type=merge", "-p", `{"spec":{"template":{"metadata":{"labels":{"rollout":"v2"}}}}}`) @@ -684,12 +684,12 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("patch", "deployment", "-n", "default", app, "--type=merge", "-p", `{"spec":{"template":{"metadata":{"labels":{"rollout":"v2"}}}}}`) - // readiness 45초를 채우기 한참 전에 회수된다 — Ready를 기다렸다면 여기서 걸린다 + // reclaimed long before the 45s readiness elapses — waiting for Ready would hang here Eventually(func(g Gomega) { out, err := kubectl("get", "pod", "-n", "default", stale, "-o", "jsonpath={.metadata.deletionTimestamp}") if err != nil { - // 이미 완전히 사라진 경우만 통과 — 일시적 API 오류는 통과가 아니다 + // only a Pod that is fully gone passes — a transient API error is not a pass g.Expect(err.Error()).To(ContainSubstring("NotFound")) return } @@ -713,7 +713,7 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("rollout", "status", "-n", "default", "deploy/"+app, "--timeout=180s") srcNode := nodeOfPod(podsOf(app)[0]) - // 착지를 결정론으로 만든다: land 하나만 남기고 나머지 워커는 손 cordon + // make the landing deterministic: keep only land and hand-cordon the other workers var land string var parked []string for _, w := range allWorkers() { @@ -749,12 +749,12 @@ var _ = Describe("soft-drain", Ordered, func() { By("draining the node the replacement landed on") mustKubectl("label", "node", land, "soft-drain.com/drain=true") - // readiness 45초를 채우기 한참 전에 회수된다 + // reclaimed long before the 45s readiness elapses Eventually(func(g Gomega) { out, err := kubectl("get", "pod", "-n", "default", repl, "-o", "jsonpath={.metadata.deletionTimestamp}") if err != nil { - // 이미 완전히 사라진 경우만 통과 — 일시적 API 오류는 통과가 아니다 + // only a Pod that is fully gone passes — a transient API error is not a pass g.Expect(err.Error()).To(ContainSubstring("NotFound")) return } @@ -773,7 +773,7 @@ var _ = Describe("soft-drain", Ordered, func() { } g.Expect(live).To(HaveLen(1)) g.Expect(live[0]).NotTo(Equal(repl)) - // land에 잠깐 앉았다 회수되는 교차가 있어도 결국 Pending으로 수렴한다 + // even if it briefly lands on land and is reclaimed, it converges to Pending g.Expect(podPhase(live[0])).To(Equal("Pending")) }, time.Minute, 2*time.Second).Should(Succeed()) @@ -812,8 +812,8 @@ var _ = Describe("soft-drain", Ordered, func() { By("scaling up mid-drain") mustKubectl("scale", "deployment", "-n", "default", app, "--replicas=3") - // 새 Pod들은 cordon된 노드에 고정돼 미스케줄 Pending — 타깃이 아니다. - // "타깃마다 하나"가 아니라 replicas 기준으로 만들었다면 여기서 초과 생성된다. + // the new Pods are pinned to the cordoned node and stay mis-scheduled Pending — not targets. + // Creating by replicas instead of "one per target" would over-create right here. Consistently(func() []string { return replacementPods(app) }, 45*time.Second, 5*time.Second).Should(HaveLen(1)) Expect(nodeStateLabel(worker)).To(Equal("InProgress")) @@ -847,8 +847,8 @@ var _ = Describe("soft-drain", Ordered, func() { cleanupApp(app) }) - // 빈 워커들을 먼저 막는다. src를 먼저 라벨하면 대체 Pod이 아직 cordon - // 안 된 워커에 스케줄돼 교착 없이 그냥 끝나버린다. + // block the empty workers first. Labelling src first would let the replacement schedule + // onto a worker not yet cordoned and finish without ever deadlocking. By("cordoning off every other worker first") for _, w := range workers { if w != src { @@ -874,8 +874,8 @@ var _ = Describe("soft-drain", Ordered, func() { g.Expect(podPhase(repls[0])).To(Equal("Pending")) }, 3*time.Minute, 3*time.Second).Should(Succeed()) - // Complete된 빈 워커 하나를 되돌린다. 라벨만 걷으면 컨트롤러가 - // 자기가 걸었던 cordon도 함께 걷는다 — 여기서 그 동작이 교착을 푼다. + // give back one Complete empty worker. Removing just the label makes the controller take + // back the cordon it placed — that behaviour is what breaks the deadlock here. var freed string for _, w := range workers { if w != src { @@ -910,7 +910,7 @@ var _ = Describe("soft-drain", Ordered, func() { Eventually(func() string { return nodeStateLabel(src) }, 3*time.Minute, 3*time.Second).Should(Equal("Complete")) - // 우리가 건 cordon이 아니므로 어노테이션이 없다 + // not a cordon we placed, so there is no annotation ann, _ := kubectl("get", "node", src, "-o", `jsonpath={.metadata.annotations.soft-drain\.com/cordoned-by-controller}`) Expect(strings.TrimSpace(ann)).To(BeEmpty()) @@ -935,7 +935,7 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("label", "node", worker, "soft-drain.com/drain=true") - // naked pod은 타깃이 아니므로 노드는 곧바로 Complete가 된다 + // a naked pod is not a target, so the node goes Complete right away Eventually(func() string { return nodeStateLabel(worker) }, 2*time.Minute, 3*time.Second).Should(Equal("Complete")) @@ -993,14 +993,14 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("rollout", "status", "-n", "default", "deploy/"+app, "--timeout=180s") origPod := podsOf(app)[0] - // 이전 스펙이나 실행이 남긴 같은 reason의 이벤트에 오염되지 않게 지우고 시작한다 + // start clean so events with the same reason from an earlier spec or run do not pollute it _, _ = kubectl("delete", "events", "-n", "default", "--field-selector", "reason=ReplacementOnDrainingNode", "--ignore-not-found") mustKubectl("label", "node", worker, "soft-drain.com/drain=true") - // 대체 Pod이 cordon을 뚫고 같은 노드에 앉아 Ready가 되면 지워지고, - // 다음 라운드가 다시 만든다. 반복 Warning Event가 그 증거다. + // a replacement that gets through the cordon onto the same node is deleted once Ready, + // and the next round creates another. The repeating Warning Events are the evidence. By("waiting for the landing-deletion loop to leave evidence") Eventually(func(g Gomega) { out, _ := kubectl("get", "events", "-n", "default", @@ -1008,12 +1008,12 @@ var _ = Describe("soft-drain", Ordered, func() { g.Expect(nonEmptyLines(out)).NotTo(BeEmpty()) }, 3*time.Minute, 5*time.Second).Should(Succeed()) - // 옮기지 못하므로 InProgress에 머물고, 원래 Pod은 산다 + // it cannot be moved, so this stays InProgress and the original Pod lives Expect(nodeStateLabel(worker)).To(Equal("InProgress")) Expect(podsOf(app)).To(ContainElement(origPod)) }) - // ─── 다중 replicas ─── + // ─── multiple replicas ─── It("r=3 packed on one node moves all three without downtime", Label("shard-d"), func() { const app = "sd-pack" @@ -1063,7 +1063,7 @@ var _ = Describe("soft-drain", Ordered, func() { Expect(violations).To(BeEmpty(), "availability dropped during drain:\n%s", strings.Join(violations, "\n")) - // 다른 노드의 Pod은 이름(오브젝트)까지 그대로다 — 엉뚱한 Pod을 안 죽였다 + // the Pod on the other node is identical down to its name (the object) — no wrong Pod was killed for name, node := range untouched { Expect(podPhase(name)).To(Equal("Running")) Expect(nodeOfPod(name)).To(Equal(node)) @@ -1123,7 +1123,7 @@ var _ = Describe("soft-drain", Ordered, func() { g.Expect(podPhase(repls[0])).To(Equal("Pending")) }, 2*time.Minute, 3*time.Second).Should(Succeed()) - // "먼저 띄우고 나중에 지운다" 방식 전체의 산술 — 막힌 채 유지된다 + // the arithmetic of "bring it up first, delete it later" in general — it stays blocked Consistently(func() string { return nodeStateLabel(src) }, 30*time.Second, 5*time.Second).Should(Equal("InProgress")) @@ -1134,7 +1134,7 @@ var _ = Describe("soft-drain", Ordered, func() { }, 2*time.Minute, 3*time.Second).Should(Succeed()) }) - // ─── 방해 행위자 ─── + // ─── disruptive actors ─── It("a controller restart mid-drain resumes without duplicates", Label("shard-c"), func() { const app = "sd-restart" @@ -1146,13 +1146,13 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("rollout", "restart", "-n", namespace, controllerDeploy) mustKubectl("rollout", "status", "-n", namespace, controllerDeploy, "--timeout=120s") - // 무기억이라 재기동 후에도 같은 판정 — 대체 Pod이 늘지도 줄지도 않는다 + // memoryless, so the decision after a restart is the same — no replacement is added or lost Consistently(func(g Gomega) { g.Expect(replacementPods(app)).To(HaveLen(1)) g.Expect(nodeStateLabel(worker)).To(Equal("InProgress")) }, 30*time.Second, 5*time.Second).Should(Succeed()) - // 재기동한 컨트롤러가 취소도 처리한다 + // the restarted controller handles the cancel too mustKubectl("label", "node", worker, "soft-drain.com/drain-") Eventually(func(g Gomega) { g.Expect(nodeStateLabel(worker)).To(BeEmpty()) @@ -1185,7 +1185,7 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("scale", "-n", namespace, controllerDeploy, "--replicas=1") mustKubectl("rollout", "status", "-n", namespace, controllerDeploy, "--timeout=120s") - // watch는 edge가 아니라 level이다 — 놓친 이벤트 없이 현재 상태에서 수렴한다 + // a watch is level, not edge — it converges from the current state with no missed events Eventually(func() string { return nodeStateLabel(src) }, 3*time.Minute, 3*time.Second).Should(Equal("Complete")) Eventually(func(g Gomega) { @@ -1211,9 +1211,9 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("label", "node", src, "soft-drain.com/drain=true") - // Complete는 옛 컨트롤러 Pod 오브젝트가 사라진 뒤에만 붙을 수 있고, - // 그 시점에 옛 인스턴스는 이미 죽어 있다 — 이주한 새 인스턴스가 - // 리더 리스를 이어받아 루프를 계속한다는 증명이다. + // Complete can only be set after the old controller Pod object is gone, and by then the + // old instance is already dead — proof that the migrated instance took over the leader + // lease and kept the loop running. Eventually(func() string { return nodeStateLabel(src) }, 3*time.Minute, 3*time.Second).Should(Equal("Complete")) @@ -1276,7 +1276,7 @@ var _ = Describe("soft-drain", Ordered, func() { cleanupDrain(src, app) }) - // pause + 템플릿 변경으로 Healthy(D)를 거짓으로 고정한다 + // pause plus a template change pins Healthy(D) to false By("pausing the deployment with a pending template change") mustKubectl("rollout", "pause", "-n", "default", "deploy/"+app) mustKubectl("patch", "deployment", "-n", "default", app, "--type=merge", @@ -1284,7 +1284,7 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("label", "node", src, "soft-drain.com/drain=true") - // 대체 Pod이 Ready가 되어도 replaces 라벨을 단 채 남아 있어야 한다 + // the replacement must stay, replaces label and all, even once it is Ready By("verifying the handover is deferred") Eventually(func() []string { return replacementPods(app) }, 2*time.Minute, 3*time.Second).Should(HaveLen(1)) @@ -1337,7 +1337,7 @@ var _ = Describe("soft-drain", Ordered, func() { }, 2*time.Minute, 3*time.Second).Should(Succeed()) }) - // ─── 워크로드 다양성 ─── + // ─── workload variety ─── It("a forever-unready replacement never costs the original its life", Label("shard-d"), func() { const app = "sd-noready" @@ -1347,7 +1347,7 @@ var _ = Describe("soft-drain", Ordered, func() { _, _ = utils.Run(exec.Command("docker", "exec", worker, "rm", "-rf", "/var/lib/sd-e2e")) }) - // 이 노드에만 marker 파일을 만든다 — 대체 Pod은 어디에 앉든 Ready가 못 된다 + // the marker file exists only on this node — a replacement can never be Ready wherever it lands By("planting the readiness marker on one node only") _, err := utils.Run(exec.Command("docker", "exec", worker, "sh", "-c", "mkdir -p /var/lib/sd-e2e && touch /var/lib/sd-e2e/ready")) @@ -1359,7 +1359,7 @@ var _ = Describe("soft-drain", Ordered, func() { monitor := watchAvailability(map[string]int{app: 1}) mustKubectl("label", "node", worker, "soft-drain.com/drain=true") - // 대체 Pod이 생기고, Ready가 못 되니 넘기기가 영영 일어나지 않는다 + // a replacement appears, and since it never goes Ready the hand-over never happens Eventually(func() []string { return replacementPods(app) }, 2*time.Minute, 3*time.Second).Should(HaveLen(1)) Consistently(func(g Gomega) { @@ -1404,7 +1404,7 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("label", "node", src, "soft-drain.com/drain=true") - // PV의 노드 어피니티가 cordon된 노드를 가리켜 대체 Pod이 영영 Pending이다 + // the PV's node affinity points at the cordoned node, so the replacement is Pending forever Eventually(func(g Gomega) { repls := replacementPods(app) g.Expect(repls).To(HaveLen(1)) @@ -1444,7 +1444,7 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("label", "node", worker, "soft-drain.com/drain=true") - // 셋 다 타깃이 아니므로 노드는 곧바로 Complete가 된다 + // none of the three is a target, so the node goes Complete right away Eventually(func() string { return nodeStateLabel(worker) }, 2*time.Minute, 3*time.Second).Should(Equal("Complete")) @@ -1516,7 +1516,7 @@ var _ = Describe("soft-drain", Ordered, func() { mustKubectl("uncordon", worker) Eventually(func() string { return nodeStateLabel(worker) }, 60*time.Second, 2*time.Second).Should(Equal("Cancelled")) - // Cancelled 래치는 플러그인이 걷고 다시 건다 — 재drain은 새 drain과 같다 + // the plugin removes and re-adds the Cancelled latch — a re-drain is just a new drain out, err = utils.Run(exec.Command(plugin, worker, "--wait=false", "--timeout", "2m")) Expect(err).NotTo(HaveOccurred(), out) Expect(out).To(ContainSubstring("clearing it first")) @@ -1573,13 +1573,13 @@ var _ = Describe("soft-drain", Ordered, func() { Expect(err).NotTo(HaveOccurred(), out) for _, o := range two { Expect(nodeStateLabel(o)).To(BeEmpty()) - // release는 우리가 걸었던 cordon도 걷는다 — 복원 patch가 원자적이라 - // state가 비었으면 uncordon도 끝나 있다 + // release also takes back the cordon we placed — the restore patch is atomic, so an + // empty state means the uncordon is done too Expect(nodeUnschedulable(o)).To(BeEmpty()) } By("draining to completion in blocking mode") - applyYAML(deployYAML(workload{name: app})) // 고정 해제 — 이제 갈 곳이 있다 + applyYAML(deployYAML(workload{name: app})) // unpinned — there is somewhere to go now mustKubectl("rollout", "status", "-n", "default", "deploy/"+app, "--timeout=180s") var src string Eventually(func(g Gomega) { @@ -1633,7 +1633,7 @@ var _ = Describe("soft-drain", Ordered, func() { } mustKubectl("label", "node", src, "soft-drain.com/drain=true") - // 접기 전 규칙이라면 대체 Pod이 착지하는 족족 지워져 영원히 안 끝난다 + // under the pre-fold rule the replacement would be deleted as fast as it lands and never finish Eventually(func() string { return nodeStateLabel(src) }, 3*time.Minute, 3*time.Second).Should(Equal("Complete")) Eventually(func(g Gomega) {