-
Notifications
You must be signed in to change notification settings - Fork 1
feat(loadtest): EBS 처리량 인과 라운드 매니페스트 + 디스크 수요 폴러 (#603) #690
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
53c21ed
a814804
ba33422
bb343d0
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
Large diffs are not rendered by default.
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,131 @@ | ||
| #!/usr/bin/env bash | ||
| # AI 워커 부하-중 장애 빈도 — 디스크 «수요» 폴러. | ||
| # | ||
| # 이 스크립트가 있는 이유: 2026-08-28 라운드가 두 번 다 박스 정지로 끝났고, 09-02 CloudWatch | ||
| # 사후 조회가 「gp3 기본 처리량 상한(125MiB/s)에 눌러붙었다」를 찾았지만 **인과를 못 세웠다** — | ||
| # 결과 문서 §6-7이 직접 적었다: | ||
| # | ||
| # "요청이 상한보다 먼저 늘었는지, 상한에 닿은 뒤에 밀린 것뿐인지를 봐야 하는데, | ||
| # 그 지표를 이 라운드가 안 걷었다 — 이건 두 인스턴스가 사라진 지금은 영영 못 메운다." | ||
| # | ||
| # 그 채널이 이것이다. CloudWatch EBS 지표는 gp3에서 **5분 해상도가 한계**인데 사고는 90초 만에 | ||
| # 났다 — 그래서 이 박스 안 폴러가 대체재가 아니라 **유일한 채널**이다. | ||
| # | ||
| # 설계: docs/decisions/ai-worker-load-soak-experiment.md | ||
| # 매니페스트: loadtest/aws/ROUND-2026-09-08-ebs-causality.md §4-ㄱ | ||
| # | ||
| # 대상 박스(shadowfit-* 컨테이너가 떠 있는 곳)에서 root로 nohup 실행. | ||
| # _monitor.sh(60초)·_diagnostics.sh(7초)와 **같이** 돈다 — 겹치는 채널은 없다. | ||
| set -uo pipefail | ||
|
|
||
| INTERVAL_SEC=${INTERVAL_SEC:-7} # _diagnostics.sh 와 같은 리듬 | ||
| DURATION_SEC=${DURATION_SEC:-10800} # 부하기 쪽과 맞춘다 | ||
| TAIL_SEC=${TAIL_SEC:-600} # 부하 종료 후에도 관찰(상한에서 언제 내려오나) | ||
| PW=${PW:-1234} | ||
| OUT=${OUT:-/root/ai_worker_load_soak_disk.csv} | ||
| DEV=${DEV:-} # 비우면 루트 파일시스템에서 유도한다 | ||
| CONTAINERS=${CONTAINERS:-"shadowfit-ai shadowfit-backend shadowfit-mysql"} | ||
|
|
||
| # ── 게이트 ──────────────────────────────────────────────────────────────── | ||
| # 🔴 여기서 죽는 것이 이 스크립트의 일이다. 장치를 못 찾은 채로 돌면 표는 멀쩡한데 값이 | ||
| # 전부 -1 이고, 그건 #271(「없는 카운터를 8판 내내 읽고 0을 찍었다」)의 재발이다. | ||
| die() { echo "🔴 $*" >&2; exit 1; } | ||
|
|
||
| [ -r /proc/diskstats ] || die "/proc/diskstats 를 못 읽는다 — 이 채널 없이는 이 라운드를 돌 이유가 없다" | ||
|
|
||
| if [ -z "$DEV" ]; then | ||
| src=$(findmnt -no SOURCE / 2>/dev/null) || src="" | ||
| [ -n "$src" ] || die "루트 파일시스템의 장치를 못 찾았다(findmnt 실패) — DEV=<장치명> 으로 직접 줄 것" | ||
| base=$(basename "$src") | ||
| # nvme0n1p1 -> nvme0n1 · xvda1 -> xvda · sda1 -> sda | ||
| case "$base" in | ||
| nvme*) DEV=$(echo "$base" | sed -E 's/p[0-9]+$//') ;; | ||
| *) DEV=$(echo "$base" | sed -E 's/[0-9]+$//') ;; | ||
| esac | ||
| fi | ||
|
|
||
| grep -qE "[[:space:]]${DEV}[[:space:]]" /proc/diskstats \ | ||
| || die "장치 '$DEV' 가 /proc/diskstats 에 없다 — DEV 를 직접 줄 것. 후보: $(awk '{print $3}' /proc/diskstats | tr '\n' ' ')" | ||
|
|
||
| # MySQL 상태 채널이 살아 있는지도 지금 확인한다 — 부하 중에 처음 알면 늦다. | ||
| if ! docker exec -e MYSQL_PWD="$PW" shadowfit-mysql mysql -uroot -N \ | ||
| -e "SHOW GLOBAL STATUS LIKE 'Innodb_data_written';" >/dev/null 2>&1; then | ||
| echo "⚠️ MySQL 상태 조회가 지금 실패한다 — 컨테이너 이름·PW 를 확인할 것." >&2 | ||
| echo " (막지는 않는다. 부하 중 무응답은 그 자체가 관측이라 -1 로 계속 찍는다)" >&2 | ||
| fi | ||
|
Comment on lines
+51
to
+55
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🗄️ Data Integrity & Integration | 🟠 Major | ⚡ Quick win 초기 MySQL 실패를 판 무효로 처리하세요. Line 51–55는 인증 오류나 컨테이너 오류를 경고만 하고 계속 진행합니다. 이후 시작 검증 실패는 Also applies to: 65-71 🤖 Prompt for AI Agents |
||
|
|
||
| # ── 읽는 것 ─────────────────────────────────────────────────────────────── | ||
| # /proc/diskstats 필드: 3=장치 4=읽기완료 6=읽은섹터 8=쓰기완료 10=쓴섹터 12=진행중IO | ||
| # 섹터는 리눅스 관례대로 **512바이트 고정**이다(장치의 물리 섹터 크기와 무관). | ||
| read_dev() { awk -v d="$DEV" '$3==d {print $6, $10, $12; found=1} END{if(!found) print -1, -1, -1}' /proc/diskstats; } | ||
|
|
||
| # 🔴 위치가 아니라 **이름**으로 받는다 — SHOW GLOBAL STATUS 는 이름 오름차순이라 | ||
| # Com_commit 이 Innodb_* 보다 먼저 나온다. 위치로 읽으면 열이 통째로 밀린 채 | ||
| # 표가 멀쩡해 보인다(#271 계열의 조용한 결함). | ||
| read_mysql() { | ||
| docker exec -e MYSQL_PWD="$PW" shadowfit-mysql mysql -uroot -N -e \ | ||
| "SHOW GLOBAL STATUS WHERE Variable_name IN | ||
| ('Innodb_data_writes','Innodb_data_written','Innodb_os_log_written','Com_commit');" 2>/dev/null \ | ||
|
Comment on lines
+66
to
+68
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🩺 Stability & Availability | 🟠 Major | ⚡ Quick win 🔎 Supported by static analysis🤖 get_repo_knowledge executed:
Length of output: 1863 🏁 Script executed: #!/bin/bash
set -eu
printf '%s\n' '--- target file ---'
cat -n loadtest/measure_ai_worker_load_soak_disk.sh | sed -n '1,150p'
printf '%s\n' '--- related timeout patterns ---'
rg -n -S 'timeout|docker (exec|stats|inspect)|mysql' loadtest README.md .github 2>/dev/null | head -200
printf '%s\n' '--- tracked status and focused diff ---'
git status --short
git diff -- loadtest/measure_ai_worker_load_soak_disk.shRepository: Shadowfit/init Length of output: 35788 Docker 및 MySQL 조회에 제한 시간을 적용하세요.
🤖 Prompt for AI Agents |
||
| | awk '{v[$1]=$2} | ||
| END{ n=split("Innodb_data_writes Innodb_data_written Innodb_os_log_written Com_commit", k, " "); | ||
| for(i=1;i<=n;i++) printf "%s ", (k[i] in v ? v[k[i]] : -1) }' | ||
| } | ||
|
|
||
| # 컨테이너별 누적 블록 I/O 와 json-file 로그 크기. | ||
| # 🔑 로그 크기가 1순위 후보다 — 1차 사고에서 07:54에 부하를 껐는데도 08:05까지 처리량이 | ||
| # 상한에 붙어 있었다. 부하가 없는데 쓰기가 계속됐다는 뜻이라 「누가 쓰는가」를 갈라야 한다. | ||
| read_containers() { | ||
| local c out="" | ||
| for c in $CONTAINERS; do | ||
| local bio logsz | ||
| bio=$(docker stats --no-stream --format '{{.BlockIO}}' "$c" 2>/dev/null | tr -d ' ' || echo "n/a") | ||
| [ -n "$bio" ] || bio="n/a" | ||
| local lp | ||
| lp=$(docker inspect --format '{{.LogPath}}' "$c" 2>/dev/null || echo "") | ||
| if [ -n "$lp" ] && [ -f "$lp" ]; then | ||
| logsz=$(stat -c %s "$lp" 2>/dev/null || echo -1) | ||
| else | ||
| logsz=-1 | ||
| fi | ||
| out="${out}${bio},${logsz}," | ||
| done | ||
| echo "${out%,}" | ||
| } | ||
|
|
||
| # ── 폴링 ────────────────────────────────────────────────────────────────── | ||
| DEADLINE=$(( $(date +%s) + DURATION_SEC + TAIL_SEC )) | ||
|
|
||
| { | ||
| echo "# 장치=$DEV interval=${INTERVAL_SEC}s duration=${DURATION_SEC}s tail=${TAIL_SEC}s" | ||
| echo "# 섹터=512B 고정 · rate 는 직전 표본과의 차분 / 실경과초" | ||
| echo "# gp3 기본 처리량 상한 = 128,000 KiB/s (=125MiB/s) — write_KiBps 가 여기 눌러붙는지가 관측 대상" | ||
| printf 'epoch,read_KiBps,write_KiBps,io_inflight,rd_sectors_cum,wr_sectors_cum,' | ||
| printf 'innodb_data_writes,innodb_data_written,innodb_os_log_written,com_commit' | ||
| for c in $CONTAINERS; do printf ',%s_blkio,%s_logbytes' "$c" "$c"; done | ||
| printf '\n' | ||
| } > "$OUT" | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🩺 Stability & Availability | 🟠 Major | ⚡ Quick win 출력 파일 생성 실패를 즉시 중단하세요. Line 106의 출력 파일을 연 직후 성공 여부를 검사하고, 이후 append 실패도 치명적 오류로 처리하세요. 🤖 Prompt for AI Agents |
||
|
|
||
| prev_rd=-1; prev_wr=-1; prev_ts=0 | ||
|
|
||
| while [ "$(date +%s)" -lt "$DEADLINE" ]; do | ||
| ts=$(date +%s) | ||
| read -r rd wr inflight <<< "$(read_dev)" | ||
|
|
||
| if [ "$prev_rd" -ge 0 ] && [ "$rd" -ge 0 ] && [ "$ts" -gt "$prev_ts" ]; then | ||
| # 512B 섹터 → KiB/s : (Δ섹터 × 512) / Δ초 / 1024 = Δ섹터 / Δ초 / 2 | ||
| rd_rate=$(awk -v a="$rd" -v b="$prev_rd" -v d="$((ts-prev_ts))" 'BEGIN{printf "%.1f",(a-b)/d/2}') | ||
| wr_rate=$(awk -v a="$wr" -v b="$prev_wr" -v d="$((ts-prev_ts))" 'BEGIN{printf "%.1f",(a-b)/d/2}') | ||
| else | ||
| rd_rate=""; wr_rate="" # 첫 표본은 차분이 없다 — 0 으로 채우지 않는다 | ||
| fi | ||
|
|
||
| read -r idw idwn iolw commit <<< "$(read_mysql)" | ||
| cstats=$(read_containers) | ||
|
|
||
| echo "$ts,$rd_rate,$wr_rate,$inflight,$rd,$wr,$idw,$idwn,$iolw,$commit,$cstats" >> "$OUT" | ||
|
|
||
| prev_rd=$rd; prev_wr=$wr; prev_ts=$ts | ||
| sleep "$INTERVAL_SEC" | ||
| done | ||
|
|
||
| echo "# END $(date -u +%FT%TZ)" >> "$OUT" | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,84 @@ | ||
| #!/usr/bin/env bash | ||
| # AI 워커 부하-중 장애 빈도 — 증거 회수기. **부하기 박스에서** 돈다. | ||
| # | ||
| # 이 스크립트가 있는 이유: 2026-08-28 라운드는 두 번 다 **대상 박스가 얼어서 그 안의 로그를 | ||
| # 못 건졌다.** 1차는 systemd 저널이 부하 시작 *전*에 죽었고, SSH 가 끝내 안 붙어 리부트 전 | ||
| # dmesg 를 통째로 잃었다. 계측을 아무리 잘 걸어도 **그 계측이 대상 박스와 같이 죽으면 0** 이다. | ||
| # | ||
| # 그래서 대상의 로그를 부하기(살아남는 쪽)로 계속 흘려 받는다. 대상이 얼어도 | ||
| # **그 순간까지의 줄은 여기 남는다.** | ||
| # | ||
| # 설계: docs/decisions/ai-worker-load-soak-experiment.md | ||
| # 매니페스트: loadtest/aws/ROUND-2026-09-08-ebs-causality.md §4-ㄴ | ||
| set -uo pipefail | ||
|
|
||
| TARGET=${TARGET:?TARGET(대상 박스 사설 IP) 필요} | ||
| SSH_USER=${SSH_USER:-ec2-user} | ||
| SSH_KEY=${SSH_KEY:-$HOME/.ssh/shadowfit-measure.pem} | ||
| DURATION_SEC=${DURATION_SEC:-10800} | ||
| TAIL_SEC=${TAIL_SEC:-600} | ||
| HB_INTERVAL=${HB_INTERVAL:-7} # 하트비트 간격 — 대상 폴러(7초)와 같은 리듬 | ||
| OUT_DIR=${OUT_DIR:-/root/pulled} | ||
| # 대상에서 빨아올 파일들. 대상 폴러 셋의 기본 출력 경로다. | ||
| FILES=${FILES:-"/root/ai_worker_load_soak_monitor.log /root/innodb_status_poll.log /root/ai_worker_load_soak_disk.csv /root/backend_threaddumps.log"} | ||
|
|
||
| die() { echo "🔴 $*" >&2; exit 1; } | ||
| SSH="ssh -i $SSH_KEY -o StrictHostKeyChecking=no -o ConnectTimeout=5 -o BatchMode=yes ${SSH_USER}@${TARGET}" | ||
|
|
||
| mkdir -p "$OUT_DIR" || die "$OUT_DIR 를 못 만든다" | ||
|
|
||
| # ── 게이트 ──────────────────────────────────────────────────────────────── | ||
| # 🔴 여기서 죽는 것이 이 스크립트의 일이다. 붙지도 않는데 조용히 돌면 빈 파일만 남고, | ||
| # 그건 "증거를 회수했다"는 착각을 만든다 — 08-28 이 잃은 것을 또 잃는 모양이다. | ||
| $SSH 'echo ok' >/dev/null 2>&1 || die "대상($TARGET)에 SSH 가 안 붙는다 — 키·보안그룹·IP 확인" | ||
|
|
||
| present="" | ||
| for f in $FILES; do | ||
| if $SSH "test -e '$f'" 2>/dev/null; then present="$present $f"; else echo "⚠️ 대상에 없다(건너뜀): $f" >&2; fi | ||
| done | ||
| [ -n "$present" ] || die "빨아올 파일이 하나도 없다 — 대상 폴러를 **먼저** 띄웠는지 확인할 것" | ||
|
|
||
| DEADLINE=$(( $(date +%s) + DURATION_SEC + TAIL_SEC )) | ||
|
|
||
| # ── 하트비트 ────────────────────────────────────────────────────────────── | ||
| # 08-28 의 결정적 관측은 「세 독립 채널이 같은 30초 창에서 동시에 멎었다」였다. 그 창을 | ||
| # **부하기 시계로** 못 박으려면 이쪽에서 찍는 시각이 필요하다 — 대상이 얼면 대상 시계도 멎는다. | ||
| HB="$OUT_DIR/heartbeat.csv" | ||
| echo "local_epoch,rc,remote_epoch,note" > "$HB" | ||
| ( | ||
| while [ "$(date +%s)" -lt "$DEADLINE" ]; do | ||
| now=$(date +%s) | ||
| remote=$(timeout 5 $SSH 'date +%s' 2>/dev/null); rc=$? | ||
| if [ "$rc" -eq 0 ] && [ -n "$remote" ]; then | ||
| echo "$now,0,$remote," >> "$HB" | ||
| else | ||
| echo "$now,$rc,,대상 무응답" >> "$HB" | ||
| fi | ||
| sleep "$HB_INTERVAL" | ||
| done | ||
| ) & | ||
| HB_PID=$! | ||
|
|
||
| # ── 스트리밍 ────────────────────────────────────────────────────────────── | ||
| # `tail -n +1 -F` 는 파일 처음부터 주고, 이후 추가분을 계속 따라간다. 연결이 끊기면 tail 이 | ||
| # 끝나므로 **그 시점이 곧 「여기까지 받았다」** 이다. 자동 재접속은 일부러 안 한다 — | ||
| # 끊긴 자리를 로그에 남기는 편이 조용히 다시 붙는 것보다 진단에 쓸모 있다. | ||
| PIDS="" | ||
| for f in $present; do | ||
| base=$(basename "$f") | ||
| { | ||
| echo "=== PULL START $(date -u +%FT%TZ) src=$f ===" | ||
| $SSH "tail -n +1 -F '$f'" 2>&1 | ||
| echo "=== PULL END $(date -u +%FT%TZ) (스트림 종료 — 대상이 멎었거나 판이 끝났다) ===" | ||
| } >> "$OUT_DIR/$base" & | ||
| PIDS="$PIDS $!" | ||
| echo " 스트리밍 시작: $f -> $OUT_DIR/$base" | ||
| done | ||
|
|
||
| echo "## 회수기 가동 — 종료 예정 epoch=$DEADLINE (하트비트 pid $HB_PID)" | ||
|
|
||
| while [ "$(date +%s)" -lt "$DEADLINE" ]; do sleep 30; done | ||
|
|
||
| kill $PIDS "$HB_PID" 2>/dev/null | ||
| echo "## 완료 — $(date -u +%FT%TZ)" | ||
| echo " 🔑 판정 먼저 볼 것: $HB 에서 rc≠0 이 처음 나온 시각 = 대상이 멎은 순간(부하기 시계)" |
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,33 @@ | ||
| EBS 처리량 인과 라운드 — 인스턴스 기록 (터미네이트 전까지 이 파일을 지우지 말 것) | ||
|
|
||
| 시작: 2026-09-08T13:28:51Z UTC | ||
| 리전: ap-northeast-2 | ||
| 매니페스트: loadtest/aws/ROUND-2026-09-08-ebs-causality.md | ||
| REF(측정 커밋): a81480467e993b6b63a4a1ec9147aecda1c57921 | ||
| = origin/main(f560dd4c) + loadtest 스크립트만 추가 → 앱 코드는 main 과 동일 | ||
|
|
||
| | 이름 | 인스턴스 | 타입 | 사설 IP | 공인 IP | 볼륨 | 볼륨 스펙 | | ||
| |-------------|----------------------|---------------|----------------|-----------------|-----------------------|------------------| | ||
| | armA-target | i-0a063b0452cb3052e | c7i.2xlarge | 172.31.42.113 | 3.34.177.251 | vol-0cf8928b0a7608a5e | gp3 3000/125 | | ||
| | armA-loader | i-0383497fc1ef6dca9 | c7i.xlarge | 172.31.42.40 | 15.164.212.27 | vol-05722163231227892 | gp3 기본 | | ||
| | armB-target | i-037f0dd465abe3c80 | c7i.2xlarge | 172.31.38.69 | 54.180.32.88 | vol-0ac2c3e8ef1e51389 | gp3 16000/1000 | | ||
| | armB-loader | i-0d2b173bafc4a1c24 | c7i.xlarge | 172.31.44.32 | 54.180.124.117 | vol-0a2ad8f1ceddb1095 | gp3 기본 | | ||
|
|
||
| ✅ 볼륨 스펙 실기 확인 (describe-volumes) — 팔 A 3000 IOPS/125 MiB/s · 팔 B 16000/1000. | ||
| 두 팔이 **볼륨 처리량 하나만** 다르다. | ||
|
|
||
| 워치독: user-data 가 각 박스에서 sleep 14400 후 shutdown -h now. | ||
| 종료 예정(최대): 약 2026-09-08T17:28:51Z UTC — 그전에 사람이 끄는 것이 정상 경로다. | ||
| --instance-initiated-shutdown-behavior terminate 로 띄웠으므로 shutdown = terminate. | ||
|
|
||
| 🔴 라운드가 끝나면 반드시: | ||
| aws ec2 terminate-instances --region ap-northeast-2 --instance-ids i-0a063b0452cb3052e i-0383497fc1ef6dca9 i-037f0dd465abe3c80 i-0d2b173bafc4a1c24 | ||
| 그리고 describe-instances 로 State=terminated 를 눈으로 확인할 것. | ||
| 볼륨은 DeleteOnTermination=true 로 붙였다 — 그래도 describe-volumes 로 확인한다. | ||
| 🔴 팔 B 볼륨은 프로비저닝(16000/1000)이라 남으면 기본 gp3 보다 비싸다. | ||
|
|
||
| === 종료 확인 === | ||
| 2026-09-08T17:13Z terminate-instances 실행 → 4대 전부 State=terminated 확인 | ||
| 태그(Project=shadowfit-measure)로 남은 볼륨 0개 — 프로비저닝 볼륨(16000/1000) 포함 삭제 확인 | ||
| 계정 전체 running/pending/stopping/stopped 인스턴스 0대 확인 | ||
| 요금: (미기입) — Cost Explorer 사후 조회로 채울 것 |
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
🎯 Functional Correctness | 🟡 Minor | ⚡ Quick win
🔎 Supported by static analysis
🤖 get_repo_knowledge executed:
get_repo_knowledge Shadowfit/init /tmp/coderabbit-repo-knowledge/shadowfit-init-70c328afLength of output: 1467
🏁 Script executed:
Repository: Shadowfit/init
Length of output: 13876
🌐 Web query:
util-linux findmnt SOURCE --noheadings root filesystem output device mapper symlink documentation💡 Result:
The findmnt command is part of the util-linux package and displays information about mounted filesystems [1][2]. When using the --noheadings option, findmnt suppresses the header line in the output, which is useful for scripting and parsing [2][3]. Regarding device mapper paths and symlinks: 1. Canonicalization: The util-linux tools, including findmnt and mount, typically canonicalize device paths [4]. For LVM volumes and other device-mapper targets, the canonical form is generally the path under /dev/mapper/ (e.g., /dev/mapper/vg-lv) [4]. 2. Symlink Resolution: Internally, findmnt and related libraries (like libudev and libblkid) may resolve symlinks to reach the underlying block device for identification purposes (such as retrieving labels or UUIDs) [5][6]. 3. Controlling Output: If you need to ensure paths are resolved to their canonical form, you can use the --canonicalize option [4]. The command: findmnt --output SOURCE --noheadings --canonicalize will output the source device column without a header, displaying the resolved, canonical paths [2][7][4]. To specifically list or format output related to device mapper paths, you can define the columns explicitly using --output [2][7]. If the default output is not providing the format you require (e.g., showing a symlink instead of a mapper path), ensure your system is referencing the canonical path in /etc/fstab or the mount table, as findmnt relies on this information [7][4].
Citations:
🏁 Script executed:
Repository: Shadowfit/init
Length of output: 289
🏁 Script executed:
Repository: Shadowfit/init
Length of output: 5440
🌐 Web query:
AWS EC2 c7i Nitro root filesystem findmnt /dev/nvme0n1p1 diskstats device name💡 Result:
On AWS Nitro-based instances, such as the C7i, EBS volumes are presented as NVMe block devices [1][2]. Because the Linux kernel dynamically enumerates these devices during the boot process based on the order they respond, names like /dev/nvme0n1p1 are not guaranteed to be persistent across reboots or detach/attach events [3][4][2]. To determine the device name for the root filesystem or other mounts, the following approaches are recommended: 1. Use findmnt or lsblk: These commands provide the current, active source device for a mount point [5][6][2]. - To find the source device for the root filesystem: findmnt -no SOURCE / [6] - To list all block devices and their mount points: lsblk [3][7] 2. Avoid Relying on NVMe Paths: Because /dev/nvme* names can change, you should not hardcode them in configuration files like /etc/fstab [7][4]. Instead, use UUIDs (Universally Unique Identifiers) or filesystem labels for persistent mounting [7][4]. You can retrieve a device's UUID using the blkid or lsblk -f commands [7][4]. 3. Identifying Specific EBS Volumes: If you need to map a Linux device name (e.g., /dev/nvme0n1) to a specific AWS EBS volume ID (e.g., vol-xxxxxxxxxxxxxxxxx), you can use the volume's serial number, which AWS sets to the EBS volume ID [8][2]. - Using lsblk: lsblk -o NAME,SERIAL [8][9] - Using nvme-cli: sudo nvme id-ctrl -v /dev/nvme0n1 [5][2] (Note: the output will provide the volume ID without dashes) [2]. For monitoring purposes (e.g., CloudWatch Agent), it is best practice to configure monitoring based on mount points (e.g., /) rather than ephemeral device names to avoid metric gaps if the device name changes [4].
Citations:
🏁 Script executed:
Repository: Shadowfit/init
Length of output: 3562
c7i.2xlarge에서 디스크 폴러의 1분 smoke test를 완료하세요.
실제 EC2 검증은 아직 수행되지 않았고, 장치명은 대상 박스에서만 확정할 수 있습니다. 부하를 시작하기 전에
findmnt -no SOURCE /, 유도된DEV,/proc/diskstats항목을 기록하세요. 값이 지원 형식과 다르면 Line 47의 검사 실패로 폴러가 CSV를 만들기 전에 종료될 수 있습니다. 이 경우DEV를 명시하거나 매핑 규칙을 확장하세요.🤖 Prompt for AI Agents