8.7 配套代码:容量计算、复测与告警
对应小节:8.7 容量规划 三部分:① 容量计算器(含约束检查);② 简化复测脚本;③ 容量告警。
一、容量计算器(含非线性约束检查)
# tools/capacity-planner.py <capacity-input.json>
"""
从拐点计算所需副本数,并检查非线性约束。
输入格式见文件末尾的示例。
"""
import json
import math
import sys
def plan(data):
print("═" * 84)
print(f"容量规划:{data.get('service', 'app')}")
print("═" * 84)
print()
knee = data["knee_qps"]
headroom = data.get("headroom", 0.4)
peak = data["peak_qps"]
redundancy = data.get("redundancy", 1)
spec = data.get("instance_spec", "?")
constraints = data.get("constraints", {})
# ── ① 基础计算 ──────────────────────────────────────────
safe_capacity = int(knee * (1 - headroom))
replicas = math.ceil(peak / safe_capacity) + redundancy
print("【基础计算】")
print(f" 实例规格 : {spec}")
print(f" 拐点(实测) : {knee} RPS")
print(f" headroom : {headroom:.0%}")
print(f" 单实例安全容量 : {safe_capacity} RPS")
print(f" 峰值需求 : {peak} RPS")
print(f" 所需副本(含 {redundancy} 台冗余): {replicas}")
print()
# ── ② 约束检查(非线性因素)────────────────────────────
print("【约束检查】⭐ 这一步常常否掉最优方案")
problems = []
# 数据库连接数
if "db_max_connections" in constraints:
pool_per_instance = constraints.get("pool_per_instance", 30)
total_conn = replicas * pool_per_instance
db_max = constraints["db_max_connections"]
reserve = constraints.get("reserve_connections", 5)
available = db_max - reserve
print(f" 数据库连接数:")
print(f" {replicas} 实例 × {pool_per_instance} 池 = {total_conn}")
print(f" 数据库上限 {db_max}(留 {reserve} 给运维)→ 可用 {available}")
if total_conn > available:
max_pool = available // replicas
problems.append(
f"连接数超出:需要 {total_conn},可用 {available}\n"
f" → 每实例池降到 {max_pool},或引入 PgBouncer,或减少副本数")
print(f" ❌ 超出!每实例池应 ≤ {max_pool}")
else:
print(f" ✅ 余量 {available - total_conn}")
print()
# 共享缓存
if "redis_max_qps" in constraints:
cache_qps_factor = constraints.get("cache_qps_per_request", 2.0)
cache_qps = peak * cache_qps_factor
redis_max = constraints["redis_max_qps"]
print(f" 共享 Redis:")
print(f" 峰值 {peak} × 每请求 {cache_qps_factor} 次 = {cache_qps:.0f} QPS")
print(f" Redis 上限 {redis_max} QPS")
if cache_qps > redis_max * 0.8:
problems.append(f"Redis 接近上限({cache_qps:.0f} / {redis_max})")
print(f" ⚠️ 接近上限")
else:
print(f" ✅ 余量 {redis_max - cache_qps:.0f}")
print()
# 下游服务
if "downstream_qps_limit" in constraints:
factor = constraints.get("downstream_calls_per_request", 1.0)
downstream_qps = peak * factor
limit = constraints["downstream_qps_limit"]
print(f" 下游服务:")
print(f" 峰值 {peak} × 每请求 {factor} 次 = {downstream_qps:.0f} QPS")
print(f" 下游上限 {limit} QPS")
if downstream_qps > limit:
problems.append(f"下游会被打垮({downstream_qps:.0f} > {limit})")
print(f" ❌ 超出!需要限流或缓存")
else:
print(f" ✅")
print()
# 负载均衡上限
if "lb_max_connections" in constraints:
conn_per_instance = constraints.get("connections_per_instance", 1000)
total = replicas * conn_per_instance
lb_max = constraints["lb_max_connections"]
print(f" 负载均衡连接数:{replicas} × {conn_per_instance} = {total}(上限 {lb_max})")
if total > lb_max:
problems.append(f"负载均衡连接数超出({total} > {lb_max})")
print(f" ❌ 超出")
else:
print(f" ✅")
print()
# ── ③ 数据量增长的影响 ──────────────────────────────────
growth = data.get("data_growth_per_quarter", 0.3)
if growth > 0:
print("【数据量增长的影响】")
print(f" 每季度数据增长 {growth:.0%}")
print(f" 经验值:数据量翻倍时拐点下降 10%~20%")
quarters = data.get("planning_horizon_quarters", 4)
# 假设数据每季度增长 growth,拐点按 -15% per doubling 衰减
import math as m
doublings = m.log(1 + growth * quarters, 2) if growth > 0 else 0
knee_after = knee * (1 - 0.15) ** doublings
safe_after = int(knee_after * (1 - headroom))
replicas_after = math.ceil(peak / safe_after) + redundancy
print(f" {quarters} 个季度后:拐点约 {knee_after:.0f} RPS,"
f"所需副本约 {replicas_after} 台(当前 {replicas} 台)")
print(f" → 建议提前规划扩容")
print()
# ── ④ 结论 ──────────────────────────────────────────────
print("═" * 84)
if problems:
print(f"❌ 存在 {len(problems)} 项约束冲突:")
for p in problems:
print(f" - {p}")
print()
print("解决方案(按优先级):")
print(" ① 优化代码/查询(降低单请求资源消耗 → 拐点上升)")
print(" ② 加缓存(降低对数据库/下游的压力)")
print(" ③ 引入连接池代理(PgBouncer 等)")
print(" ④ 提高单实例规格(减少副本数 → 减少总连接数)")
print(" ⑤ 减少副本数 + 接受更低的安全容量(谨慎)")
else:
print("✅ 所有约束检查通过")
print()
print("最终建议:")
print(f" 实例规格 : {spec}")
print(f" 副本数 : {replicas}(含 {redundancy} 台冗余)")
print(f" 安全容量 : {safe_capacity} RPS/实例,共 {safe_capacity * replicas} RPS")
print(f" 峰值利用率: {peak / (safe_capacity * replicas):.0%}")
print("═" * 84)
return 1 if problems else 0
if __name__ == "__main__":
if len(sys.argv) < 2:
print(__doc__)
print()
print("示例输入:")
print(json.dumps({
"service": "orders-api",
"instance_spec": "4C8G",
"knee_qps": 900,
"headroom": 0.4,
"peak_qps": 3000,
"redundancy": 1,
"data_growth_per_quarter": 0.3,
"planning_horizon_quarters": 4,
"constraints": {
"db_max_connections": 100,
"reserve_connections": 5,
"pool_per_instance": 30,
"redis_max_qps": 50000,
"cache_qps_per_request": 2.0,
"downstream_qps_limit": 5000,
"downstream_calls_per_request": 1.5,
"lb_max_connections": 20000,
"connections_per_instance": 1000,
},
}, ensure_ascii=False, indent=2))
sys.exit(0)
data = json.loads(open(sys.argv[1], encoding="utf-8").read())
sys.exit(plan(data))
预期输出(含约束冲突):
【基础计算】
实例规格 : 4C8G
拐点(实测) : 900 RPS
headroom : 40%
单实例安全容量 : 540 RPS
峰值需求 : 3000 RPS
所需副本(含 1 台冗余): 7
【约束检查】⭐ 这一步常常否掉最优方案
数据库连接数:
7 实例 × 30 池 = 210
数据库上限 100(留 5 给运维)→ 可用 95
❌ 超出!每实例池应 ≤ 13
注意:纯按 QPS 算需要 7 台,但数据库连接数只允许每实例 13 个连接——这个约束会彻底改变方案(要么降池,要么加 PgBouncer,要么换更大规格减少副本数)。
二、简化复测脚本
#!/usr/bin/env bash
# tools/capacity-retest.sh <EXP_ID>
#
# 简化版容量复测(30 分钟):在当前安全容量下跑,确认没有退化。
#
# 与完整阶梯加压的区别:
# 完整版:找新拐点(2 小时)
# 简化版:确认"还是安全的"(30 分钟)
set -uo pipefail
EXP_ID="${1:?usage: capacity-retest.sh <exp_id>}"
DIR="docs/experiments/${EXP_ID}"
mkdir -p "$DIR/results"
SAFE_CAPACITY="${SAFE_CAPACITY:?需要设置 SAFE_CAPACITY 环境变量(来自容量表)}"
PEAK_QPS="${PEAK_QPS:-$(python3 -c "print(int($SAFE_CAPACITY * 0.8))")}"
echo "═══════════════════════════════════════════════════════════════"
echo "容量复测(简化版)"
echo " 安全容量: $SAFE_CAPACITY RPS(来自容量表)"
echo " 本次负载: $PEAK_QPS RPS(安全容量的 80%)"
echo "═══════════════════════════════════════════════════════════════"
echo
# ① 环境与元数据
tools/collect-env.sh "$EXP_ID" > /dev/null 2>&1 || true
tools/check-environment.sh "$EXP_ID" > /dev/null 2>&1 || true
{
echo "retest_date=$(date -Iseconds)"
echo "safe_capacity=$SAFE_CAPACITY"
echo "test_qps=$PEAK_QPS"
echo "data_rows=$(psql -tAc 'select count(*) from orders' 2>/dev/null || echo n/a)"
} | tee "$DIR/results/retest-meta.txt"
echo
# ② 重启 + 预热
scripts/restart-app.sh "$DIR/results/gc.log"
sleep 5
BASE_URL="$TARGET_URL" k6 run --quiet --vus 20 --duration 60s loadtest/profile-constant.js > /dev/null
echo "✅ 预热完成"
echo
# ③ 在安全容量的 80% 下跑 10 分钟
echo "运行 10 分钟..."
BASE_URL="$TARGET_URL" RATE="$PEAK_QPS" DURATION=10m \
k6 run --summary-export="$DIR/results/k6-retest.json" \
loadtest/profile-constant.js > "$DIR/results/k6-retest-stdout.txt" 2>&1
# ④ 采集饱和度
curl -s --max-time 5 "${TARGET_URL%/}/metrics" > "$DIR/results/metrics-retest.txt" 2>/dev/null || true
# ⑤ 判定
echo
echo "═══ 判定 ═══"
python3 - "$DIR/results" "$SAFE_CAPACITY" <<'PY'
import json, pathlib, re, sys
d, safe = pathlib.Path(sys.argv[1]), int(sys.argv[2])
k6 = json.loads((d / "k6-retest.json").read_text())["metrics"]
dur = k6["http_req_duration"]
metrics = (d / "metrics-retest.txt").read_text(errors="ignore") \
if (d / "metrics-retest.txt").exists() else ""
def extract(pattern):
m = re.search(pattern, metrics, re.MULTILINE)
return float(m.group(1)) if m else None
pending = extract(r"^db_pool_pending\s+([\d.]+)")
queue = extract(r"^executor_queue_depth\s+([\d.]+)")
gc_p99 = extract(r"^jvm_gc_pause_seconds.*\s+([\d.]+)")
print(f" P50 : {dur['med']:.1f} ms")
print(f" P95 : {dur['p(95)']:.1f} ms")
print(f" P99 : {dur['p(99)']:.1f} ms")
print(f" 错误率 : {k6['http_req_failed']['rate']:.2%}")
print(f" 实测 QPS : {k6['http_reqs']['rate']:.1f}(设定 {safe * 0.8:.0f})")
print(f" 连接池 pending: {pending if pending is not None else 'n/a'}")
print(f" 队列深度 : {queue if queue is not None else 'n/a'}")
print()
problems = []
if dur["p(99)"] > 200:
problems.append(f"P99 超标({dur['p(99)']:.1f}ms > 200ms)")
if k6["http_req_failed"]["rate"] > 0.001:
problems.append(f"错误率超标({k6['http_req_failed']['rate']:.2%})")
if pending is not None and pending > 0:
problems.append(f"连接池排队(pending={pending:.0f})")
if queue is not None and queue > 0:
problems.append(f"队列积压(depth={queue:.0f})")
print("═" * 70)
if problems:
print(f"❌ 容量可能已退化({len(problems)} 项)")
for p in problems:
print(f" - {p}")
print()
print("下一步:")
print(" ① 跑完整的阶梯加压找新拐点(tools/retest-knee.sh)")
print(" ② 或按第 6 章的流程定位原因")
print(" ③ 更新容量表,评估是否需要扩容")
else:
print("✅ 在安全容量的 80% 下指标正常")
print(" → 容量没有明显退化,本次复测通过")
print("═" * 70)
PY
echo
echo "✅ 复测完成 → $DIR/results"
echo
echo "记录进容量表:"
echo " | $(grep '^safe_capacity=' "$DIR/results/retest-meta.txt" | cut -d= -f2) RPS | 复测通过 | $(date +%Y-%m-%d) |"
三、容量表模板
<!-- docs/capacity.md -->
# 容量表
> 最后更新:YYYY-MM-DD | 下次复测:YYYY-MM-DD(建议每季度)
## 服务:orders-api
| 实例规格 | 拐点(实测) | headroom | 安全容量 | 峰值需求 | 所需副本 | 数据来源 | 测量日期 |
| --- | --- | --- | --- | --- | --- | --- | --- |
| 4C8G | 900 RPS | 40% | 540 RPS | 3000 | 7 | E02 | 2025-01-10 |
| 8C16G | 1600 RPS | 40% | 960 RPS | 3000 | 4 | E05 | 2025-01-15 |
**当前采用**:8C16G × 4(含 1 台冗余 = 5 台)
## 约束条件
| 约束 | 上限 | 当前使用 | 余量 | 备注 |
| --- | --- | --- | --- | --- |
| 数据库 max_connections | 100 | 4 × 20 = 80 | 20 | 已引入 PgBouncer |
| Redis QPS | 50000 | 30000 | 20000 | — |
| 下游 user-service QPS | 5000 | 3200 | 1800 | 需放缓 |
| 负载均衡连接数 | 20000 | 4000 | 16000 | — |
## 月度复测记录
| 日期 | 测试负载 | P99 | pending | 结论 | 备注 |
| --- | --- | --- | --- | --- | --- |
| 2025-01-15 | 768 RPS | 92 ms | 0 | ✅ 通过 | 建立基线 |
| 2025-02-15 | 768 RPS | 98 ms | 0 | ✅ 通过 | P99 +6%(噪声内) |
| 2025-03-15 | 768 RPS | 118 ms | 3 | ⚠️ 关注 | pending 出现,需要排查 |
| 2025-04-15 | 768 RPS | 145 ms | 12 | ❌ 失败 | 数据量翻倍导致拐点下移 |
## 数据量趋势
| 日期 | orders 行数 | 相对基线 |
| --- | --- | --- |
| 2025-01-15 | 120 万 | 1.0× |
| 2025-04-15 | 280 万 | 2.3× |
**观察**:数据量翻倍时,拐点下降了约 20%(900 → 720 RPS),
与经验值(每翻倍降 10%~20%)一致。
## 扩容决策记录
| 日期 | 决策 | 依据 | 成本影响 |
| --- | --- | --- | --- |
| 2025-01-15 | 采用 8C16G × 5 | E05 实验:4 台比 7 台更省 | 月成本 -15% |
| 2025-04-20 | 从 5 台扩到 7 台 | 4 月复测失败,拐点下移 | 月成本 +40% |
四、动手改造
| 改动 | 观察什么 |
|---|---|
用 capacity-planner.py 输入你的数据 |
看约束检查会不会否掉纯 QPS 算出的方案 |
把 pool_per_instance 从 30 改成 10 |
连接数约束满足——理解"降池"这个选项 |
把 instance_spec 换成更大的规格 |
副本数减少 → 总连接数减少——“换大规格"是解决连接数约束的手段 |
用 capacity-retest.sh 跑一次季度复测 |
30 分钟得到"容量是否退化"的结论 |
| 在容量表里记录数据量趋势 | 观察拐点与数据量的关系 |
五、这段代码的局限
data_growth_per_quarter与「拐点下降 15% per doubling」是经验值:应该用你自己的历史数据校准(对比不同数据量下的容量曲线)。capacity-retest.sh需要SAFE_CAPACITY环境变量:它来自容量表,需要手工维护。- 简化复测只能确认"还是安全的”,不能发现"容量提升了"(如果优化让拐点上升,简化版看不出来)。
- 约束检查需要人工提供的数据(
db_max_connections等):如果填错,检查结果也会错。 - 容量表需要人工维护:可以做成 Markdown(可 review)或结构化文件(可自动检查),但都要有人负责更新。