6.2 配套代码:延迟分解的 PromQL 与自动计算
对应小节:6.2 延迟分解 把「三段减法」变成可执行的查询与脚本。
一、三段分解的 PromQL
-- ═══════════════════════════════════════════════════════════
-- ① 服务端完整延迟(进入 handler 到响应写完)
-- ═══════════════════════════════════════════════════════════
histogram_quantile(0.99,
sum by (le) (rate(http_server_requests_seconds_bucket{uri="/orders/{id}"}[5m])))
-- ═══════════════════════════════════════════════════════════
-- ② 应用内延迟(handler 内部)
-- ═══════════════════════════════════════════════════════════
histogram_quantile(0.99,
sum by (le) (rate(app_handler_duration_seconds_bucket[5m])))
-- ═══════════════════════════════════════════════════════════
-- ③ 服务排队 = ① − ②
-- ═══════════════════════════════════════════════════════════
histogram_quantile(0.99,
sum by (le) (rate(http_server_requests_seconds_bucket{uri="/orders/{id}"}[5m])))
-
histogram_quantile(0.99,
sum by (le) (rate(app_handler_duration_seconds_bucket[5m])))
-- ═══════════════════════════════════════════════════════════
-- ④ 各依赖
-- ═══════════════════════════════════════════════════════════
histogram_quantile(0.99, sum by (le) (rate(dep_postgres_seconds_bucket[5m])))
histogram_quantile(0.99, sum by (le) (rate(dep_redis_seconds_bucket[5m])))
histogram_quantile(0.99, sum by (le) (rate(dep_downstream_seconds_bucket[5m])))
-- ═══════════════════════════════════════════════════════════
-- ⑤ 应用自身开销 = ② − Σ依赖
-- ⚠️ 只在【串行调用】时成立;并行调用会算出负数
-- ═══════════════════════════════════════════════════════════
histogram_quantile(0.99, sum by (le) (rate(app_handler_duration_seconds_bucket[5m])))
-
(
histogram_quantile(0.99, sum by (le) (rate(dep_postgres_seconds_bucket[5m])))
+
histogram_quantile(0.99, sum by (le) (rate(dep_redis_seconds_bucket[5m])))
+
histogram_quantile(0.99, sum by (le) (rate(dep_downstream_seconds_bucket[5m])))
)
⚠️ 重要提醒:分位数相减是近似(第 6.2 节)。它适合判断「哪段是大头」,不适合精确归因。精确归因需要 tracing 的 span 树。
二、自动分解脚本(从 Prometheus 拉数据)
# tools/decompose-latency.py <PROMETHEUS_URL> <URI_LABEL> [--percentile 0.99]
"""
从 Prometheus 拉取各段延迟,生成分解表。
用法:
python3 tools/decompose-latency.py http://localhost:9090 '/orders/{id}'
"""
import json
import sys
import urllib.parse
import urllib.request
def query(prom_url, expr):
url = f"{prom_url}/api/v1/query?" + urllib.parse.urlencode({"query": expr})
try:
with urllib.request.urlopen(url, timeout=10) as r:
data = json.load(r)
if data["status"] != "success":
return None
result = data["data"]["result"]
return float(result[0]["value"][1]) if result else None
except Exception as e:
print(f" ⚠️ 查询失败: {e}", file=sys.stderr)
return None
def hq(metric, selector="", q=0.99):
return (f'histogram_quantile({q}, sum by (le) '
f'(rate({metric}_bucket{selector}[5m])))')
def main():
prom = sys.argv[1] if len(sys.argv) > 1 else "http://localhost:9090"
uri = sys.argv[2] if len(sys.argv) > 2 else ""
q = 0.99
if "--percentile" in sys.argv:
q = float(sys.argv[sys.argv.index("--percentile") + 1])
print("═" * 70)
print(f"延迟分解(P{q*100:.0f},URI={uri or '全部'})")
print("═" * 70)
print()
selector = f'{{uri="{uri}"}}' if uri else ""
segments = {
"① 服务端完整延迟": hq("http_server_requests_seconds", selector, q),
"② 应用内延迟": hq("app_handler_duration_seconds", "", q),
"③ dep.postgres": hq("dep_postgres_seconds", "", q),
"④ dep.redis": hq("dep_redis_seconds", "", q),
"⑤ dep.downstream": hq("dep_downstream_seconds", "", q),
}
values = {}
for label, expr in segments.items():
v = query(prom, expr)
values[label] = v
server = values["① 服务端完整延迟"]
inapp = values["② 应用内延迟"]
deps = [values["③ dep.postgres"], values["④ dep.redis"], values["⑤ dep.downstream"]]
def fmt(v):
return f"{v*1000:.1f} ms" if v is not None else "n/a"
# 打印各段
print(f"{'段':<22}{'P' + str(int(q*100)):<10}{'占比'}")
print("-" * 70)
print(f"{'① 服务端完整延迟':<22}{fmt(server):<12}100%")
print(f"{'② 应用内延迟':<22}{fmt(inapp):<12}"
f"{(inapp/server*100):.1f}%" if server and inapp else "n/a")
if server and inapp:
queued = server - inapp
print(f"{'③ 服务排队(①−②)':<22}{fmt(queued):<12}{queued/server*100:.1f}%")
for label, v in zip(["④ dep.postgres", "⑤ dep.redis", "⑥ dep.downstream"], deps):
if v is not None and server:
print(f"{label:<22}{fmt(v):<12}{v/server*100:.1f}%")
# 应用自身开销
if inapp and all(d is not None for d in deps):
dep_sum = sum(deps)
self_cost = inapp - dep_sum
print("-" * 70)
if self_cost < 0:
print(f"{'⑦ 应用自身(②−Σ依赖)':<22}{fmt(self_cost):<12}"
f"❌ 负数!说明链路中有并行调用,不能简单相减")
print(" → 用 tracing 的 span 树看关键路径(第 6.2 节)")
else:
print(f"{'⑦ 应用自身(②−Σ依赖)':<22}{fmt(self_cost):<12}"
f"{self_cost/server*100:.1f}%")
print()
print("═" * 70)
print("判读:")
print(" 哪一段占比最大,就从那一段开始排查。")
print(" ⚠️ 分位数相减是近似,只用于判断「哪段是大头」。")
print(" 精确归因请用 tracing span 树(同一批请求的各段耗时)。")
print("═" * 70)
if __name__ == "__main__":
main()
三、基于 tracing 的精确分解(推荐)
分位数相减是近似。精确做法是用 tracing:每个请求的 span 树里都有各段的真实耗时。
# tools/decompose-from-traces.py <spans.json>
"""
从 tracing 导出的 span 数据做精确分解。
输入格式(简化):每个请求一行 JSON
{"trace_id": "...", "spans": [
{"name": "GET /orders/{id}", "duration_ms": 400, "parent": null},
{"name": "db.query.orders", "duration_ms": 65, "parent": "GET /orders/{id}"},
...]}
"""
import json
import statistics as st
import sys
def analyze(trace):
"""把一个 trace 的各段耗时提取出来"""
spans = {s["name"]: s for s in trace["spans"]}
root = next((s for s in trace["spans"] if s["parent"] is None), None)
if not root:
return None
# 各直接子 span 的耗时
children = [s for s in trace["spans"] if s["parent"] == root["name"]]
child_sum = sum(s["duration_ms"] for s in children)
return {
"total": root["duration_ms"],
"deps": {s["name"]: s["duration_ms"] for s in children},
"dep_sum": child_sum,
"self_cost": root["duration_ms"] - child_sum, # 精确的自身开销
}
def pctl(xs, q):
s = sorted(xs)
return s[min(int(len(s) * q), len(s) - 1)]
def main(path):
traces = [json.loads(line) for line in open(path) if line.strip()]
results = [r for r in (analyze(t) for t in traces) if r]
if not results:
print("❌ 没有解析到有效的 trace")
return
print("═" * 74)
print(f"基于 tracing 的精确延迟分解({len(results)} 个请求)")
print("═" * 74)
print()
totals = [r["total"] for r in results]
self_costs = [r["self_cost"] for r in results]
print(f"{'段':<26}{'P50':>10}{'P95':>10}{'P99':>10}")
print("-" * 74)
print(f"{'总延迟':<26}{pctl(totals,.5):>10.1f}"
f"{pctl(totals,.95):>10.1f}{pctl(totals,.99):>10.1f}")
# 各依赖
dep_names = set()
for r in results:
dep_names.update(r["deps"].keys())
for name in sorted(dep_names):
vals = [r["deps"].get(name, 0) for r in results]
print(f"{name:<26}{pctl(vals,.5):>10.1f}{pctl(vals,.95):>10.1f}{pctl(vals,.99):>10.1f}")
print("-" * 74)
print(f"{'应用自身开销(精确)':<26}{pctl(self_costs,.5):>10.1f}"
f"{pctl(self_costs,.95):>10.1f}{pctl(self_costs,.99):>10.1f}")
print()
print("说明:这是【精确】的自身开销(每个请求各自相减),")
print(" 不像分位数相减那样是近似。")
print(" 如果有负数,说明该请求里有并行调用(span 之和 > 总耗时)。")
neg = sum(1 for c in self_costs if c < 0)
if neg:
print(f" ⚠️ {neg}/{len(results)} 个请求出现负数 → 存在并行调用")
print(f" 此时应该看 span 树的关键路径,而不是简单相减")
if __name__ == "__main__":
main(sys.argv[1] if len(sys.argv) > 1 else "traces.json")
四、动手改造
| 改动 | 观察什么 |
|---|---|
对并行调用的接口跑 decompose-latency.py |
会出现负数——这就是"并行不能相减"的实证 |
对比 decompose-latency.py(近似)与 decompose-from-traces.py(精确) |
两者的差距有多大?(取决于各段 P99 是否落在同一批请求上) |
把 --percentile 从 0.99 改成 0.5 |
分解结果会变——因为不同分位数上的瓶颈可能不同 |
| 在 tracing 里给每个依赖加更细的 span | 分解表会更细(例如把 db.query 拆成"等连接"和"执行") |
五、这段代码的局限
decompose-latency.py依赖 Prometheus 的指标命名:如果你的依赖指标名不同,需要改脚本。- 分位数相减的误差在高分位数上更大:P99 以上的样本少,两个分位数很可能是不同的请求(第 6.2 节)。
- tracing 只有在采样率足够时才有代表性:如果只采了 1%,样本可能不含最慢的那些请求(第 4 章 4.6 节)。
- span 树需要正确的埋点:如果依赖调用没有独立的 span,就无法分解。