7.4 配套代码:缓存的三重防护与代价评估
对应小节:7.4 缓存 一份可直接抄用的 Caffeine 实现(含三重防护),以及缓存对 P99 影响的计算工具。
一、完整的缓存实现(含三重防护)
// src/main/kotlin/cache/OrderCache.kt
package cache
import com.github.benmanes.caffeine.cache.AsyncCache
import com.github.benmanes.caffeine.cache.Caffeine
import com.github.benmanes.caffeine.cache.Expiry
import com.github.benmanes.caffeine.cache.stats.CacheStats
import io.micrometer.core.instrument.Gauge
import io.micrometer.core.instrument.MeterRegistry
import kotlinx.coroutines.future.await
import java.time.Duration
import java.util.concurrent.CompletableFuture
import java.util.concurrent.Executor
import kotlin.random.Random
data class Order(val id: Long, val userId: Long, val amount: Long)
/** 空值哨兵:防穿透 */
private val NULL_SENTINEL = Order(id = -1, userId = -1, amount = -1)
class OrderCache(
private val loader: (Long) -> Order?, // 回源函数(阻塞)
private val executor: Executor, // 回源用的线程池
registry: MeterRegistry,
) {
/**
* 用 AsyncCache 而不是同步 cache:
* ① get(key, mapping) 是原子的 → 天然「单飞」(防击穿)
* ② 通过 CompletableFuture.await() 桥接到协程
*
* 三重防护中的两重在这里:
* 防护一(单飞):Caffeine 的原子 get
* 防护二(TTL 抖动):下面的 Expiry 实现
* 防护三(空值缓存):NULL_SENTINEL
*/
private val cache: AsyncCache<Long, Order> = Caffeine.newBuilder()
.maximumSize(100_000) // ⭐ 必须有界(防 OOM)
.expireAfter(expireWithJitter()) // ⭐ TTL + 抖动(防雪崩)
.recordStats() // ⭐ 必须有指标
.buildAsync()
init {
// 命中率(最核心的指标)
Gauge.builder("app.cache.hit_rate") { cache.synchronous().stats().hitRate() }
.description("缓存命中率")
.register(registry)
// 未命中时的延迟(关键:只看命中率不够)
Gauge.builder("app.cache.miss_latency_p99") { missLatencyP99.get().toDouble() }
.description("未命中时的 P99 延迟(ms)")
.register(registry)
// 缓存大小与驱逐
Gauge.builder("app.cache.size") { cache.synchronous().estimatedSize().toDouble() }
.register(registry)
Gauge.builder("app.cache.evictions") { cache.synchronous().stats().evictionCount().toDouble() }
.register(registry)
}
/** 防护二:TTL 加随机抖动(±20%),避免大批 key 同时失效 */
private fun expireWithJitter(): Expiry<Long, Order> =
object : Expiry<Long, Order> {
private val baseMs = Duration.ofMinutes(10).toMillis()
private fun ttl() = baseMs + Random.nextLong(-baseMs / 5, baseMs / 5)
override fun expireAfterCreate(key: Long, value: Order, currentTime: Long) = ttl()
override fun expireAfterUpdate(key: Long, value: Order, currentTime: Long, currentDuration: Long) = ttl()
override fun expireAfterRead(key: Long, value: Order, currentTime: Long, currentDuration: Long) = currentDuration
}
private val missLatencyP99 = java.util.concurrent.atomic.AtomicLong(0)
/**
* 读(含三重防护)
*
* 防护一(单飞):cache.get(key) { ... } 保证同一 key 的并发回源只执行一次
*/
suspend fun get(id: Long): Order? {
val result = cache.get(id) { k, _ ->
CompletableFuture.supplyAsync({
val t0 = System.nanoTime()
val value = loader(k) ?: NULL_SENTINEL // 防护三:缓存空值
missLatencyP99.set((System.nanoTime() - t0) / 1_000_000)
value
}, executor)
}.await()
return if (result === NULL_SENTINEL) null else result
}
/** 写:先更新数据库,再删缓存(Cache-Aside 的正确顺序) */
suspend fun invalidate(id: Long) {
cache.synchronous().invalidate(id)
}
fun stats(): CacheStats = cache.synchronous().stats()
/** 供测试/监控使用 */
fun estimatedSize(): Long = cache.synchronous().estimatedSize()
}
使用示例:
val cache = OrderCache(loader = { id -> repo.findById(id) }, executor = ioExecutor, registry = registry)
// 读
val order = cache.get(123L)
// 写(注意顺序:先更新数据库,再删缓存)
suspend fun updateOrder(order: Order) {
repo.update(order) // ① 先更新数据库
cache.invalidate(order.id) // ② 再删缓存
}
二、缓存对 P99 的影响计算
# tools/cache-p99-impact.py
"""
计算缓存对 P50/P99 的影响。
核心公式(简化模型):
设命中率为 h,命中延迟为 t_hit,未命中延迟为 t_miss_effective
(t_miss_effective = 回源延迟 + 缓存查询开销)
P50 ≈ t_hit (如果 h > 50%)
P99 ≈ t_hit if h > 99% (P99 落在命中区间)
≈ t_miss_effective (如果 h < 99%)
"""
import sys
def analyze(hit_rate, t_hit_ms, t_miss_ms, t_no_cache_ms):
"""
hit_rate: 命中率 (0~1)
t_hit_ms: 命中时的延迟(缓存查询)
t_miss_ms: 未命中时的延迟(缓存查询 + 回源 + 写缓存)
t_no_cache_ms: 不用缓存时的延迟(直接回源)
"""
miss_rate = 1 - hit_rate
print("═" * 74)
print("缓存对延迟分布的影响")
print("═" * 74)
print()
print(f"命中率 : {hit_rate:.1%}")
print(f"命中延迟 : {t_hit_ms:.2f} ms")
print(f"未命中延迟 : {t_miss_ms:.2f} ms(含缓存开销,比不用缓存更慢)")
print(f"不用缓存的延迟 : {t_no_cache_ms:.2f} ms")
print()
# 各分位数的估算
def estimate(q):
# 如果未命中比例 > (1-q),则该分位数落在未命中区间
return t_miss_ms if miss_rate > (1 - q) else t_hit_ms
print(f"{'分位数':<10}{'有缓存':>12}{'无缓存':>12}{'改善':>12} 说明")
print("-" * 74)
for q, label in [(0.50, "P50"), (0.90, "P90"), (0.95, "P95"), (0.99, "P99"), (0.999, "P999")]:
with_cache = estimate(q)
improvement = (t_no_cache_ms - with_cache) / t_no_cache_ms * 100
note = "落在命中区间" if miss_rate <= (1 - q) else "⚠️ 落在未命中区间"
print(f"{label:<10}{with_cache:>11.2f}ms{t_no_cache_ms:>11.2f}ms{improvement:>11.1f}% {note}")
print()
print("═" * 74)
print("关键结论:")
print()
# 判断 P99 是否改善
p99_with = estimate(0.99)
if p99_with >= t_miss_ms - 0.01:
required = 0.99
print(f"❌ P99 未改善({p99_with:.1f}ms,与不用缓存的 {t_no_cache_ms:.1f}ms 相比改善有限)")
print(f" 原因:未命中率 {miss_rate:.1%} > 1%,P99 落在未命中的那批请求上")
print(f" → 要让 P99 改善,命中率必须 > {required:.0%}")
print(f" → 或者降低未命中延迟({t_miss_ms:.1f}ms)")
else:
print(f"✅ P99 改善到 {p99_with:.1f}ms(提升 {(t_no_cache_ms-p99_with)/t_no_cache_ms*100:.0f}%)")
print()
print("改进方向:")
print(" ① 提高命中率:加大容量、延长 TTL、优化 key 设计、加 L1 本地缓存")
print(f" 当前命中率 {hit_rate:.0%},每提高 1% 都有帮助(尤其接近 99% 时)")
print(" ② 降低未命中延迟:")
print(" - 优化回源查询(加索引、消除 N+1)")
print(" - 三防护(单飞 / TTL 抖动 / 空值)")
print(" - 异步写缓存(不阻塞返回)")
print(" ③ 如果以上都不行 → 缓存对这个接口可能【不值得】")
print("═" * 74)
if __name__ == "__main__":
# 示例:命中率 85%,命中 0.5ms,未命中 200ms,无缓存 180ms
analyze(hit_rate=0.85, t_hit_ms=0.5, t_miss_ms=200.0, t_no_cache_ms=180.0)
print()
print()
# 对比:命中率提升到 99.5%
analyze(hit_rate=0.995, t_hit_ms=0.5, t_miss_ms=200.0, t_no_cache_ms=180.0)
预期输出(85% 命中率):
分位数 有缓存 无缓存 改善 说明
--------------------------------------------------------------------------
P50 0.50ms 180.00ms 99.7% 落在命中区间
P90 200.00ms 180.00ms -11.1% ⚠️ 落在未命中区间
P95 200.00ms 180.00ms -11.1% ⚠️ 落在未命中区间
P99 200.00ms 180.00ms -11.1% ⚠️ 落在未命中区间
P999 200.00ms 180.00ms -11.1% ⚠️ 落在未命中区间
❌ P99 未改善(200.0ms,与不用缓存的 180.0ms 相比改善有限)
看到关键了吗:P50 改善 99.7%,但 P90 以上全部恶化 11%——这就是「缓存改善 P50 但可能让 P99 变差」的量化证据。
对比 99.5% 命中率:
P50 0.50ms 180.00ms 99.7% 落在命中区间
P90 0.50ms 180.00ms 99.7% 落在命中区间
P95 0.50ms 180.00ms 99.7% 落在命中区间
P99 0.50ms 180.00ms 99.7% 落在命中区间 ← 这时才改善 P99
P999 200.00ms 180.00ms -11.1% ⚠️ 落在未命中区间
三、判断「该不该用缓存」
# tools/should-cache.py
"""
判断一个接口是否值得加缓存。
需要回答四个问题——任何一个答"否"都可能否决缓存方案。
"""
import sys
def should_cache(
read_write_ratio: float, # 读写比(读/写)
estimated_hit_rate: float, # 预估命中率
miss_latency_ms: float, # 未命中延迟
consistency_tolerance_s: int, # 可容忍的不一致时长(秒)
current_p99_ms: float, # 当前 P99
slo_p99_ms: float, # SLO 目标
):
print("═" * 74)
print("该不该加缓存?")
print("═" * 74)
print()
verdict = []
print(f"① 读写比 : {read_write_ratio:.1f} : 1")
if read_write_ratio > 10:
print(" ✅ 读多写少,缓存有意义")
elif read_write_ratio > 3:
print(" ⚠️ 读写比一般,缓存收益可能有限")
else:
verdict.append("读写比过低(缓存刚写入就被更新,命中率低)")
print(" ❌ 写多读少,缓存意义不大")
print()
print(f"② 预估命中率 : {estimated_hit_rate:.1%}")
if estimated_hit_rate > 0.99:
print(" ✅ 命中率足够高,能改善 P99")
elif estimated_hit_rate > 0.9:
print(" ⚠️ 能改善 P50/P90,但 P99 可能改善有限")
elif estimated_hit_rate > 0.7:
print(" ⚠️ 只改善 P50;P99 可能反而变差")
else:
verdict.append(f"命中率过低({estimated_hit_rate:.0%}),P99 不会改善")
print(" ❌ 命中率太低,不值得")
print()
print(f"③ 可容忍不一致时长 : {consistency_tolerance_s} 秒")
if consistency_tolerance_s >= 60:
print(" ✅ 一致性要求宽松,缓存风险低")
elif consistency_tolerance_s >= 5:
print(" ⚠️ 需要短 TTL + 主动失效(复杂度上升)")
else:
verdict.append("一致性要求强,缓存的复杂度不值得")
print(" ❌ 一致性要求太强,缓存风险高")
print()
print(f"④ 当前 P99 / SLO : {current_p99_ms:.0f} / {slo_p99_ms:.0f} ms")
gap = (current_p99_ms - slo_p99_ms) / slo_p99_ms * 100
if current_p99_ms > slo_p99_ms:
print(f" ✅ 超标 {gap:.0f}%,有优化动力")
else:
verdict.append("当前已达标,收益不足以覆盖复杂度")
print(" ❌ 已达标,不值得引入复杂度")
print()
print("═" * 74)
if verdict:
print("❌ 建议:【暂不加缓存】")
print()
print("否决理由:")
for v in verdict:
print(f" - {v}")
print()
print("替代方向(优先级更高):")
print(" ① 先确认是不是 N+1 / 缺索引(第 7.2 节,收益更确定)")
print(" ② 减少返回字段(少传数据,三重收益)")
print(" ③ 如果确实需要缓存,考虑只缓存最热的那部分数据(而不是全部)")
else:
print("✅ 建议:【可以加缓存】")
print()
print("实施要求:")
print(" ① 必须有界(maximumSize)")
print(" ② 必须做三重防护(单飞 / TTL 抖动 / 空值缓存)")
print(" ③ 必须监控命中率 + 【未命中时的延迟】")
print(" ④ 必须有降级方案(缓存挂了直接查库)")
print(" ⑤ 先更新数据库,再删缓存")
print("═" * 74)
if __name__ == "__main__":
print("示例 1:典型的『该加缓存』")
should_cache(read_write_ratio=50, estimated_hit_rate=0.995,
miss_latency_ms=200, consistency_tolerance_s=300,
current_p99_ms=400, slo_p99_ms=200)
print()
print()
print("示例 2:典型的『不该加缓存』")
should_cache(read_write_ratio=2, estimated_hit_rate=0.4,
miss_latency_ms=200, consistency_tolerance_s=1,
current_p99_ms=95, slo_p99_ms=200)
四、动手改造
| 改动 | 观察什么 |
|---|---|
用 cache-p99-impact.py 试不同命中率(0.5 / 0.9 / 0.99 / 0.999) |
找到「P99 开始改善」的命中率阈值 |
在 OrderCache 里去掉 TTL 抖动 |
用压测观察:大批 key 同时失效时 DB QPS 的脉冲 |
去掉单飞(改成 getIfPresent + put) |
并发回源的数量会飙升(用 DB 并发查询数验证) |
把 maximumSize 改成 Integer.MAX_VALUE(无界) |
长时间运行后内存会持续增长(第 3 章 3.5 的浸泡分析) |
给 Lab 6 的 /slow-db 跑 should-cache.py |
看它是否会建议「暂不加缓存」(因为批量查询已经解决) |
五、这段代码的局限
cache-p99-impact.py用的是简化模型:它假设所有未命中请求的延迟相同,真实分布更复杂(重尾)。它适合做量级判断,不适合精确预测。- 命中率的预估很难准:
should-cache.py需要你提供estimated_hit_rate,而这个数字通常要靠生产日志的真实分布来估(第 5.5 节)。 OrderCache混用了同步与异步 API:cache.get()是异步的(返回CompletableFuture),而stats()用的是synchronous().stats()——这是 Caffeine 的AsyncCache的正常用法。- TTL 抖动的实现用了
Math.random:在高并发下会有锁竞争,生产可以用ThreadLocalRandom。 - 没有实现「延迟双删」:对于强一致性要求的场景,
先更新库再删缓存仍有极小窗口的不一致,需要额外的保护。