文档目录

7.4 配套代码:缓存的三重防护与代价评估

对应小节:7.4 缓存 一份可直接抄用的 Caffeine 实现(含三重防护),以及缓存对 P99 影响的计算工具。

一、完整的缓存实现(含三重防护)

// src/main/kotlin/cache/OrderCache.kt
package cache

import com.github.benmanes.caffeine.cache.AsyncCache
import com.github.benmanes.caffeine.cache.Caffeine
import com.github.benmanes.caffeine.cache.Expiry
import com.github.benmanes.caffeine.cache.stats.CacheStats
import io.micrometer.core.instrument.Gauge
import io.micrometer.core.instrument.MeterRegistry
import kotlinx.coroutines.future.await
import java.time.Duration
import java.util.concurrent.CompletableFuture
import java.util.concurrent.Executor
import kotlin.random.Random

data class Order(val id: Long, val userId: Long, val amount: Long)

/** 空值哨兵:防穿透 */
private val NULL_SENTINEL = Order(id = -1, userId = -1, amount = -1)

class OrderCache(
    private val loader: (Long) -> Order?,          // 回源函数(阻塞)
    private val executor: Executor,                // 回源用的线程池
    registry: MeterRegistry,
) {
    /**
     * 用 AsyncCache 而不是同步 cache:
     *   ① get(key, mapping) 是原子的 → 天然「单飞」(防击穿)
     *   ② 通过 CompletableFuture.await() 桥接到协程
     *
     * 三重防护中的两重在这里:
     *   防护一(单飞):Caffeine 的原子 get
     *   防护二(TTL 抖动):下面的 Expiry 实现
     *   防护三(空值缓存):NULL_SENTINEL
     */
    private val cache: AsyncCache<Long, Order> = Caffeine.newBuilder()
        .maximumSize(100_000)                       // ⭐ 必须有界(防 OOM)
        .expireAfter(expireWithJitter())            // ⭐ TTL + 抖动(防雪崩)
        .recordStats()                              // ⭐ 必须有指标
        .buildAsync()

    init {
        // 命中率(最核心的指标)
        Gauge.builder("app.cache.hit_rate") { cache.synchronous().stats().hitRate() }
            .description("缓存命中率")
            .register(registry)

        // 未命中时的延迟(关键:只看命中率不够)
        Gauge.builder("app.cache.miss_latency_p99") { missLatencyP99.get().toDouble() }
            .description("未命中时的 P99 延迟(ms)")
            .register(registry)

        // 缓存大小与驱逐
        Gauge.builder("app.cache.size") { cache.synchronous().estimatedSize().toDouble() }
            .register(registry)
        Gauge.builder("app.cache.evictions") { cache.synchronous().stats().evictionCount().toDouble() }
            .register(registry)
    }

    /** 防护二:TTL 加随机抖动(±20%),避免大批 key 同时失效 */
    private fun expireWithJitter(): Expiry<Long, Order> =
        object : Expiry<Long, Order> {
            private val baseMs = Duration.ofMinutes(10).toMillis()
            private fun ttl() = baseMs + Random.nextLong(-baseMs / 5, baseMs / 5)

            override fun expireAfterCreate(key: Long, value: Order, currentTime: Long) = ttl()
            override fun expireAfterUpdate(key: Long, value: Order, currentTime: Long, currentDuration: Long) = ttl()
            override fun expireAfterRead(key: Long, value: Order, currentTime: Long, currentDuration: Long) = currentDuration
        }

    private val missLatencyP99 = java.util.concurrent.atomic.AtomicLong(0)

    /**
     * 读(含三重防护)
     *
     * 防护一(单飞):cache.get(key) { ... } 保证同一 key 的并发回源只执行一次
     */
    suspend fun get(id: Long): Order? {
        val result = cache.get(id) { k, _ ->
            CompletableFuture.supplyAsync({
                val t0 = System.nanoTime()
                val value = loader(k) ?: NULL_SENTINEL     // 防护三:缓存空值
                missLatencyP99.set((System.nanoTime() - t0) / 1_000_000)
                value
            }, executor)
        }.await()

        return if (result === NULL_SENTINEL) null else result
    }

    /** 写:先更新数据库,再删缓存(Cache-Aside 的正确顺序) */
    suspend fun invalidate(id: Long) {
        cache.synchronous().invalidate(id)
    }

    fun stats(): CacheStats = cache.synchronous().stats()

    /** 供测试/监控使用 */
    fun estimatedSize(): Long = cache.synchronous().estimatedSize()
}

使用示例:

val cache = OrderCache(loader = { id -> repo.findById(id) }, executor = ioExecutor, registry = registry)

// 读
val order = cache.get(123L)

// 写(注意顺序:先更新数据库,再删缓存)
suspend fun updateOrder(order: Order) {
    repo.update(order)        // ① 先更新数据库
    cache.invalidate(order.id) // ② 再删缓存
}

二、缓存对 P99 的影响计算

# tools/cache-p99-impact.py
"""
计算缓存对 P50/P99 的影响。

核心公式(简化模型):
  设命中率为 h,命中延迟为 t_hit,未命中延迟为 t_miss_effective
  (t_miss_effective = 回源延迟 + 缓存查询开销)

  P50 ≈ t_hit                (如果 h > 50%)
  P99 ≈ t_hit  if h > 99%    (P99 落在命中区间)
      ≈ t_miss_effective     (如果 h < 99%)
"""
import sys


def analyze(hit_rate, t_hit_ms, t_miss_ms, t_no_cache_ms):
    """
    hit_rate: 命中率 (0~1)
    t_hit_ms: 命中时的延迟(缓存查询)
    t_miss_ms: 未命中时的延迟(缓存查询 + 回源 + 写缓存)
    t_no_cache_ms: 不用缓存时的延迟(直接回源)
    """
    miss_rate = 1 - hit_rate

    print("═" * 74)
    print("缓存对延迟分布的影响")
    print("═" * 74)
    print()
    print(f"命中率          : {hit_rate:.1%}")
    print(f"命中延迟        : {t_hit_ms:.2f} ms")
    print(f"未命中延迟      : {t_miss_ms:.2f} ms(含缓存开销,比不用缓存更慢)")
    print(f"不用缓存的延迟  : {t_no_cache_ms:.2f} ms")
    print()

    # 各分位数的估算
    def estimate(q):
        # 如果未命中比例 > (1-q),则该分位数落在未命中区间
        return t_miss_ms if miss_rate > (1 - q) else t_hit_ms

    print(f"{'分位数':<10}{'有缓存':>12}{'无缓存':>12}{'改善':>12}   说明")
    print("-" * 74)
    for q, label in [(0.50, "P50"), (0.90, "P90"), (0.95, "P95"), (0.99, "P99"), (0.999, "P999")]:
        with_cache = estimate(q)
        improvement = (t_no_cache_ms - with_cache) / t_no_cache_ms * 100
        note = "落在命中区间" if miss_rate <= (1 - q) else "⚠️ 落在未命中区间"
        print(f"{label:<10}{with_cache:>11.2f}ms{t_no_cache_ms:>11.2f}ms{improvement:>11.1f}%   {note}")

    print()
    print("═" * 74)
    print("关键结论:")
    print()

    # 判断 P99 是否改善
    p99_with = estimate(0.99)
    if p99_with >= t_miss_ms - 0.01:
        required = 0.99
        print(f"❌ P99 未改善({p99_with:.1f}ms,与不用缓存的 {t_no_cache_ms:.1f}ms 相比改善有限)")
        print(f"   原因:未命中率 {miss_rate:.1%} > 1%,P99 落在未命中的那批请求上")
        print(f"   → 要让 P99 改善,命中率必须 > {required:.0%}")
        print(f"   → 或者降低未命中延迟({t_miss_ms:.1f}ms)")
    else:
        print(f"✅ P99 改善到 {p99_with:.1f}ms(提升 {(t_no_cache_ms-p99_with)/t_no_cache_ms*100:.0f}%)")

    print()
    print("改进方向:")
    print("  ① 提高命中率:加大容量、延长 TTL、优化 key 设计、加 L1 本地缓存")
    print(f"     当前命中率 {hit_rate:.0%},每提高 1% 都有帮助(尤其接近 99% 时)")
    print("  ② 降低未命中延迟:")
    print("     - 优化回源查询(加索引、消除 N+1)")
    print("     - 三防护(单飞 / TTL 抖动 / 空值)")
    print("     - 异步写缓存(不阻塞返回)")
    print("  ③ 如果以上都不行 → 缓存对这个接口可能【不值得】")
    print("═" * 74)


if __name__ == "__main__":
    # 示例:命中率 85%,命中 0.5ms,未命中 200ms,无缓存 180ms
    analyze(hit_rate=0.85, t_hit_ms=0.5, t_miss_ms=200.0, t_no_cache_ms=180.0)
    print()
    print()
    # 对比:命中率提升到 99.5%
    analyze(hit_rate=0.995, t_hit_ms=0.5, t_miss_ms=200.0, t_no_cache_ms=180.0)

预期输出(85% 命中率):

分位数            有缓存        无缓存          改善   说明
--------------------------------------------------------------------------
P50              0.50ms     180.00ms       99.7%   落在命中区间
P90            200.00ms     180.00ms      -11.1%   ⚠️ 落在未命中区间
P95            200.00ms     180.00ms      -11.1%   ⚠️ 落在未命中区间
P99            200.00ms     180.00ms      -11.1%   ⚠️ 落在未命中区间
P999           200.00ms     180.00ms      -11.1%   ⚠️ 落在未命中区间

❌ P99 未改善(200.0ms,与不用缓存的 180.0ms 相比改善有限)

看到关键了吗:P50 改善 99.7%,但 P90 以上全部恶化 11%——这就是「缓存改善 P50 但可能让 P99 变差」的量化证据。

对比 99.5% 命中率:

P50              0.50ms     180.00ms       99.7%   落在命中区间
P90              0.50ms     180.00ms       99.7%   落在命中区间
P95              0.50ms     180.00ms       99.7%   落在命中区间
P99              0.50ms     180.00ms       99.7%   落在命中区间   ← 这时才改善 P99
P999           200.00ms     180.00ms      -11.1%   ⚠️ 落在未命中区间

三、判断「该不该用缓存」

# tools/should-cache.py
"""
判断一个接口是否值得加缓存。

需要回答四个问题——任何一个答"否"都可能否决缓存方案。
"""
import sys


def should_cache(
    read_write_ratio: float,      # 读写比(读/写)
    estimated_hit_rate: float,    # 预估命中率
    miss_latency_ms: float,       # 未命中延迟
    consistency_tolerance_s: int, # 可容忍的不一致时长(秒)
    current_p99_ms: float,        # 当前 P99
    slo_p99_ms: float,            # SLO 目标
):
    print("═" * 74)
    print("该不该加缓存?")
    print("═" * 74)
    print()

    verdict = []
    print(f"① 读写比           : {read_write_ratio:.1f} : 1")
    if read_write_ratio > 10:
        print("   ✅ 读多写少,缓存有意义")
    elif read_write_ratio > 3:
        print("   ⚠️  读写比一般,缓存收益可能有限")
    else:
        verdict.append("读写比过低(缓存刚写入就被更新,命中率低)")
        print("   ❌ 写多读少,缓存意义不大")

    print()
    print(f"② 预估命中率       : {estimated_hit_rate:.1%}")
    if estimated_hit_rate > 0.99:
        print("   ✅ 命中率足够高,能改善 P99")
    elif estimated_hit_rate > 0.9:
        print("   ⚠️  能改善 P50/P90,但 P99 可能改善有限")
    elif estimated_hit_rate > 0.7:
        print("   ⚠️  只改善 P50;P99 可能反而变差")
    else:
        verdict.append(f"命中率过低({estimated_hit_rate:.0%}),P99 不会改善")
        print("   ❌ 命中率太低,不值得")

    print()
    print(f"③ 可容忍不一致时长 : {consistency_tolerance_s} 秒")
    if consistency_tolerance_s >= 60:
        print("   ✅ 一致性要求宽松,缓存风险低")
    elif consistency_tolerance_s >= 5:
        print("   ⚠️  需要短 TTL + 主动失效(复杂度上升)")
    else:
        verdict.append("一致性要求强,缓存的复杂度不值得")
        print("   ❌ 一致性要求太强,缓存风险高")

    print()
    print(f"④ 当前 P99 / SLO   : {current_p99_ms:.0f} / {slo_p99_ms:.0f} ms")
    gap = (current_p99_ms - slo_p99_ms) / slo_p99_ms * 100
    if current_p99_ms > slo_p99_ms:
        print(f"   ✅ 超标 {gap:.0f}%,有优化动力")
    else:
        verdict.append("当前已达标,收益不足以覆盖复杂度")
        print("   ❌ 已达标,不值得引入复杂度")

    print()
    print("═" * 74)
    if verdict:
        print("❌ 建议:【暂不加缓存】")
        print()
        print("否决理由:")
        for v in verdict:
            print(f"  - {v}")
        print()
        print("替代方向(优先级更高):")
        print("  ① 先确认是不是 N+1 / 缺索引(第 7.2 节,收益更确定)")
        print("  ② 减少返回字段(少传数据,三重收益)")
        print("  ③ 如果确实需要缓存,考虑只缓存最热的那部分数据(而不是全部)")
    else:
        print("✅ 建议:【可以加缓存】")
        print()
        print("实施要求:")
        print("  ① 必须有界(maximumSize)")
        print("  ② 必须做三重防护(单飞 / TTL 抖动 / 空值缓存)")
        print("  ③ 必须监控命中率 + 【未命中时的延迟】")
        print("  ④ 必须有降级方案(缓存挂了直接查库)")
        print("  ⑤ 先更新数据库,再删缓存")
    print("═" * 74)


if __name__ == "__main__":
    print("示例 1:典型的『该加缓存』")
    should_cache(read_write_ratio=50, estimated_hit_rate=0.995,
                 miss_latency_ms=200, consistency_tolerance_s=300,
                 current_p99_ms=400, slo_p99_ms=200)
    print()
    print()
    print("示例 2:典型的『不该加缓存』")
    should_cache(read_write_ratio=2, estimated_hit_rate=0.4,
                 miss_latency_ms=200, consistency_tolerance_s=1,
                 current_p99_ms=95, slo_p99_ms=200)

四、动手改造

改动 观察什么
用 cache-p99-impact.py 试不同命中率(0.5 / 0.9 / 0.99 / 0.999) 找到「P99 开始改善」的命中率阈值
在 OrderCache 里去掉 TTL 抖动 用压测观察:大批 key 同时失效时 DB QPS 的脉冲
去掉单飞(改成 getIfPresent + put) 并发回源的数量会飙升(用 DB 并发查询数验证)
把 maximumSize 改成 Integer.MAX_VALUE(无界) 长时间运行后内存会持续增长(第 3 章 3.5 的浸泡分析)
给 Lab 6 的 /slow-db 跑 should-cache.py 看它是否会建议「暂不加缓存」(因为批量查询已经解决)

五、这段代码的局限

  • cache-p99-impact.py 用的是简化模型:它假设所有未命中请求的延迟相同,真实分布更复杂(重尾)。它适合做量级判断,不适合精确预测。
  • 命中率的预估很难准:should-cache.py 需要你提供 estimated_hit_rate,而这个数字通常要靠生产日志的真实分布来估(第 5.5 节)。
  • OrderCache 混用了同步与异步 API:cache.get() 是异步的(返回 CompletableFuture),而 stats() 用的是 synchronous().stats()——这是 Caffeine 的 AsyncCache 的正常用法。
  • TTL 抖动的实现用了 Math.random:在高并发下会有锁竞争,生产可以用 ThreadLocalRandom。
  • 没有实现「延迟双删」:对于强一致性要求的场景,先更新库再删缓存仍有极小窗口的不一致,需要额外的保护。