文档目录

2.6 配套代码:并发原语成本表与伪共享

对应小节:2.6 并发原语成本与伪共享 两个实验:① 测出你自己的并发成本表;② 复现伪共享并验证修法。

一、实验一:并发原语成本表

// src/main/kotlin/lesson02/PrimitiveCosts.kt
package lesson02

import java.util.concurrent.CountDownLatch
import java.util.concurrent.atomic.AtomicLong
import java.util.concurrent.atomic.LongAdder
import kotlin.concurrent.thread

private val lock = Any()
private var sharedCounter = 0L

fun bench(label: String, iterations: Int, body: () -> Unit): Double {
    repeat(iterations / 4) { body() }                 // 粗略预热
    val t0 = System.nanoTime()
    repeat(iterations) { body() }
    val t1 = System.nanoTime()
    val ns = (t1 - t0).toDouble() / iterations
    println("%-40s %10.1f ns/op".format(label, ns))
    return ns
}

/** 多线程版本的测量:N 个线程各自跑 iterations 次,返回每个线程的平均耗时 */
fun benchThreads(label: String, threads: Int, iterationsPerThread: Int, body: (Int) -> Unit): Double {
    repeat(2) {   // 预热两轮
        val start = CountDownLatch(1); val done = CountDownLatch(threads)
        repeat(threads) { t ->
            thread { start.await(); repeat(iterationsPerThread) { body(t) }; done.countDown() }
        }
        start.countDown(); done.await()
    }
    val start = CountDownLatch(1); val done = CountDownLatch(threads)
    val t0 = System.nanoTime()
    repeat(threads) { t ->
        thread { start.await(); repeat(iterationsPerThread) { body(t) }; done.countDown() }
    }
    start.countDown(); done.await()
    val elapsed = System.nanoTime() - t0
    val ns = elapsed.toDouble() / iterationsPerThread   // 每线程的"墙上时间"摊到每次操作
    println("%-40s %10.1f ns/op".format(label, ns))
    return ns
}

fun main() {
    val n = 2_000_000

    println("=== 单线程:无竞争 ===")
    val uncontended = bench("synchronized(无竞争)", n) {
        synchronized(lock) { sharedCounter++ }
    }

    val atomic = AtomicLong()
    bench("AtomicLong.incrementAndGet(无竞争)", n) {
        atomic.incrementAndGet()
    }

    val adder = LongAdder()
    bench("LongAdder.increment(无竞争)", n) {
        adder.increment()
    }

    println()
    println("=== 4 线程:有竞争 ===")
    val contendedLock = benchThreads("synchronized(4 线程竞争)", 4, n / 4) {
        synchronized(lock) { sharedCounter++ }
    }
    val contendedAtomic = AtomicLong()
    benchThreads("AtomicLong(4 线程竞争)", 4, n / 4) {
        contendedAtomic.incrementAndGet()
    }
    val contendedAdder = LongAdder()
    benchThreads("LongAdder(4 线程竞争)", 4, n / 4) {
        contendedAdder.increment()
    }

    println()
    println("=== 线程创建成本 ===")
    bench("Thread 创建 + join", 20_000) {
        val t = thread { }
        t.join()
    }

    println()
    println("=== 结论对照 ===")
    println("无竞争 vs 有竞争        : %.1f 倍".format(contendedLock / uncontended))
    println("LongAdder vs synchronized: %.1f 倍(竞争下)".format(contendedLock / contendedAdder))
    println()
    println("预期:无竞争的锁很便宜(几十 ns),有竞争时贵一个数量级;")
    println("      LongAdder 通过分片把竞争打散,在高竞争下明显更快。")
    println()
    println("⚠️ 并发基准对线程调度极其敏感,请跑 3 次看稳定性。")
}

二、预期输出形态

=== 单线程:无竞争 ===
synchronized(无竞争)                         18.4 ns/op
AtomicLong.incrementAndGet(无竞争)           11.2 ns/op
LongAdder.increment(无竞争)                   9.8 ns/op

=== 4 线程:有竞争 ===
synchronized(4 线程竞争)                     94.7 ns/op
AtomicLong(4 线程竞争)                      112.3 ns/op
LongAdder(4 线程竞争)                        23.6 ns/op

=== 线程创建成本 ===
Thread 创建 + join                         41250.0 ns/op     ← 约 41 微秒

=== 结论对照 ===
无竞争 vs 有竞争        : 5.1 倍
LongAdder vs synchronized: 4.0 倍(竞争下)

四个关键量级:

操作 量级
无竞争锁 / CAS 10–20 ns
有竞争锁 100 ns 级(取决于竞争强度,可能更差)
LongAdder 竞争下 20–30 ns(分片的效果)
线程创建 约 40 微秒(比普通操作贵 1000 倍以上)

线程创建那一行是本节最震撼的数字:一次线程创建 ≈ 40000 ns,而一次锁操作 ≈ 18 ns。差了三个数量级。 这就是「一万个请求开一万个线程」不可能成立的原因,也是协程存在的理由。

三、实验二:伪共享的复现与修复

// src/main/kotlin/lesson02/FalseSharing.kt
package lesson02

import java.util.concurrent.CountDownLatch
import kotlin.concurrent.thread

/** ❌ 两个字段紧挨着,通常落在同一个缓存行里 */
class PackedCounters {
    @Volatile var a = 0L
    @Volatile var b = 0L
}

/** ✅ 用填充把它们隔开(需要 -XX:-RestrictContended 才生效) */
class PaddedCounters {
    @Volatile var a = 0L
    @Suppress("unused") private val p1 = 0L
    @Suppress("unused") private val p2 = 0L
    @Suppress("unused") private val p3 = 0L
    @Suppress("unused") private val p4 = 0L
    @Suppress("unused") private val p5 = 0L
    @Suppress("unused") private val p6 = 0L
    @Volatile var b = 0L
}

/** ✅✅ 更好的做法:完全不共享(每个线程操作自己的槽位) */
class ShardedCounters(threads: Int) {
    val slots = LongArray(threads)
    fun add(idx: Int, v: Long) { slots[idx] += v }
}

fun measure(label: String, threads: Int, iterations: Int, writer: (Int, Int) -> Unit): Double {
    fun once(): Long {
        val start = CountDownLatch(1); val done = CountDownLatch(threads)
        val t0 = System.nanoTime()
        repeat(threads) { t ->
            thread { start.await(); repeat(iterations) { writer(t, it) }; done.countDown() }
        }
        start.countDown(); done.await()
        return System.nanoTime() - t0
    }
    repeat(2) { once() }                       // 预热
    val ns = once().toDouble() / iterations
    println("%-42s %10.1f ns/op".format(label, ns))
    return ns
}

fun main() {
    val threads = 4
    val iterations = 10_000_000

    println("=== 伪共享实验(每个线程高频写自己的字段)===")
    println("场景 A:两个线程写同一个对象的相邻字段(伪共享)")

    val packed = PackedCounters()
    val a = measure("Packed(a 和 b 相邻)", 2, iterations) { t, i ->
        if (t == 0) packed.a += i else packed.b += i
    }

    println("\n场景 B:两个线程各写一个独立对象(无共享)")
    val sep1 = PackedCounters(); val sep2 = PackedCounters()
    val b = measure("Separate objects(各写自己的对象)", 2, iterations) { t, i ->
        if (t == 0) sep1.a += i else sep2.a += i
    }

    println("\n场景 C:4 线程分别写自己的数组槽位(分片,最佳)")
    val sharded = ShardedCounters(threads)
    val c = measure("Sharded array(分片)", threads, iterations) { t, i ->
        sharded.add(t, i.toLong())
    }

    println()
    println("=== 结论 ===")
    println("伪共享 vs 独立对象 : %.2f 倍".format(a / b))
    println("伪共享 vs 分片     : %.2f 倍".format(a / c))
    println()
    println("预期:伪共享明显更慢(通常 1.5 ~ 5 倍,取决于 CPU 与访问频率)。")
    println("      『分片』比『填充』更可靠 —— 因为 JVM 的对象布局不由你完全控制。")
}

运行注意:

# 如果要验证 @Contended 的效果(而不是手工填充),需要解除限制
java -XX:-RestrictContended -cp build/classes/kotlin/main lesson02.FalseSharingKt

为什么用手工填充而不是 @Contended:@Contended 是 JDK 内部注解(jdk.internal.vm.annotation.Contended),业务代码使用它需要特殊处理,且会增加内存占用。手工填充也并不可靠——JVM 可能重排字段。所以本实验的重点结论是:「分片」比「填充」更值得依赖。

四、三个工程结论

结论 依据
锁的成本主要在竞争,不在加锁 无竞争 18 ns,有竞争 95 ns(5 倍以上)
线程很贵,不要按请求创建 创建一次约 40 微秒
伪共享真实存在,优先用「不共享」解决 分片比分填充更可靠,且效果更好

遇到竞争时的优先顺序(重申第 2.6 节正文):

① 消除共享(分片 / 线程封闭 / 不可变)
        ↓
② 缩小临界区(只锁必须锁的部分)
        ↓
③ 用现成的分片结构(LongAdder、ConcurrentHashMap 的分段)
        ↓
④ 最后才考虑自己写无锁

五、动手改造

改动 观察什么
把竞争线程数从 4 改成 2、8、16 找到「竞争开始显著」和「加线程反而更慢」的两个点
把 synchronized 的临界区内容减半 竞争成本不成比例地下降(临界区越短,冲突概率越低)
把 LongAdder 换成 AtomicLong 再跑 8 线程 LongAdder 的优势在高竞争下更明显
在 Packed 的字段之间加 7 个 Long 填充 性能是否恢复到接近「独立对象」的水平?
把线程创建实验改成用线程池 摊薄后的单次成本会低几个数量级——这就是池化的价值

六、这段代码的局限

  • 并发基准极不稳定:线程调度、CPU 亲和性、后台任务都会影响结果。请跑 3 次以上,看趋势而不是单次值。
  • benchThreads 的计时方式粗糙:它用「墙上时间 ÷ 每线程迭代数」,混合了排队与调度时间。严格测量应该用 JMH 的 @Threads + @BenchmarkMode(Mode.Throughput)。
  • 伪共享的倍率高度依赖 CPU 架构(缓存行大小、一致性协议实现),不同机器差异很大。
  • 不要把这些数字当成绝对性能指标——它们的作用是让你建立量级直觉。