Skip to content

parallel/false_sharing.cpp ​

配套源码,运行方法见同目录 README。返回实验总览。

cpp
// Same independent atomic increments, two memory layouts. No data race.
#include <algorithm>
#include <atomic>
#include <chrono>
#include <cstdint>
#include <iomanip>
#include <iostream>
#include <memory>
#include <stdexcept>
#include <string>
#include <thread>
#include <vector>

struct Packed { std::atomic<std::uint64_t> value{0}; };
struct alignas(128) Padded { std::atomic<std::uint64_t> value{0}; };
using Clock = std::chrono::steady_clock;

static std::size_t number(const char *text, std::size_t maximum) {
    std::string s(text);
    if (s.empty() || s.find_first_not_of("0123456789") != std::string::npos)
        throw std::invalid_argument("positive decimal integers required");
    auto value = std::stoull(s);
    if (!value || value > maximum) throw std::invalid_argument("argument outside supported range");
    return static_cast<std::size_t>(value);
}

template<class Slot>
static void run(const char *name, std::size_t workers, std::size_t iterations, std::size_t repeats) {
    auto slots = std::make_unique<Slot[]>(workers);
    std::vector<double> times;
    for (std::size_t trial = 0; trial <= repeats; ++trial) {
        for (std::size_t id = 0; id < workers; ++id) slots[id].value.store(0, std::memory_order_relaxed);
        auto begin = Clock::now();
        std::vector<std::thread> threads;
        threads.reserve(workers);
        try {
            for (std::size_t id = 0; id < workers; ++id) threads.emplace_back([&, id] {
                for (std::size_t i = 0; i < iterations; ++i)
                    slots[id].value.fetch_add(1, std::memory_order_relaxed);
            });
        } catch (...) {
            for (auto &thread : threads) thread.join();
            throw;
        }
        for (auto &thread : threads) thread.join();
        auto end = Clock::now();
        for (std::size_t id = 0; id < workers; ++id)
            if (slots[id].value.load(std::memory_order_relaxed) != iterations)
                throw std::runtime_error("counter mismatch");
        if (trial) times.push_back(std::chrono::duration<double, std::milli>(end-begin).count());
    }
    std::sort(times.begin(), times.end());
    double median = times[times.size()/2];
    if (times.size()%2 == 0) median = (times[times.size()/2-1] + median)/2;
    std::cout << name << ',' << sizeof(Slot) << ',' << alignof(Slot) << ',' << workers << ','
              << iterations << ',' << repeats << ',' << times.front() << ',' << median << ','
              << times.back() << ',' << iterations * workers << '\n';
}

int main(int argc, char **argv) {
    try {
        if (argc > 4) throw std::invalid_argument("usage: false_sharing [threads [iterations [repeats]]]");
        std::size_t workers = argc > 1 ? number(argv[1], 128) : 4;
        std::size_t iterations = argc > 2 ? number(argv[2], 100000000) : 1000000;
        std::size_t repeats = argc > 3 ? number(argv[3], 31) : 5;
        std::cout << "layout,stride,alignment,threads,iterations,repeats,min_ms,median_ms,max_ms,total\n"
                  << std::fixed << std::setprecision(6);
        run<Packed>("packed", workers, iterations, repeats);
        run<Padded>("padded128", workers, iterations, repeats);
        std::cerr << "all per-thread counters matched; timings include launch/join; no affinity set\n";
    } catch (const std::exception &error) { std::cerr << error.what() << '\n'; return 1; }
}