外观
parallel/false_sharing.cpp
配套源码,运行方法见同目录 README。返回实验总览。
cpp
// Same independent atomic increments, two memory layouts. No data race.
#include <algorithm>
#include <atomic>
#include <chrono>
#include <cstdint>
#include <iomanip>
#include <iostream>
#include <memory>
#include <stdexcept>
#include <string>
#include <thread>
#include <vector>
struct Packed { std::atomic<std::uint64_t> value{0}; };
struct alignas(128) Padded { std::atomic<std::uint64_t> value{0}; };
using Clock = std::chrono::steady_clock;
static std::size_t number(const char *text, std::size_t maximum) {
std::string s(text);
if (s.empty() || s.find_first_not_of("0123456789") != std::string::npos)
throw std::invalid_argument("positive decimal integers required");
auto value = std::stoull(s);
if (!value || value > maximum) throw std::invalid_argument("argument outside supported range");
return static_cast<std::size_t>(value);
}
template<class Slot>
static void run(const char *name, std::size_t workers, std::size_t iterations, std::size_t repeats) {
auto slots = std::make_unique<Slot[]>(workers);
std::vector<double> times;
for (std::size_t trial = 0; trial <= repeats; ++trial) {
for (std::size_t id = 0; id < workers; ++id) slots[id].value.store(0, std::memory_order_relaxed);
auto begin = Clock::now();
std::vector<std::thread> threads;
threads.reserve(workers);
try {
for (std::size_t id = 0; id < workers; ++id) threads.emplace_back([&, id] {
for (std::size_t i = 0; i < iterations; ++i)
slots[id].value.fetch_add(1, std::memory_order_relaxed);
});
} catch (...) {
for (auto &thread : threads) thread.join();
throw;
}
for (auto &thread : threads) thread.join();
auto end = Clock::now();
for (std::size_t id = 0; id < workers; ++id)
if (slots[id].value.load(std::memory_order_relaxed) != iterations)
throw std::runtime_error("counter mismatch");
if (trial) times.push_back(std::chrono::duration<double, std::milli>(end-begin).count());
}
std::sort(times.begin(), times.end());
double median = times[times.size()/2];
if (times.size()%2 == 0) median = (times[times.size()/2-1] + median)/2;
std::cout << name << ',' << sizeof(Slot) << ',' << alignof(Slot) << ',' << workers << ','
<< iterations << ',' << repeats << ',' << times.front() << ',' << median << ','
<< times.back() << ',' << iterations * workers << '\n';
}
int main(int argc, char **argv) {
try {
if (argc > 4) throw std::invalid_argument("usage: false_sharing [threads [iterations [repeats]]]");
std::size_t workers = argc > 1 ? number(argv[1], 128) : 4;
std::size_t iterations = argc > 2 ? number(argv[2], 100000000) : 1000000;
std::size_t repeats = argc > 3 ? number(argv[3], 31) : 5;
std::cout << "layout,stride,alignment,threads,iterations,repeats,min_ms,median_ms,max_ms,total\n"
<< std::fixed << std::setprecision(6);
run<Packed>("packed", workers, iterations, repeats);
run<Padded>("padded128", workers, iterations, repeats);
std::cerr << "all per-thread counters matched; timings include launch/join; no affinity set\n";
} catch (const std::exception &error) { std::cerr << error.what() << '\n'; return 1; }
}