Aleph-w 3.0
A C++ Library for Data Structures and Algorithms
Loading...
Searching...
No Matches
ca_bench_example.cc
Go to the documentation of this file.
1/*
2 Aleph_w
3
4 Data structures & Algorithms
5 version 2.0.0b
6 https://github.com/lrleon/Aleph-w
7
8 This file is part of Aleph-w library
9
10 Copyright (c) 2002-2026 Leandro Rabindranath Leon
11
12 Permission is hereby granted, free of charge, to any person obtaining a copy
13 of this software and associated documentation files (the "Software"), to deal
14 in the Software without restriction, including without limitation the rights
15 to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
16 copies of the Software, and to permit persons to whom the Software is
17 furnished to do so, subject to the following conditions:
18
19 The above copyright notice and this permission notice shall be included in all
20 copies or substantial portions of the Software.
21
22 THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
23 IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
24 FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
25 AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
26 LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
27 OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
28 SOFTWARE.
29*/
30
46#include <array>
47#include <cstdint>
48#include <cstdio>
49#include <random>
50#include <vector>
51
52#include <thread_pool.H>
53
54#include <ca-bench.H>
55#include <ca-traits.H>
56#include <tpl_ca_engine.H>
57#include <tpl_ca_lattice.H>
58#include <tpl_ca_neighborhood.H>
60#include <tpl_ca_rule.H>
61#include <tpl_ca_storage.H>
62
63using namespace Aleph;
64using namespace Aleph::CA;
65
67
68namespace {
69
70Lat_t make_random_seed(std::size_t n, std::uint32_t seed)
71{
72 Lat_t lat({n, n}, 0);
73 std::mt19937 rng(seed);
74 std::bernoulli_distribution flip(0.4);
75 for (std::size_t i = 0; i < n; ++i)
76 for (std::size_t j = 0; j < n; ++j)
77 lat.set({static_cast<ca_index_t>(i), static_cast<ca_index_t>(j)},
78 flip(rng) ? 1 : 0);
79 return lat;
80}
81
82bool frames_equal(const Lat_t &a, const Lat_t &b)
83{
84 if (a.extents() != b.extents())
85 return false;
86 for (std::size_t i = 0; i < a.size(0); ++i)
87 for (std::size_t j = 0; j < a.size(1); ++j)
88 if (a.at({static_cast<ca_index_t>(i), static_cast<ca_index_t>(j)})
89 != b.at({static_cast<ca_index_t>(i), static_cast<ca_index_t>(j)}))
90 return false;
91 return true;
92}
93
94double run_sequential(const Lat_t &seed, std::size_t steps)
95{
98 return bench_seconds([&] { eng.run(steps); });
99}
100
101double run_parallel(const Lat_t &seed,
102 std::size_t steps,
103 std::size_t threads,
105{
106 ThreadPool pool(threads);
108 cfg.pool = &pool;
109 cfg.num_partitions = threads;
110 cfg.min_parallel_cells = 0;
113
114 const double secs = bench_seconds([&] { eng.run(steps); });
115 out_frame = eng.frame();
116 return secs;
117}
118
119} // namespace
120
121int main()
122{
123 std::printf("Aleph::CA Phase-5 microbench (Game of Life, toroidal)\n");
124 std::printf("------------------------------------------------------\n");
125 std::printf("hardware_concurrency = %u\n",
126 static_cast<unsigned>(std::thread::hardware_concurrency()));
127
128 // Modest defaults so the bench finishes quickly when run as part of
129 // the example suite. Bump these for serious profiling work.
130 const std::array<std::size_t, 3> sizes{256, 512, 1024};
131 const std::array<std::size_t, 5> thread_counts{1, 2, 4, 8, 16};
132 const std::size_t steps = 100;
133
134 std::printf(
135 "\n%-7s %-9s %-9s %-18s %-9s\n", "N", "threads", "wall (s)", "throughput", "speedup");
136 std::printf("%-7s %-9s %-9s %-18s %-9s\n", "-------", "-------", "--------",
137 "-----------------", "-------");
138
139 for (std::size_t N : sizes)
140 {
141 const Lat_t seed = make_random_seed(N, /*seed=*/0xCAFEu ^ N);
142
143 // Compute a reference sequential frame for validation and a
144 // baseline time for speedup calculations.
147 const double seq_seconds = bench_seconds([&] { ref.run(steps); });
148 const double seq_throughput
149 = static_cast<double>(N) * N * steps / seq_seconds;
150
151 std::printf("%-7zu %-9s %-9.4f %-18s %-9s\n",
152 N,
153 "seq",
156 "1.00x");
157
158 for (std::size_t T : thread_counts)
159 {
161 const double par_seconds = run_parallel(seed, steps, T, par_frame);
162
163 if (not frames_equal(ref.frame(), par_frame))
164 {
165 std::printf(" *** divergence at N=%zu T=%zu ***\n", N, T);
166 return 1;
167 }
168
169 const double throughput
170 = static_cast<double>(N) * N * steps / par_seconds;
171 const double speedup = seq_seconds / par_seconds;
172 std::printf("%-7zu %-9zu %-9.4f %-18s %4.2fx\n",
173 N,
174 T,
177 speedup);
178 }
179 std::printf("\n");
180 }
181
182 // Touch the lazy default pool to make sure later examples don't pay
183 // its construction cost; sanity check only.
185
186 // Throwaway sequential baseline to satisfy the seq_seconds reference
187 // path used above (kept as a hint that a single sequential reference
188 // is enough to validate every parallel configuration).
190
191 std::printf("Microbench finished. All parallel runs matched the sequential frame.\n");
192 return 0;
193}
Tiny chrono-based timer used by the CA module Examples.
size_t steps
Definition ca-c-api.h:126
Common typedefs and tag types for the Cellular Automata module.
int main()
Lattice that adds boundary-aware access on top of a storage.
const extents_type & extents() const noexcept
ca_size_t size() const noexcept
state_type at(const coord_type &c) const
Strict access: throws if c is out of range.
Moore (Chebyshev) neighborhood of radius R in N dimensions.
Parallel synchronous double-buffered engine.
const Lattice & frame() const noexcept
Return the current frame.
Synchronous double-buffered engine.
A reusable thread pool for efficient parallel task execution.
static mt19937 rng
#define N
Definition fib.C:294
size_t blossom_maximum_cardinality_matching(const GT &g, DynDlist< typename GT::Arc * > &matching, SA sa=SA())
Alias of compute_maximum_cardinality_general_matching().
Definition Blossom.H:466
constexpr Game_Of_Life_Rule make_game_of_life_rule() noexcept
Build the canonical Game of Life rule.
const char * format_throughput(double cells_per_second)
Format a "cells per second" rate as "X.XX M cells/s".
Definition ca-bench.H:126
bool frames_equal(const Lattice &a, const Lattice &b)
Definition ca-metrics.H:323
double bench_seconds(F &&f)
Run f() once and return the wall-clock time it took, in seconds.
Definition ca-bench.H:111
Main namespace for Aleph-w library functions.
Definition ah-arena.H:89
ThreadPool & default_pool()
Return the default shared thread pool instance.
std::decay_t< typename HeadC::Item_Type > T
Definition ah-zip.H:105
Configuration for Parallel_Synchronous_Engine.
ThreadPool * pool
Thread pool to schedule on. nullptr means Aleph::default_pool().
The lattice wraps around on every axis.
Definition ca-traits.H:124
ValueArg< size_t > seed
Definition testHash.C:53
A modern, efficient thread pool for parallel task execution.
Synchronous double-buffered engine for cellular automata.
Cellular automata lattice with pluggable boundary policies.
Neighborhoods catalogue for Aleph::CA.
Parallel synchronous engine for cellular automata (Phase 5).
Rule mechanisms for Aleph::CA.
Dense, contiguous storage for cellular automata cells (1D/2D/3D).