7 Parallelism in C++

Threads, Mutexes, Condition Variables, Futures, Promises, Async, OpenMP und MPI
Author

Daniel Schwarzenbach

Threads

Parallelising a for Loop

Let’s say we want to compute a function for each index in a range, e.g. sin(i * π/2) for i = 0, 1, 2, 3, 4. We can do this in parallel using threads.

#include <iostream>
#include <thread>
#include <vector>
#include <functional>
#include <cmath>

// using declarations
using std::cout;
using std::endl;
using std::vector;
using std::thread;
using std::function;

// overload << operator for vector
template<typename T>
std::ostream& operator<<(std::ostream& os, const vector<T>& vec) {
    os << "[";
    for (size_t i = 0; i < vec.size(); ++i) {
        os << vec[i];
        if (i != vec.size() - 1) os << ", ";
    }
    os << "]";
    return os;
}

/**
 *  @brief  parallel_for applies a function to each index in a range in parallel
 *  @details This function launches a thread for each index in the range [min, supr) 
 and applies the given function to that index. The results are collected in a vector 
 and returned.
 * @param  {int} min                    : the starting index (inclusive)
 * @param  {int} supr                   : the ending index (exclusive)
 * @param  {function<Out(int)> &} const : the function to apply to each index
 * @return {vector<Out>}                : a vector of results
 */
 template<typename Out>
auto parallel_for(int min, int supr, function<Out(int)> const& function) -> vector<Out> {
  // make sure min < supr
  if (min >= supr) throw std::invalid_argument("min >= supr");
  vector<thread> threads;
  vector<Out> results(supr - min);
  // launch threads
  for (int i = min; i < supr; ++i) {
    threads.emplace_back([&results, min, i, &function]() {
      results[i - min] = function(i);
    });
  }
  // join threads
  for (auto& t : threads) {
    t.join();
  }
  return results;
}


int main() {
  // π/2
  constexpr double π_2 = 3.14159265358979323846 / 2.0;
  // compute sin(i * π/2) for i = 0, 1, 2, 3, 4 in parallel
  vector<int> results = parallel_for<int>(0, 5, [](int i) -> int {
    return (int)round(sin(i * π_2));
  });
  // print results
  cout << results << endl;
}