文章/答案/技术大牛

发布

社区首页 >问答首页 >RcppParallel并行距离计算:分段故障

问RcppParallel并行距离计算:分段故障
EN

Stack Overflow用户

提问于 2017-08-10 15:55:57

回答 1查看 418关注 0票数 1

我有一个矩阵，对于这个矩阵，我想要计算_i_th行和每一个其他行之间的距离(比如说欧几里得)(也就是说，我想要成对距离矩阵的_i_th行)。

#include <Rcpp.h>
#include <cmath>
#include <algorithm>
#include <RcppParallel.h>
//#include <RcppArmadillo.h>
#include <queue>

using namespace std;
using namespace Rcpp;
using namespace RcppParallel;

// [[Rcpp::export]]
double dist_fun(NumericVector row1, NumericVector row2){
   double rval = 0;
   for (int i = 0; i < row1.length(); i++){
      rval += (row1[i] - row2[i]) * (row1[i] - row2[i]);
   }
   return rval;
}

// [[Rcpp::export]]
NumericVector dist_row(NumericMatrix mat, int i){
   NumericVector row(mat.nrow());

   NumericMatrix::Row row1 = mat.row(i - 1);
   for (int j = 0; j < mat.nrow(); j++){
      NumericMatrix::Row row2 = mat.row(j);

      row(j) = dist_fun(row1, row2);
   }
   return row;
}

// [[Rcpp::depends(RcppParallel)]]
struct JsDistance: public Worker {

   // input matrix to read from
   const NumericMatrix mat;
   int i;

   // output vector to write to
   NumericVector output;

   // initialize from Rcpp input and output matrixes (the RMatrix class
   // can be automatically converted to from the Rcpp matrix type)
   JsDistance(const NumericMatrix mat, int i, NumericVector output)
      : mat(mat), i(i), output(output) {}


   // function call operator that work for the specified range (begin/end)
    void operator()(std::size_t begin, std::size_t end) {
      NumericVector row1 = mat.row(i);
      for (std::size_t j = begin; j < end; j++) {

         NumericVector row2 = mat.row(j);
         output[j] = dist_fun(row1, row2);
      }
    }
};

// [[Rcpp::export]]
NumericVector parallel_dist_row(NumericMatrix mat, int i) {

   // allocate the matrix we will return
   NumericVector output(mat.nrow());

   // create the worker
   JsDistance JsDistance(mat, i, output);

   // call it with parallelFor
   parallelFor(0, mat.nrow(), JsDistance);
   return output;
}

使用Rcpp的顺序方式是上面所写的'row_dist‘函数。然而，我想要处理的矩阵非常大，所以我想并行化它。但是，我会遇到一个分段错误，我不太明白为什么。要触发错误，可以运行以下代码：

library(Rcpp)
library(RcppParallel)

setThreadOptions(numThreads = 20)


set.seed(42)
X = matrix(rnorm(10000 * 400), 10000, 400)
sourceCpp("question.cpp")


start1 = proc.time()
print(dist_row(X, 2)[1:30])
print(proc.time() - start1)

start2 = proc.time()
print(parallel_dist_row(X, 2)[1:30])
print(proc.time() - start2)

有人能告诉我我做错了什么吗？提前谢谢你的时间！

=======================================================================

编辑：

inline double d(double a, double b){
   return fabs(a - b);
}

    // [[Rcpp::depends(RcppParallel)]
struct dtwDistance: public Worker {

  // Input matrix to read from must be of the RMatrix<T> form
  // if using Rcpp objects
  const RMatrix<double> mat;
  int i;

  // Output vector to write to must be of the RVector<T> form
  // if using Rcpp objects
  RVector<double> output;

  // initialize from Rcpp input and output matrixes (the RMatrix class
  // can be automatically converted to from the Rcpp matrix type)
  dtwDistance(const NumericMatrix mat, int i, NumericVector output)
    : mat(mat), i(i - 1), output(output) {}

  // Note the -1 ^^^^ to match results from prior function

  // Function call operator to iterate over a specified range (begin/end)
  void operator()(std::size_t begin, std::size_t end) {

    RMatrix<double>::Row row1 = mat.row(i);

    for (std::size_t j = begin; j < end; ++j) {

      RMatrix<double>::Row row2 = mat.row(j);

      size_t n = row1.length();
      size_t m = row2.length();
      NumericMatrix cost(n + 1, m + 1);

      for (int ii = 1; ii <= n; ii++){
        cost(i, 0) = numeric_limits<double>::infinity();
      }

      for (int jj = 1; jj <= m; jj++){
        cost(0, j) = numeric_limits<double>::infinity();
      }

      for (int ii = 1; ii <= n; ii++){
          for (int jj = 1; jj <= m; jj++){
            double dist = d(row1[ii - 1], row2[jj - 1]);
            cost(ii, jj) = dist + min(min(cost(ii - 1, jj), cost(ii, jj - 1)), cost(ii - 1, jj - 1));
            //cout << ii << ", " << jj << ", " << cost(ii, jj) << "\n";
          }
      }
      output[j] = cost(n, m);

    }
  }
};
// [[Rcpp::export]]
NumericVector parallel_dist_row_dtw(NumericMatrix mat, int i) {

   // allocate the matrix we will return
   //RMatrix<double> input(mat);
   NumericVector y(mat.nrow());
   //RVector<double> output(y);

   // create the worker
   dtwDistance dtwDistance(mat, i, y);

   // call it with parallelFor
   parallelFor(0, mat.nrow(), dtwDistance);
   return y;
}

我需要计算的距离是动态时间翘曲距离。我如上地实现了它。然而，当运行时，它会发出“堆栈不平衡”的警告。在几次跑完之后会有一个分段故障。我想知道现在有什么问题。

为了引发这个问题，我做了：

library(Rcpp)
library(RcppParallel)
setThreadOptions(numThreads = 4)
sourceCpp("scripts/chisq_dtw.cpp")
set.seed(42)
X = matrix(rnorm(1000), 100, 10)

parallel_dist_row_dtw(X, 1)
parallel_dist_row_dtw(X, 2)
parallel_dist_row_dtw(X, 3)
parallel_dist_row_dtw(X, 4)
parallel_dist_row_dtw(X, 5)

distance

rcpp

rcppparallel

c++

回答 1

Stack Overflow用户

回答已采纳

发布于 2017-08-10 17:58:37

问题是您没有通过RMatrix<T>和RVector<T>使用R对象的线程安全包装器。这些类很重要，因为在后台线程上执行并行化是不安全的，因此调用R或Rcpp是不安全的。正式文件在安全存取器部分强调了这一点。

特别是，我们：

为了提供对R向量和矩阵的数组的安全和方便的访问，RcppParallel引入了几个访问器类： RVector<T> -不同类型的R向量 RMatrix<T> -包扎各种类型的R矩阵(还包括Row和Column类) 要为Rcpp向量或矩阵创建线程安全访问器，只需使用它构造RVector或RMatrix实例即可。

代码修正

因此，您的工作可以通过将*Matrix切换到RMatrix<T>，将*Vector切换到RVector<T>来解决。

struct JsDistance: public Worker {

  // Input matrix to read from must be of the RMatrix<T> form
  // if using Rcpp objects
  const RMatrix<double> mat;
  int i;

  // Output vector to write to must be of the RVector<T> form
  // if using Rcpp objects
  RVector<double> output;

  // initialize from Rcpp input and output matrixes (the RMatrix class
  // can be automatically converted to from the Rcpp matrix type)
  JsDistance(const NumericMatrix mat, int i, NumericVector output)
    : mat(mat), i(i - 1), output(output) {}

  // Note the -1 ^^^^ to match results from prior function

  // Function call operator to iterate over a specified range (begin/end)
  void operator()(std::size_t begin, std::size_t end) {

    RMatrix<double>::Row row1 = mat.row(i);

    for (std::size_t j = begin; j < end; ++j) {

      RMatrix<double>::Row row2 = mat.row(j);

      double rval = 0;
      for (unsigned int k = 0; k < row1.length(); ++k) {
        rval += (row1[k] - row2[k]) * (row1[k] - row2[k]);
      }

      output[j] = rval;

    }
  }
};

特别是，这里使用的数据类型是表单RMatrix<double>，甚至用于访问矩阵。

此外，在并行版本中缺少一个i-1语句。为了弥补这一点，我选择在JSDistance的构造函数中处理它。

测试

set.seed(42)
X = matrix(rnorm(10000 * 400), 10000, 400)

start1 = proc.time()

print(dist_row(X, 2)[1:30])
# [1] 811.8873   0.0000 799.8153 810.1442 720.3232 730.6083 797.8441 781.8066 827.1511 834.1863 842.9392 850.2476 724.5842 673.1428 775.0994
# [16] 805.5752 804.9281 774.9770 799.7669 870.3187 815.1129 934.7581 726.1554 804.2097 758.4943 772.8931 806.6026 715.8257 847.8980 831.7555

print(proc.time() - start1)
# user  system elapsed 
# 0.22    0.00    0.23 

start2 = proc.time()

print(parallel_dist_row(X, 2)[1:30])
# [1] 811.8873   0.0000 799.8153 810.1442 720.3232 730.6083 797.8441 781.8066 827.1511 834.1863 842.9392 850.2476 724.5842 673.1428 775.0994
# [16] 805.5752 804.9281 774.9770 799.7669 870.3187 815.1129 934.7581 726.1554 804.2097 758.4943 772.8931 806.6026 715.8257 847.8980 831.7555

print(proc.time() - start2)
#   user  system elapsed 
#   0.28    0.00    0.06 

all.equal(parallel_dist_row(X, 2), dist_row(X, 2))
# [1] TRUE

票数 4

页面原文内容由Stack Overflow提供。腾讯云小微IT领域专用引擎提供翻译支持

原文链接：

https://stackoverflow.com/questions/45618375

复制

相似问题

问RcppParallel并行距离计算:分段故障
EN

回答 1

Stack Overflow用户

社区

活动

圈层

关于

腾讯云开发者

热门产品

热门推荐

更多推荐

问RcppParallel并行距离计算:分段故障EN

回答 1

Stack Overflow用户

社区

活动

圈层

关于

腾讯云开发者

热门产品

热门推荐

更多推荐

问RcppParallel并行距离计算:分段故障
EN