v1 version and measurements
This commit is contained in:
@@ -55,7 +55,9 @@ struct session_t {
|
||||
std::string outMtxFile {"out.hdf5"}; //!< output matrix file name in HDF5 format
|
||||
std::string outMtxIdxDataSet {"/Idx"}; //!< Index output dataset name in HDF5 matrix file
|
||||
std::string outMtxDstDataSet {"/Dst"}; //!< Distance output dataset name in HDF5 matrix file
|
||||
std::size_t max_threads {}; //!< Maximum threads to use
|
||||
std::size_t max_threads {0}; //!< Maximum threads to use
|
||||
std::size_t slices {0}; //!< Slices/threads to use
|
||||
std::size_t accuracy {100}; //!< The neighbor finding accuracy
|
||||
bool timing {false}; //!< Enable timing prints of the program
|
||||
bool verbose {false}; //!< Flag to enable verbose output to stdout
|
||||
};
|
||||
|
||||
+16
-125
@@ -134,6 +134,9 @@ struct Matrix {
|
||||
Matrix& operator=(Matrix&& m) noexcept { moves(std::move(m)); return *this; }
|
||||
Matrix(const Matrix& m) = delete; //!< No copy ctor
|
||||
Matrix& operator=(const Matrix& m) = delete; //!< No copy
|
||||
//Matrix(const Matrix& m);
|
||||
//Matrix& operator=(const Matrix& m) { copy(m); }
|
||||
|
||||
//! @}
|
||||
|
||||
//! \name Data exposure
|
||||
@@ -233,6 +236,11 @@ struct Matrix {
|
||||
|
||||
// a basic serial iterator support
|
||||
DataType* data() noexcept { return data_; }
|
||||
DataType* begin() noexcept { return data_; }
|
||||
const DataType* begin() const noexcept { return data_; }
|
||||
DataType* end() noexcept { return data_ + capacity(rows_, cols_); }
|
||||
const DataType* end() const noexcept { return data_ + capacity(rows_, cols_); }
|
||||
|
||||
// IndexType begin_idx() noexcept { return 0; }
|
||||
// IndexType end_idx() noexcept { return capacity(rows_, cols_); }
|
||||
|
||||
@@ -265,17 +273,19 @@ struct Matrix {
|
||||
std::swap(rows_, src.rows_);
|
||||
std::swap(cols_, src.cols_);
|
||||
}
|
||||
|
||||
private:
|
||||
//! move helper
|
||||
void moves(Matrix&& src) noexcept {
|
||||
data_ = std::move(src.vector_storage_);
|
||||
data_ = std::move(src.raw_storage_);
|
||||
data_ = std::move(src.data_);
|
||||
data_ = std::move(src.use_vector_);
|
||||
rows_ = std::move(src.rows_);
|
||||
cols_ = std::move(src.cols_);
|
||||
vector_storage_ = std::move(src.vector_storage_);
|
||||
raw_storage_ = std::move(src.raw_storage_);
|
||||
data_ = std::move(src.data_);
|
||||
use_vector_ = std::move(src.use_vector_);
|
||||
rows_ = std::move(src.rows_);
|
||||
cols_ = std::move(src.cols_);
|
||||
}
|
||||
|
||||
// Storage
|
||||
std::vector<DataType>
|
||||
vector_storage_; //!< Internal storage (if used).
|
||||
DataType* raw_storage_; //!< External storage (if used).
|
||||
@@ -528,125 +538,6 @@ private:
|
||||
};
|
||||
|
||||
|
||||
template<typename ...> struct Matrix_view { };
|
||||
|
||||
/*!
|
||||
* @struct Matrix_view
|
||||
* @tparam MatrixType
|
||||
*/
|
||||
template<template <typename, typename, MatrixType, MatrixOrder, bool> class Matrix,
|
||||
typename DataType,
|
||||
typename IndexType,
|
||||
MatrixType Type,
|
||||
MatrixOrder Order>
|
||||
struct Matrix_view<Matrix<DataType, IndexType, Type, Order, false>> {
|
||||
using owner_t = Matrix<DataType, IndexType, Type, Order, false>;
|
||||
|
||||
using dataType = DataType; //!< meta:export of underling data type
|
||||
using indexType = IndexType; //!< meta:export of underling index type
|
||||
static constexpr MatrixOrder matrixOrder = Order; //!< meta:export of array order
|
||||
static constexpr MatrixType matrixType = Type; //!< meta:export of array type
|
||||
|
||||
/*!
|
||||
* \name Obj lifetime
|
||||
*/
|
||||
//! @{
|
||||
|
||||
//! Construct a matrix view to entire matrix
|
||||
Matrix_view(const owner_t* owner) noexcept :
|
||||
owner_(owner), m_(owner->data()), rows_(owner->rows()), cols_(owner->columns()) { }
|
||||
|
||||
Matrix_view(const owner_t* owner, IndexType begin, IndexType end) noexcept :
|
||||
owner_(owner) {
|
||||
if constexpr (Order == MatrixOrder::ROWMAJOR) {
|
||||
m_ = owner->data() + begin * owner->columns();
|
||||
rows_ = end - begin;
|
||||
cols_ = owner->columns();
|
||||
} else if (Order == MatrixOrder::COLMAJOR) {
|
||||
m_ = owner->data() + begin * owner->rows();
|
||||
rows_ = owner->rows();
|
||||
cols_ = end - begin;
|
||||
}
|
||||
}
|
||||
|
||||
Matrix_view(Matrix_view&& m) = delete; //! No move
|
||||
Matrix_view& operator=(Matrix_view&& m) = delete;
|
||||
Matrix_view(const Matrix_view& m) = delete; //!< No copy
|
||||
Matrix_view& operator=(const Matrix_view& m) = delete;
|
||||
//! @}
|
||||
|
||||
//! Get/Set the size of each dimension
|
||||
const IndexType rows() const noexcept { return rows_; }
|
||||
const IndexType columns() const noexcept { return cols_; }
|
||||
|
||||
//! Get the interface size of the Matrix (what appears to be the size)
|
||||
IndexType size() const {
|
||||
return rows_ * cols_;
|
||||
}
|
||||
|
||||
//! Actual memory capacity of the symmetric matrix
|
||||
static constexpr IndexType capacity(IndexType M, IndexType N) {
|
||||
return M*N;
|
||||
}
|
||||
/*
|
||||
* virtual 2D accessors
|
||||
*/
|
||||
const DataType get (IndexType i, IndexType j) const {
|
||||
if constexpr (Order == MatrixOrder::COLMAJOR)
|
||||
return m_[i + j*rows_];
|
||||
else
|
||||
return m_[i*cols_ + j];
|
||||
}
|
||||
|
||||
DataType set (DataType v, IndexType i, IndexType j) {
|
||||
if constexpr (Order == MatrixOrder::COLMAJOR)
|
||||
return m_[i + j*rows_] = v;
|
||||
else
|
||||
return m_[i*cols_ + j] = v;
|
||||
}
|
||||
// DataType operator()(IndexType i, IndexType j) { return get(i, j); }
|
||||
/*!
|
||||
* Return a proxy MatVal object with read and write capabilities.
|
||||
* @param i The row number
|
||||
* @param j The column number
|
||||
* @return tHE MatVal object
|
||||
*/
|
||||
MatVal<Matrix_view> operator()(IndexType i, IndexType j) noexcept {
|
||||
return MatVal<Matrix_view>(this, get(i, j), i, j);
|
||||
}
|
||||
|
||||
// a basic serial iterator support
|
||||
DataType* data() noexcept { return m_.data(); }
|
||||
// IndexType begin_idx() noexcept { return 0; }
|
||||
// IndexType end_idx() noexcept { return capacity(rows_, cols_); }
|
||||
|
||||
const DataType* data() const noexcept { return m_; }
|
||||
const IndexType begin_idx() const noexcept { return 0; }
|
||||
const IndexType end_idx() const noexcept { return capacity(rows_, cols_); }
|
||||
//! @}
|
||||
|
||||
/*!
|
||||
* \name Safe iteration API
|
||||
*
|
||||
* This api automates the iteration over the array based on
|
||||
* MatrixType
|
||||
*/
|
||||
//! @{
|
||||
template<typename F, typename... Args>
|
||||
void for_each_in (IndexType begin, IndexType end, F&& lambda, Args&&... args) {
|
||||
for (IndexType it=begin ; it<end ; ++it) {
|
||||
std::forward<F>(lambda)(std::forward<Args>(args)..., it);
|
||||
}
|
||||
}
|
||||
//! @}
|
||||
//!
|
||||
private:
|
||||
const owner_t* owner_ {nullptr}; //!< Pointer to Matrix
|
||||
DataType* m_ {nullptr}; //!< Starting address of the slice/view
|
||||
IndexType rows_{}; //!< the virtual size of rows.
|
||||
IndexType cols_{}; //!< the virtual size of columns.
|
||||
};
|
||||
|
||||
/*!
|
||||
* A view/iterator hybrid object for Matrix columns.
|
||||
*
|
||||
|
||||
@@ -54,10 +54,11 @@ void pdist2(const Matrix& X, const Matrix& Y, Matrix& D2) {
|
||||
for (int i = 0; i < M ; ++i) {
|
||||
for (int j = 0; j < N; ++j) {
|
||||
D2.set(D2.get(i, j) + X_norms[i] + Y_norms[j], i, j);
|
||||
//D2.set(std::max(D2.get(i, j), 0.0), i, j); // Ensure non-negative
|
||||
D2.set(std::max(D2.get(i, j), 0.0), i, j); // Ensure non-negative
|
||||
D2.set(std::sqrt(D2.get(i, j)), i, j); // Take the square root of each
|
||||
}
|
||||
}
|
||||
M++;
|
||||
}
|
||||
|
||||
template<typename DataType, typename IndexType>
|
||||
@@ -82,7 +83,7 @@ void quickselect(std::vector<std::pair<DataType, IndexType>>& vec, int k) {
|
||||
* point of Q
|
||||
*/
|
||||
template<typename MatrixD, typename MatrixI>
|
||||
void knnsearch(const MatrixD& C, const MatrixD& Q, size_t idx_offset, size_t k, size_t m, MatrixI& idx, MatrixD& dst) {
|
||||
void knnsearch(MatrixD& C, MatrixD& Q, size_t idx_offset, size_t k, size_t m, MatrixI& idx, MatrixD& dst) {
|
||||
|
||||
using DstType = typename MatrixD::dataType;
|
||||
using IdxType = typename MatrixI::dataType;
|
||||
|
||||
+115
-34
@@ -1,5 +1,5 @@
|
||||
/**
|
||||
* \file v0.hpp
|
||||
* \file v1.hpp
|
||||
* \brief
|
||||
*
|
||||
* \author
|
||||
@@ -16,6 +16,26 @@
|
||||
#include "v0.hpp"
|
||||
#include "config.h"
|
||||
|
||||
#if defined CILK
|
||||
#include <cilk/cilk.h>
|
||||
#include <cilk/cilk_api.h>
|
||||
//#include <cilk/reducer_opadd.h>
|
||||
|
||||
#elif defined OMP
|
||||
#include <omp.h>
|
||||
|
||||
#elif defined PTHREADS
|
||||
#include <thread>
|
||||
#include <numeric>
|
||||
#include <functional>
|
||||
//#include <random>
|
||||
|
||||
#else
|
||||
#endif
|
||||
|
||||
|
||||
void init_workers();
|
||||
|
||||
namespace v1 {
|
||||
|
||||
template <typename DataType, typename IndexType>
|
||||
@@ -57,49 +77,110 @@ void mergeResultsWithM(mtx::Matrix<IndexType>& N1, mtx::Matrix<DataType>& D1,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
template<typename MatrixD, typename MatrixI>
|
||||
void knnsearch(const MatrixD& C, const MatrixD& Q, size_t idx_offset, size_t k, size_t m, MatrixI& idx, MatrixD& dst) {
|
||||
|
||||
void worker_body (std::vector<MatrixD>& corpus_slices,
|
||||
std::vector<MatrixD>& query_slices,
|
||||
MatrixI& idx,
|
||||
MatrixD& dst,
|
||||
size_t slice,
|
||||
size_t num_slices, size_t corpus_slice_size, size_t query_slice_size,
|
||||
size_t k,
|
||||
size_t m) {
|
||||
// "load" types
|
||||
using DstType = typename MatrixD::dataType;
|
||||
using IdxType = typename MatrixI::dataType;
|
||||
|
||||
if (C.rows() <= 8 || Q.rows() <= 4) {
|
||||
// Base case: Call knnsearch directly
|
||||
v0::knnsearch(C, Q, idx_offset, k, m, idx, dst);
|
||||
return;
|
||||
}
|
||||
|
||||
// Divide Corpus and Query into subsets
|
||||
IdxType midC = C.rows() / 2;
|
||||
IdxType midQ = Q.rows() / 2;
|
||||
for (size_t ci = 0; ci < num_slices; ++ci) {
|
||||
size_t idx_offset = ci * corpus_slice_size;
|
||||
|
||||
// Slice corpus and query matrixes
|
||||
MatrixD C1((DstType*)C.data(), 0, midC, C.columns());
|
||||
MatrixD C2((DstType*)C.data(), midC, midC, C.columns());
|
||||
MatrixD Q1((DstType*)Q.data(), 0, midQ, Q.columns());
|
||||
MatrixD Q2((DstType*)Q.data(), midQ, midQ, Q.columns());
|
||||
// Intermediate matrixes for intermediate results
|
||||
MatrixI temp_idx(query_slices[slice].rows(), k);
|
||||
MatrixD temp_dst(query_slices[slice].rows(), k);
|
||||
|
||||
// Allocate temporary matrixes for all permutations
|
||||
MatrixI N1_1(midQ, k), N1_2(midQ, k), N2_1(midQ, k), N2_2(midQ, k);
|
||||
MatrixD D1_1(midQ, k), D1_2(midQ, k), D2_1(midQ, k), D2_2(midQ, k);
|
||||
// kNN for each combination
|
||||
v0::knnsearch(corpus_slices[ci], query_slices[slice], idx_offset, k, m, temp_idx, temp_dst);
|
||||
|
||||
// Recursive calls
|
||||
knnsearch(C1, Q1, idx_offset, k, m, N1_1, D1_1);
|
||||
knnsearch(C2, Q1, idx_offset + midC, k, m, N1_2, D1_2);
|
||||
knnsearch(C1, Q2, idx_offset, k, m, N2_1, D2_1);
|
||||
knnsearch(C2, Q2, idx_offset + midC, k, m, N2_2, D2_2);
|
||||
// Merge temporary results to final results
|
||||
MatrixI idx_slice((IdxType*)idx.data(), slice * query_slice_size, query_slices[slice].rows(), k);
|
||||
MatrixD dst_slice((DstType*)dst.data(), slice * query_slice_size, query_slices[slice].rows(), k);
|
||||
|
||||
// slice output matrixes
|
||||
MatrixI N1((IdxType*)idx.data(), 0, midQ, k);
|
||||
MatrixI N2((IdxType*)idx.data(), midQ, midQ, k);
|
||||
MatrixD D1((DstType*)dst.data(), 0, midQ, k);
|
||||
MatrixD D2((DstType*)dst.data(), midQ, midQ, k);
|
||||
mergeResultsWithM(idx_slice, dst_slice, temp_idx, temp_dst, k, m, idx_slice, dst_slice);
|
||||
}
|
||||
}
|
||||
|
||||
// Merge results in place
|
||||
mergeResultsWithM(N1_1, D1_1, N1_2, D1_2, k, m, N1, D1);
|
||||
mergeResultsWithM(N2_1, D2_1, N2_2, D2_2, k, m, N2, D2);
|
||||
template<typename MatrixD, typename MatrixI>
|
||||
void knnsearch(MatrixD& C, MatrixD& Q, size_t num_slices, size_t k, size_t m, MatrixI& idx, MatrixD& dst) {
|
||||
using DstType = typename MatrixD::dataType;
|
||||
using IdxType = typename MatrixI::dataType;
|
||||
|
||||
//Slice calculations
|
||||
size_t corpus_slice_size = C.rows() / ((num_slices == 0)? 1:num_slices);
|
||||
size_t query_slice_size = Q.rows() / ((num_slices == 0)? 1:num_slices);
|
||||
|
||||
// Make slices
|
||||
std::vector<MatrixD> corpus_slices;
|
||||
std::vector<MatrixD> query_slices;
|
||||
|
||||
for (size_t i = 0; i < num_slices; ++i) {
|
||||
corpus_slices.emplace_back(
|
||||
(DstType*)C.data(),
|
||||
i * corpus_slice_size,
|
||||
(i == num_slices - 1 ? C.rows() - i * corpus_slice_size : corpus_slice_size),
|
||||
C.columns());
|
||||
query_slices.emplace_back(
|
||||
(DstType*)Q.data(),
|
||||
i * query_slice_size,
|
||||
(i == num_slices - 1 ? Q.rows() - i * query_slice_size : query_slice_size),
|
||||
Q.columns());
|
||||
}
|
||||
|
||||
// Intermediate results
|
||||
for (size_t i = 0; i < dst.rows(); ++i) {
|
||||
for (size_t j = 0; j < dst.columns(); ++j) {
|
||||
dst.set(std::numeric_limits<DstType>::infinity(), i, j);
|
||||
idx.set(static_cast<IdxType>(-1), i, j);
|
||||
}
|
||||
}
|
||||
|
||||
// Main loop
|
||||
#if defined OMP
|
||||
#pragma omp parallel for
|
||||
for (size_t qi = 0; qi < num_slices; ++qi) {
|
||||
for (size_t qi = 0; qi < num_slices; ++qi) {
|
||||
worker_body (corpus_slices, query_slices, idx, dst, qi, num_slices, corpus_slice_size, query_slice_size, k, m);
|
||||
}
|
||||
}
|
||||
#elif defined CILK
|
||||
cilk_for (size_t qi = 0; qi < num_slices; ++qi) {
|
||||
for (size_t qi = 0; qi < num_slices; ++qi) {
|
||||
worker_body (corpus_slices, query_slices, idx, dst, qi, num_slices, corpus_slice_size, query_slice_size, k, m);
|
||||
}
|
||||
}
|
||||
#elif defined PTHREADS
|
||||
std::vector<std::thread> workers;
|
||||
for (size_t qi = 0; qi < num_slices; ++qi) {
|
||||
workers.push_back(
|
||||
std::thread (worker_body<MatrixD, MatrixI>,
|
||||
std::ref(corpus_slices), std::ref(query_slices),
|
||||
std::ref(idx), std::ref(dst),
|
||||
qi,
|
||||
num_slices, corpus_slice_size, query_slice_size,
|
||||
k, m)
|
||||
);
|
||||
}
|
||||
// Join threads
|
||||
std::for_each(workers.begin(), workers.end(), [](std::thread& t){
|
||||
t.join();
|
||||
});
|
||||
|
||||
#else
|
||||
for (size_t qi = 0; qi < num_slices; ++qi) {
|
||||
for (size_t qi = 0; qi < num_slices; ++qi) {
|
||||
worker_body (corpus_slices, query_slices, idx, dst, qi, num_slices, corpus_slice_size, query_slice_size, k, m);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user