64 Output_*
const output,
67 const auto left_NR = left.
nrow();
68 const auto common_dim = left.
ncol();
69 const auto right_NC = right.
ncol();
74 populate_sparse_buffers(
true, common_dim, right_NC, right, right_vbuffers, right_ibuffers, right_ranges, options.
num_threads);
77 auto right_non_empty = filter_non_empty_sparse(
79 [&](RightIndex_) ->
void {}
84 std::fill_n(output, sanisizer::product_unsafe<std::size_t>(left_NR, right_NC), 0);
92 std::optional<std::vector<Output_> > tmp_row;
97 for (LeftIndex_ lr = 0; lr < length; ++lr) {
98 const auto lptr = ext->fetch(dbuffer.data());
99 const auto optr = output + sanisizer::product_unsafe<std::size_t>(start + lr, right_NC);
100 const auto tmp_optr = (do_parallel ? tmp_row->data() : optr);
102 auto loop_body = [&](LeftIndex_ cd) ->
void {
103 const auto rrange = right_ranges[cd];
104 const Output_ mult = lptr[cd];
105 for (RightIndex_ x = 0; x < rrange.number; ++x) {
106 tmp_optr[rrange.index[x]] += mult *
static_cast<Output_
>(rrange.value[x]);
110 if (right_non_empty.has_value()) {
111 for (
const auto cd : *right_non_empty) {
115 for (LeftIndex_ cd = 0; cd < common_dim; ++cd) {
121 std::copy_n(tmp_optr, right_NC, optr);
128 std::fill_n(tmp_optr, right_NC, 0);
137 const LeftIndex_ max_block_rows = sanisizer::min(length, options.
block_size);
138 std::vector<std::vector<LeftValue_> > lbuffers;
139 lbuffers.reserve(max_block_rows);
140 for (LeftIndex_ b = 0; b < max_block_rows; ++b) {
145 std::optional<std::vector<Output_> > tmp_rows;
147 tmp_rows.emplace(sanisizer::product<I<
decltype(tmp_rows->size())> >(max_block_rows, right_NC));
151 while (lr < length) {
152 const LeftIndex_ lr_num = sanisizer::min(options.
block_size, length - lr);
153 for (LeftIndex_ lr_counter = 0; lr_counter < lr_num; ++lr_counter) {
154 lptrs[lr_counter] = ext->fetch(lbuffers[lr_counter].data());
156 const auto optr = output + sanisizer::product_unsafe<std::size_t>(start + lr, right_NC);
157 const auto tmp_optr = (do_parallel ? tmp_rows->data() : optr);
159 auto loop_body = [&](LeftIndex_ cd) ->
void {
160 const auto rrange = right_ranges[cd];
161 for (LeftIndex_ lr_counter = 0; lr_counter < lr_num; ++lr_counter) {
162 const Output_ mult = lptrs[lr_counter][cd];
163 for (RightIndex_ x = 0; x < rrange.number; ++x) {
164 tmp_optr[sanisizer::nd_offset<std::size_t>(rrange.index[x], right_NC, lr_counter)] += mult *
static_cast<Output_
>(rrange.value[x]);
169 if (right_non_empty.has_value()) {
170 for (
const auto cd : *right_non_empty) {
174 for (LeftIndex_ cd = 0; cd < common_dim; ++cd) {
180 const auto output_size = sanisizer::product_unsafe<std::size_t>(lr_num, right_NC);
181 std::copy_n(tmp_optr, output_size, optr);
182 std::fill_n(tmp_optr, output_size, 0);
void multiply_dense_row_with_sparse_row_matrix_to_row_output(const tatami::Matrix< LeftValue_, LeftIndex_ > &left, const tatami::Matrix< RightValue_, RightIndex_ > &right, Output_ *const output, const MultiplyDenseRowWithSparseRowMatrixToRowOutputOptions &options)
Definition row_to_row.hpp:61