69 const RightColumns_ right_columns,
70 GetRightRow_ get_right_row,
71 Output_*
const output,
74 const auto left_NR = left.
nrow();
75 const auto common_dim = left.
ncol();
79 std::fill_n(output, sanisizer::product_unsafe<std::size_t>(left_NR, right_columns), 0);
82 std::optional<std::vector<std::optional<std::vector<Output_> > > > tmp_results;
84 tmp_results.emplace(sanisizer::cast<I<
decltype(tmp_results->size())> >(options.
num_threads - 1));
87 const auto num_used =
tatami::parallelize([&](
int t, LeftIndex_ start, LeftIndex_ length) ->
void {
90 std::optional<std::vector<Output_> > tmp_output;
92 if (!do_parallel || t == 0) {
95 tmp_output.emplace(sanisizer::product<I<
decltype(tmp_output->size())> >(left_NR, right_columns));
96 outptr = tmp_output->data();
101 for (LeftIndex_ cd = 0; cd < length; ++cd) {
102 const auto left_ptr = left_ext->fetch(left_buffer.data());
103 const auto right_ptr = get_right_row(start + cd);
104 for (RightColumns_ rc = 0; rc < right_columns; ++rc) {
105 const Output_ mult = right_ptr[rc];
106 for (LeftIndex_ lr = 0; lr < left_NR; ++lr) {
107 outptr[sanisizer::nd_offset<std::size_t>(lr, left_NR, rc)] += mult *
static_cast<Output_
>(left_ptr[lr]);
113 std::vector<std::vector<LeftValue_> > left_buffers;
114 std::vector<const LeftValue_*> left_ptrs;
117 left_buffers.reserve(max_block_cols);
119 for (LeftIndex_ cd = 0; cd < max_block_cols; ++cd) {
125 while (cd < length) {
127 for (LeftIndex_ cd_counter = 0; cd_counter < cd_num; ++cd_counter) {
128 left_ptrs[cd_counter] = left_ext->fetch(left_buffers[cd_counter].data());
131 RightColumns_ rc = 0;
132 while (rc < right_columns) {
133 const RightColumns_ rc_end = rc + sanisizer::min(options.
primary_block_size, right_columns - rc);
135 while (lr < left_NR) {
138 for (LeftIndex_ cd_counter = 0; cd_counter < cd_num; ++cd_counter) {
139 const auto leftcol = left_ptrs[cd_counter];
140 const auto rightrow = get_right_row(start + cd + cd_counter);
141 for (
auto rc_copy = rc; rc_copy < rc_end; ++rc_copy) {
142 const Output_ mult = rightrow[rc_copy];
143 for (
auto lr_copy = lr; lr_copy < lr_end; ++lr_copy) {
144 outptr[sanisizer::nd_offset<std::size_t>(lr_copy, left_NR, rc_copy)] += mult *
static_cast<Output_
>(leftcol[lr_copy]);
157 if (do_parallel && t > 0) {
158 (*tmp_results)[t - 1] = std::move(tmp_output);
163 for (
int u = 1; u < num_used; ++u) {
164 const auto& tmp = *((*tmp_results)[u - 1]);
165 const auto N = tmp.size();
166 for (I<
decltype(N)> x = 0; x < N; ++x) {
196 Output_*
const output,
199 const auto left_NR = left.
nrow();
200 const auto common_dim = left.
ncol();
201 const auto right_NC = right.
ncol();
202 std::fill_n(output, sanisizer::product_unsafe<std::size_t>(left_NR, right_NC), 0);
205 std::optional<std::vector<std::optional<std::vector<Output_> > > > tmp_results;
207 tmp_results.emplace(sanisizer::cast<I<
decltype(tmp_results->size())> >(options.
num_threads - 1));
210 const auto num_used =
tatami::parallelize([&](
int t, LeftIndex_ start, LeftIndex_ length) ->
void {
214 std::optional<std::vector<Output_> > tmp_output;
216 if (!do_parallel || t == 0) {
219 tmp_output.emplace(sanisizer::product<I<
decltype(tmp_output->size())> >(left_NR, right_NC));
220 outptr = tmp_output->data();
227 for (LeftIndex_ cd = 0; cd < length; ++cd) {
228 const auto left_ptr = left_ext->fetch(left_buffer.data());
229 const auto right_ptr = right_ext->fetch(right_buffer.data());
230 for (RightIndex_ rc = 0; rc < right_NC; ++rc) {
231 const Output_ mult = right_ptr[rc];
232 for (LeftIndex_ lr = 0; lr < left_NR; ++lr) {
233 outptr[sanisizer::nd_offset<std::size_t>(lr, left_NR, rc)] += mult *
static_cast<Output_
>(left_ptr[lr]);
239 std::vector<std::vector<LeftValue_> > left_buffers;
240 std::vector<std::vector<RightValue_> > right_buffers;
241 std::vector<const LeftValue_*> left_ptrs;
242 std::vector<const RightValue_*> right_ptrs;
245 left_buffers.reserve(max_block_cols);
247 right_buffers.reserve(max_block_cols);
249 for (LeftIndex_ cd = 0; cd < max_block_cols; ++cd) {
256 while (cd < length) {
258 for (LeftIndex_ cd_counter = 0; cd_counter < cd_num; ++cd_counter) {
259 left_ptrs[cd_counter] = left_ext->fetch(left_buffers[cd_counter].data());
260 right_ptrs[cd_counter] = right_ext->fetch(right_buffers[cd_counter].data());
264 while (rc < right_NC) {
265 const RightIndex_ rc_end = rc + sanisizer::min(options.
primary_block_size, right_NC - rc);
267 while (lr < left_NR) {
270 for (LeftIndex_ cd_counter = 0; cd_counter < cd_num; ++cd_counter) {
271 const auto leftcol = left_ptrs[cd_counter];
272 const auto rightrow = right_ptrs[cd_counter];
273 for (
auto rc_copy = rc; rc_copy < rc_end; ++rc_copy) {
274 const Output_ mult = rightrow[rc_copy];
275 for (
auto lr_copy = lr; lr_copy < lr_end; ++lr_copy) {
276 outptr[sanisizer::nd_offset<std::size_t>(lr_copy, left_NR, rc_copy)] += mult *
static_cast<Output_
>(leftcol[lr_copy]);
289 if (do_parallel && t > 0) {
290 (*tmp_results)[t - 1] = std::move(tmp_output);
295 for (
int u = 1; u < num_used; ++u) {
296 const auto& tmp = *((*tmp_results)[u - 1]);
297 const auto N = tmp.size();
298 for (I<
decltype(N)> x = 0; x < N; ++x) {
void multiply_dense_column_with_dense_row_matrix_to_column_output(const tatami::Matrix< LeftValue_, LeftIndex_ > &left, const RightColumns_ right_columns, GetRightRow_ get_right_row, Output_ *const output, const MultiplyDenseColumnWithDenseRowMatrixToColumnOutputOptions &options)
Definition row_to_column.hpp:67