MNN/source/backend/cpu/CPUMatrixBandPart.cpp

//
//  CPUMatrixBandPart.cpp
//  MNN
//
//  Created by MNN on 2019/09/17.
//  Copyright © 2018, Alibaba Group Holding Limited
//

#include "CPUMatrixBandPart.hpp"
#include "ConvOpt.h"
#include "TensorUtils.hpp"
#include "Macro.h"
namespace MNN {
ErrorCode CPUMatrixBandPart::onResize(const std::vector<Tensor *> &inputs, const std::vector<Tensor *> &outputs) {
    MNN_ASSERT(3 == inputs.size());
    auto dimensions = inputs[0]->dimensions();
    auto height     = inputs[0]->length(dimensions - 2);
    auto width      = inputs[0]->length(dimensions - 1);
    mMask.reset(Tensor::createDevice<float>({2, height*width}, Tensor::CAFFE_C4));
    auto res                                               = backend()->onAcquireBuffer(mMask.get(), Backend::DYNAMIC);
    if (!res) {
        return OUT_OF_MEMORY;
    }
    backend()->onReleaseBuffer(mMask.get(), Backend::DYNAMIC);
    return NO_ERROR;
}
ErrorCode CPUMatrixBandPart::onExecute(const std::vector<Tensor *> &inputs, const std::vector<Tensor *> &outputs) {
    // Generate Mask
    auto lower   = inputs[1]->host<int32_t>()[0];
    auto upper   = inputs[2]->host<int32_t>()[0];
    auto maskPtr = mMask->host<float>() + mMask->stride(0);
    auto dimensions = inputs[0]->dimensions();
    auto height     = inputs[0]->length(dimensions - 2);
    auto width      = inputs[0]->length(dimensions - 1);

    for (int y = 0; y < height; ++y) {
        auto maskY = maskPtr + y * width;
        for (int x = 0; x < width; ++x) {
            bool valid = (lower < 0 || (y - x) <= lower) && (upper < 0 || (x - y) <= upper);
            maskY[x]   = valid ? 1.0f : 0.0f;
        }
    }

    // Run Mul
    auto outputPtr = outputs[0]->host<float>();
    auto inputPtr  = inputs[0]->host<float>();
    int outside    = 1;
    for (int i = 0; i < inputs[0]->dimensions() - 2; ++i) {
        outside *= inputs[0]->length(i);
    }
    auto inside = height * width;
    // For SSE the SIMD will crash when the memory is not aligned
    if (inside % 4 == 0) {
        for (int i = 0; i < outside; ++i) {
            MNNMatrixProdCommon(outputPtr + i * inside, inputPtr + i * inside, maskPtr, inside, 0, 0, 0, 1);
        }
    } else {
        for (int i = 0; i < outside; ++i) {
            ::memcpy(mMask->host<float>(), inputPtr + i * inside, inside*sizeof(float));
            MNNMatrixProdCommon(mMask->host<float>(), mMask->host<float>(), maskPtr, inside, 0, 0, 0, 1);
            ::memcpy(outputPtr + i * inside, mMask->host<float>(), inside*sizeof(float));
        }
    }
    return NO_ERROR;
}

class CPUMatrixBandPartCreator : public CPUBackend::Creator {
public:
    virtual Execution *onCreate(const std::vector<Tensor *> &inputs, const std::vector<Tensor *> &outputs,
                                const MNN::Op *op, Backend *backend) const override {
        return new CPUMatrixBandPart(backend);
    }
};

REGISTER_CPU_OP_CREATOR(CPUMatrixBandPartCreator, OpType_MatrixBandPart);
} // namespace MNN
- build: - unify schema building in core and converter; - add more build script for android; - add linux build script for python; - ops impl: - add floor mod support in binary; - use eltwise impl in add/max/sub/mul binary for optimization; - remove fake double support in cast; - fix 5d support for concat; - add adjX and adjY support for batch matmul; - optimize conv2d back prop filter; - add pad mode support for conv3d; - fix bug in conv2d & conv depthwise with very small feature map; - optimize binary without broacast; - add data types support for gather; - add gather ND support; - use uint8 data type in gather v2; - add transpose support for matmul; - add matrix band part; - add dim != 4 support for padding, reshape & tensor convert; - add pad type support for pool3d; - make ops based on TensorFlow Lite quantization optional; - add all & any support for reduction; - use type in parameter as output type in reduction; - add int support for unary; - add variable weight support for conv2d; - fix conv2d depthwise weights initialization; - fix type support for transpose; - fix grad outputs count for reduce grad and reshape grad; - fix priorbox & detection output; - fix metal softmax error; - python: - add runSessionWithCallBackInfo interface; - add max nodes limit (1400) for visualization tool; - fix save error in python3; - align default dim; - convert: - add extra design for optimization; - add more post converting optimizers; - add caffe v1 weights blob support; - add cast, unary, conv transpose support for onnx model; - optimize batchnorm, conv with variable weights, prelu, reshape, slice, upsample for onnx model; - add cos/sin/atan/tan support for unary for tensorflow model; - add any/all support for reduction for tensorflow model; - add elu, conv3d, pool3d support for tensorflow model; - optimize argmax, batchnorm, concat, batch to space, conv with variable weights, prelu, slice for tensorflow model; - others: - fix size computer lock; - fix thread pool deadlock; - add express & parameters in express; - rewrite blitter chooser without static map; - add tests for expr; 2019-10-29 13:37:26 +08:00			`//`
			`// CPUMatrixBandPart.cpp`
			`// MNN`
			`//`
			`// Created by MNN on 2019/09/17.`
			`// Copyright © 2018, Alibaba Group Holding Limited`
			`//`

			`#include "CPUMatrixBandPart.hpp"`
			`#include "ConvOpt.h"`
			`#include "TensorUtils.hpp"`
			`#include "Macro.h"`
			`namespace MNN {`
			`ErrorCode CPUMatrixBandPart::onResize(const std::vector<Tensor > &inputs, const std::vector<Tensor > &outputs) {`
			`MNN_ASSERT(3 == inputs.size());`
			`auto dimensions = inputs[0]->dimensions();`
			`auto height = inputs[0]->length(dimensions - 2);`
			`auto width = inputs[0]->length(dimensions - 1);`
			`mMask.reset(Tensor::createDevice<float>({2, height*width}, Tensor::CAFFE_C4));`
			`auto res = backend()->onAcquireBuffer(mMask.get(), Backend::DYNAMIC);`
			`if (!res) {`
			`return OUT_OF_MEMORY;`
			`}`
			`backend()->onReleaseBuffer(mMask.get(), Backend::DYNAMIC);`
			`return NO_ERROR;`
			`}`
			`ErrorCode CPUMatrixBandPart::onExecute(const std::vector<Tensor > &inputs, const std::vector<Tensor > &outputs) {`
			`// Generate Mask`
			`auto lower = inputs[1]->host<int32_t>()[0];`
			`auto upper = inputs[2]->host<int32_t>()[0];`
			`auto maskPtr = mMask->host<float>() + mMask->stride(0);`
			`auto dimensions = inputs[0]->dimensions();`
			`auto height = inputs[0]->length(dimensions - 2);`
			`auto width = inputs[0]->length(dimensions - 1);`

			`for (int y = 0; y < height; ++y) {`
			`auto maskY = maskPtr + y * width;`
			`for (int x = 0; x < width; ++x) {`
			`bool valid = (lower < 0 \|\| (y - x) <= lower) && (upper < 0 \|\| (x - y) <= upper);`
			`maskY[x] = valid ? 1.0f : 0.0f;`
			`}`
			`}`

			`// Run Mul`
			`auto outputPtr = outputs[0]->host<float>();`
			`auto inputPtr = inputs[0]->host<float>();`
			`int outside = 1;`
			`for (int i = 0; i < inputs[0]->dimensions() - 2; ++i) {`
			`outside *= inputs[0]->length(i);`
			`}`
			`auto inside = height * width;`
			`// For SSE the SIMD will crash when the memory is not aligned`
			`if (inside % 4 == 0) {`
			`for (int i = 0; i < outside; ++i) {`
			`MNNMatrixProdCommon(outputPtr + i * inside, inputPtr + i * inside, maskPtr, inside, 0, 0, 0, 1);`
			`}`
			`} else {`
			`for (int i = 0; i < outside; ++i) {`
			`::memcpy(mMask->host<float>(), inputPtr + i * inside, inside*sizeof(float));`
			`MNNMatrixProdCommon(mMask->host<float>(), mMask->host<float>(), maskPtr, inside, 0, 0, 0, 1);`
			`::memcpy(outputPtr + i * inside, mMask->host<float>(), inside*sizeof(float));`
			`}`
			`}`
			`return NO_ERROR;`
			`}`

			`class CPUMatrixBandPartCreator : public CPUBackend::Creator {`
			`public:`
			`virtual Execution onCreate(const std::vector<Tensor > &inputs, const std::vector<Tensor *> &outputs,`
			`const MNN::Op op, Backend backend) const override {`
			`return new CPUMatrixBandPart(backend);`
			`}`
			`};`

			`REGISTER_CPU_OP_CREATOR(CPUMatrixBandPartCreator, OpType_MatrixBandPart);`
			`} // namespace MNN`