From d2940a69f9126175fcc65eb5b91ea6c05a6151ef Mon Sep 17 00:00:00 2001 From: Shubham Agrawal Date: Sat, 20 Aug 2022 02:10:16 +0800 Subject: [PATCH] fixed size bug in convolution And fixed bug in padding layer --- src/mlpack/methods/ann/layer/convolution.hpp | 9 ++ .../methods/ann/layer/convolution_impl.hpp | 87 +++++++++++++------ src/mlpack/methods/ann/layer/padding_impl.hpp | 54 ++++++------ .../tests/ann/convolutional_network_test.cpp | 22 ++--- .../tests/ann/feedforward_network_test.cpp | 15 ++-- src/mlpack/tests/ann/layer/padding.cpp | 2 +- 6 files changed, 115 insertions(+), 74 deletions(-) diff --git a/src/mlpack/methods/ann/layer/convolution.hpp b/src/mlpack/methods/ann/layer/convolution.hpp index 08a9c7d641..ecf6c9e592 100644 --- a/src/mlpack/methods/ann/layer/convolution.hpp +++ b/src/mlpack/methods/ann/layer/convolution.hpp @@ -377,6 +377,9 @@ class ConvolutionType : public Layer //! Locally-stored padding layer. ann::Padding padding; + //! Locally-stored padding layer for backward pass. + ann::Padding paddingBackward; + //! Type of padding. std::string paddingType; @@ -384,6 +387,12 @@ class ConvolutionType : public Layer size_t inMaps; //! Locally-cached higher-order input dimensions. size_t higherInDimensions; + + //! Locally-stored apparent width. + size_t apparentWidth; + + //! Locally-stored apparent height. + size_t apparentHeight; }; // class Convolution // Standard Convolution layer. diff --git a/src/mlpack/methods/ann/layer/convolution_impl.hpp b/src/mlpack/methods/ann/layer/convolution_impl.hpp index 67a5476b79..3427fa31c6 100644 --- a/src/mlpack/methods/ann/layer/convolution_impl.hpp +++ b/src/mlpack/methods/ann/layer/convolution_impl.hpp @@ -130,9 +130,12 @@ ConvolutionType< padHTop(other.padHTop), useBias(other.useBias), padding(other.padding), + paddingBackward(other.paddingBackward), paddingType(other.paddingType), inMaps(other.inMaps), - higherInDimensions(other.higherInDimensions) + higherInDimensions(other.higherInDimensions), + apparentWidth(other.apparentWidth), + apparentHeight(other.apparentHeight) { // Nothing to do. } @@ -161,9 +164,12 @@ ConvolutionType< padHTop(std::move(other.padHTop)), useBias(std::move(other.useBias)), padding(std::move(other.padding)), + paddingBackward(std::move(other.paddingBackward)), paddingType(std::move(other.paddingType)), inMaps(std::move(other.inMaps)), - higherInDimensions(std::move(other.higherInDimensions)) + higherInDimensions(std::move(other.higherInDimensions)), + apparentWidth(std::move(other.apparentWidth)), + apparentHeight(std::move(other.apparentHeight)) { // Nothing to do. } @@ -201,9 +207,12 @@ ConvolutionType< padHTop = other.padHTop; useBias = other.useBias; padding = other.padding; + paddingBackward = other.paddingBackward; paddingType = other.paddingType; inMaps = other.inMaps; higherInDimensions = other.higherInDimensions; + apparentWidth = other.apparentWidth; + apparentHeight = other.apparentHeight; } return *this; @@ -242,9 +251,12 @@ ConvolutionType< padHTop = std::move(other.padHTop); useBias = std::move(other.useBias); padding = std::move(other.padding); + paddingBackward = std::move(other.paddingBackward); paddingType = std::move(other.paddingType); inMaps = std::move(other.inMaps); higherInDimensions = std::move(other.higherInDimensions); + apparentWidth = std::move(other.apparentWidth); + apparentHeight = std::move(other.apparentHeight); } return *this; @@ -380,7 +392,8 @@ void ConvolutionType< arma::Cube dilatedMappedError; if (strideHeight == 1 && strideWidth == 1) { - dilatedMappedError = mappedError; + MakeAlias(dilatedMappedError, mappedError.memptr(), + mappedError.n_rows, mappedError.n_cols, mappedError.n_slices); } else { @@ -407,6 +420,12 @@ void ConvolutionType< Rotate180(weight.slice(map), rotatedFilters.slice(map)); } + MatType output(apparentWidth * apparentHeight * inMaps * higherInDimensions, + batchSize, arma::fill::zeros); + arma::Cube outputCube; + MakeAlias(outputCube, output.memptr(), apparentWidth, apparentHeight, + inMaps * higherInDimensions * batchSize); + // See Forward() for the overall iteration strategy. for (size_t offset = 0; offset < (higherInDimensions * batchSize); ++offset) { @@ -418,36 +437,38 @@ void ConvolutionType< for (size_t inMap = 0; inMap < (size_t) inMaps; ++inMap) { // Iterate over output maps. - MatType output; for (size_t outMap = 0; outMap < maps; ++outMap) { BackwardConvolutionRule::Convolution( dilatedMappedError.slice(outMap + fullOutputOffset), rotatedFilters.slice((outMap * inMaps) + inMap), - output, + outputCube.slice(inMap + fullInputOffset), 1, 1, 1, 1, - outMap > 0); - } - // If the stride width or height is greater than 1, then we have to - // insert columns and rows into the convolution output. - MatType& curGTemp = gTemp.slice(inMap + fullInputOffset); - if (usingPadding) - { - curGTemp = output.submat( - padWLeft, - padHTop, - padWLeft + gTemp.n_rows - 1, - padHTop + gTemp.n_cols - 1); - } - else - { - curGTemp = output; + true); } } } + MatType temp(padding.OutputDimensions()[0] * padding.OutputDimensions()[1] * inMaps * higherInDimensions, + batchSize); + arma::Cube tempCube; + MakeAlias(tempCube, temp.memptr(), padding.OutputDimensions()[0], + padding.OutputDimensions()[1], inMaps * higherInDimensions * batchSize); + paddingBackward.Forward(output, temp); + if (usingPadding) + { + gTemp = tempCube.tube( + padWLeft, + padHTop, + padWLeft + gTemp.n_rows - 1, + padHTop + gTemp.n_cols - 1); + } + else + { + gTemp = tempCube; + } } template< @@ -482,6 +503,13 @@ void ConvolutionType< const_cast(usingPadding ? inputPadded : input).memptr(), paddedRows, paddedCols, inMaps * batchSize, false, false); + MatType temp(apparentWidth * apparentHeight * inMaps * higherInDimensions, + batchSize); + arma::Cube tempCube; + MakeAlias(tempCube, temp.memptr(), apparentWidth, apparentHeight, + inMaps * higherInDimensions * batchSize); + paddingBackward.Backward(input, usingPadding ? inputPadded : input, temp); + // We will make an alias for the gradient, but note that this is only for the // convolution map weights! The bias will be handled by direct accesses into // `gradient`. @@ -501,17 +529,15 @@ void ConvolutionType< MatType& curError = mappedError.slice(outMap + fullOutputOffset); for (size_t inMap = 0; inMap < inMaps; ++inMap) { - MatType output; GradientConvolutionRule::Convolution( - inputTemp.slice(inMap + fullInputOffset), + tempCube.slice(inMap + fullInputOffset), curError, - output, + gradientTemp.slice((outMap * inMaps) + inMap), 1, 1, strideWidth, - strideHeight); - - gradientTemp.slice((outMap * inMaps) + inMap) += output; + strideHeight, + true); } if (useBias) @@ -570,6 +596,13 @@ void ConvolutionType< this->outputDimensions[i] = this->inputDimensions[i]; } + apparentWidth = (this->outputDimensions[0] - 1) * strideWidth + kernelWidth; + apparentHeight = (this->outputDimensions[1] - 1) * strideHeight + kernelHeight; + + paddingBackward = ann::Padding(0, padding.OutputDimensions()[0] - apparentWidth, 0, padding.OutputDimensions()[1] - apparentHeight); + paddingBackward.InputDimensions() = std::vector({ apparentWidth, apparentHeight, inMaps * higherInDimensions }); + paddingBackward.ComputeOutputDimensions(); + this->outputDimensions[2] = maps; } diff --git a/src/mlpack/methods/ann/layer/padding_impl.hpp b/src/mlpack/methods/ann/layer/padding_impl.hpp index 69aa6dde60..ef48f43992 100644 --- a/src/mlpack/methods/ann/layer/padding_impl.hpp +++ b/src/mlpack/methods/ann/layer/padding_impl.hpp @@ -107,43 +107,43 @@ void PaddingType::Forward(const MatType& input, MatType& output) output.n_cols, false, true); // Set the padding parts to 0. - if (padWLeft > 0) + if (padHTop > 0) { reshapedOutput.tube(0, 0, reshapedOutput.n_rows - 1, - padWLeft - 1).zeros(); + padHTop - 1).zeros(); } - if (padHTop > 0) + if (padWLeft > 0) { reshapedOutput.tube(0, - padWLeft, - padHTop - 1, - padWLeft + this->inputDimensions[1] - 1).zeros(); - } - - if (padWRight > 0) - { - reshapedOutput.tube(0, - padWLeft + this->inputDimensions[1], - reshapedOutput.n_rows - 1, - reshapedOutput.n_cols - 1).zeros(); + padHTop, + padWLeft - 1, + padHTop + this->inputDimensions[1] - 1).zeros(); } if (padHBottom > 0) { - reshapedOutput.tube(padHTop + this->inputDimensions[0], - padWLeft, + reshapedOutput.tube(0, + padHTop + this->inputDimensions[1], reshapedOutput.n_rows - 1, - padWLeft + this->inputDimensions[1] - 1).zeros(); + reshapedOutput.n_cols - 1).zeros(); + } + + if (padWRight > 0) + { + reshapedOutput.tube(padWLeft + this->inputDimensions[0], + padHTop, + reshapedOutput.n_rows - 1, + padHTop + this->inputDimensions[1] - 1).zeros(); } // Copy the input matrix. - reshapedOutput.tube(padHTop, - padWLeft, - padHTop + this->inputDimensions[0] - 1, - padWLeft + this->inputDimensions[1] - 1) = reshapedInput; + reshapedOutput.tube(padWLeft, + padHTop, + padWLeft + this->inputDimensions[0] - 1, + padHTop + this->inputDimensions[1] - 1) = reshapedInput; } template @@ -161,10 +161,10 @@ void PaddingType::Backward( this->inputDimensions[0], this->inputDimensions[1], totalInMaps * g.n_cols, false, true); - reshapedG = reshapedGy.tube(padHTop, - padWLeft, - padHTop + this->inputDimensions[0] - 1, - padWLeft + this->inputDimensions[1] - 1); + reshapedG = reshapedGy.tube(padWLeft, + padHTop, + padWLeft + this->inputDimensions[0] - 1, + padHTop + this->inputDimensions[1] - 1); } template @@ -172,8 +172,8 @@ void PaddingType::ComputeOutputDimensions() { this->outputDimensions = this->inputDimensions; - this->outputDimensions[0] += padHTop + padHBottom; - this->outputDimensions[1] += padWLeft + padWRight; + this->outputDimensions[0] += padWLeft + padWRight; + this->outputDimensions[1] += padHTop + padHBottom; // Higher dimensions remain unchanged. But, we will cache the product of // these higher dimensions. diff --git a/src/mlpack/tests/ann/convolutional_network_test.cpp b/src/mlpack/tests/ann/convolutional_network_test.cpp index 927ca18154..baf45d907f 100644 --- a/src/mlpack/tests/ann/convolutional_network_test.cpp +++ b/src/mlpack/tests/ann/convolutional_network_test.cpp @@ -85,29 +85,29 @@ TEST_CASE("PaddingTest", "[ConvolutionalNetworktest]") model.Forward(X, results); // Ensure that things are correctly padded. - arma::cube reshapedResults(results.memptr(), 35, 31, results.n_cols, false, + arma::cube reshapedResults(results.memptr(), 31, 35, results.n_cols, false, true); for (size_t i = 0; i < reshapedResults.n_slices; ++i) { // Check left. - for (size_t j = 0; j < reshapedResults.n_rows; ++j) - REQUIRE(reshapedResults(j, 0, i) == 0.0); + for (size_t j = 0; j < reshapedResults.n_cols; ++j) + REQUIRE(reshapedResults(0, j, i) == 0.0); // Check top. for (size_t j = 0; j < 3; ++j) - for (size_t k = 0; k < reshapedResults.n_cols; ++k) - REQUIRE(reshapedResults(j, k, i) == 0.0); + for (size_t k = 0; k < reshapedResults.n_rows; ++k) + REQUIRE(reshapedResults(k, j, i) == 0.0); // Check bottom. - for (size_t j = 31; j < reshapedResults.n_rows; ++j) - for (size_t k = 0; k < reshapedResults.n_cols; ++k) - REQUIRE(reshapedResults(j, k, i) == 0.0); + for (size_t j = 31; j < reshapedResults.n_cols; ++j) + for (size_t k = 0; k < reshapedResults.n_rows; ++k) + REQUIRE(reshapedResults(k, j, i) == 0.0); // Check right. - for (size_t j = 0; j < reshapedResults.n_rows; ++j) - for (size_t k = 29; k < reshapedResults.n_cols; ++k) - REQUIRE(reshapedResults(j, k, i) == 0.0); + for (size_t j = 0; j < reshapedResults.n_cols; ++j) + for (size_t k = 29; k < reshapedResults.n_rows; ++k) + REQUIRE(reshapedResults(k, j, i) == 0.0); } } diff --git a/src/mlpack/tests/ann/feedforward_network_test.cpp b/src/mlpack/tests/ann/feedforward_network_test.cpp index 5894fc26b8..a74c52be0d 100644 --- a/src/mlpack/tests/ann/feedforward_network_test.cpp +++ b/src/mlpack/tests/ann/feedforward_network_test.cpp @@ -80,8 +80,7 @@ void CheckCopyFunction(ModelType* network1, template void CheckMoveFunction(ModelType* network1, MatType& trainData, - MatType& trainLabels, - const size_t maxEpochs) + MatType& trainLabels) { ens::RMSProp opt(0.01, 32, 0.88, 1e-8, trainData.n_cols, -1); network1->Train(trainData, trainLabels, opt); @@ -150,7 +149,7 @@ TEST_CASE("CheckCopyMovingVanillaNetworkTest", "[FeedForwardNetworkTest]") CheckCopyFunction(model, trainData, trainLabels); // Check whether move constructor is working or not. - CheckMoveFunction(model1, trainData, trainLabels, 1); + CheckMoveFunction(model1, trainData, trainLabels); } /** @@ -184,7 +183,7 @@ TEST_CASE("CheckCopyMovingLinear3DNetworkTest", "[FeedForwardNetworkTest]") CheckCopyFunction(model, trainData, trainLabels); // Check whether move constructor is working or not. - CheckMoveFunction(model1, trainData, trainLabels, 1); + CheckMoveFunction(model1, trainData, trainLabels); } /** @@ -215,7 +214,7 @@ TEST_CASE("CheckCopyMovingNoisyLinearTest", "[FeedForwardNetworkTest]") model2->Add(); // Check whether move constructor is working or not. - CheckMoveFunction(model2, input, output, 1); + CheckMoveFunction(model2, input, output); } /** @@ -261,7 +260,7 @@ TEST_CASE("CheckCopyMovingConcatenateTest", "[FeedForwardNetworkTest]") model2->Add(); // Check whether move constructor is working or not. - CheckMoveFunction(model2, input, output, 1); + CheckMoveFunction(model2, input, output); } /** @@ -295,7 +294,7 @@ TEST_CASE("CheckCopyMovingDropoutNetworkTest", "[FeedForwardNetworkTest]") CheckCopyFunction(model, trainData, trainLabels); // Check whether move constructor is working or not. - CheckMoveFunction(model1, trainData, trainLabels, 1); + CheckMoveFunction(model1, trainData, trainLabels); } /** @@ -342,7 +341,7 @@ TEST_CASE("CheckCopyMovingVanillaNetworkTestNoBias", "[FeedForwardNetworkTest]") CheckCopyFunction<>(model, trainData, trainLabels); // Check whether move constructor is working or not. - CheckMoveFunction<>(model1, trainData, trainLabels, 1); + CheckMoveFunction<>(model1, trainData, trainLabels); } /** diff --git a/src/mlpack/tests/ann/layer/padding.cpp b/src/mlpack/tests/ann/layer/padding.cpp index 2a25baed50..5ff697b478 100644 --- a/src/mlpack/tests/ann/layer/padding.cpp +++ b/src/mlpack/tests/ann/layer/padding.cpp @@ -43,7 +43,7 @@ TEST_CASE("SimplePaddingLayerTest", "[ANNLayerTest]") output.randu(); module.Forward(input, output); REQUIRE(arma::accu(input) == Approx(arma::accu(output))); - REQUIRE(output.n_rows == (9 * 8)); // 2x5 --> 9x8 + REQUIRE(output.n_rows == (5 * 12)); // 2x5 --> 5x12 // Test the Backward function. delta.set_size(input.n_rows, input.n_cols);