diff --git a/README.md b/README.md index 6fbca03c..5cc0c2e9 100644 --- a/README.md +++ b/README.md @@ -19,6 +19,7 @@ * [Object Detection and Tracking](mediapipe/docs/object_tracking_mobile_gpu.md) * [Objectron: 3D Object Detection and Tracking](mediapipe/docs/objectron_mobile_gpu.md) * [AutoFlip: Intelligent Video Reframing](mediapipe/docs/autoflip.md) +* [KNIFT: Template Matching with Neural Image Features](mediapipe/docs/template_matching_mobile_cpu.md) ![face_detection](mediapipe/docs/images/mobile/face_detection_android_gpu_small.gif) ![face_mesh](mediapipe/docs/images/mobile/face_mesh_android_gpu_small.gif) @@ -29,6 +30,7 @@ ![object_tracking](mediapipe/docs/images/mobile/object_tracking_android_gpu_small.gif) ![objectron_shoes](mediapipe/docs/images/mobile/objectron_shoe_android_gpu_small.gif) ![objectron_chair](mediapipe/docs/images/mobile/objectron_chair_android_gpu_small.gif) +![template_matching](mediapipe/docs/images/mobile/template_matching_android_cpu_small.gif) ## Installation Follow these [instructions](mediapipe/docs/install.md). @@ -53,6 +55,7 @@ Search MediaPipe Github repository using [Google Open Source code search](https: * [YouTube Channel](https://www.youtube.com/channel/UCObqmpuSMx-usADtL_qdMAw) ## Publications +* [MediaPipe KNIFT: Template-based Feature Matching](https://mediapipe.page.link/knift-blog) * [Alfred Camera: Smart camera features using MediaPipe](https://developers.googleblog.com/2020/03/alfred-camera-smart-camera-features-using-mediapipe.html) * [MediaPipe Objectron: Real-time 3D Object Detection on Mobile Devices](https://mediapipe.page.link/objectron-aiblog) * [AutoFlip: An Open Source Framework for Intelligent Video Reframing](https://mediapipe.page.link/autoflip) diff --git a/mediapipe/calculators/image/image_properties_calculator.cc b/mediapipe/calculators/image/image_properties_calculator.cc index ea6c06c4..be0a65e0 100644 --- a/mediapipe/calculators/image/image_properties_calculator.cc +++ b/mediapipe/calculators/image/image_properties_calculator.cc @@ -19,6 +19,11 @@ #include "mediapipe/gpu/gpu_buffer.h" #endif // !MEDIAPIPE_DISABLE_GPU +namespace { +constexpr char kImageFrameTag[] = "IMAGE"; +constexpr char kGpuBufferTag[] = "IMAGE_GPU"; +} // namespace + namespace mediapipe { // Extracts image properties from the input image and outputs the properties. @@ -40,13 +45,14 @@ namespace mediapipe { class ImagePropertiesCalculator : public CalculatorBase { public: static ::mediapipe::Status GetContract(CalculatorContract* cc) { - RET_CHECK(cc->Inputs().HasTag("IMAGE") ^ cc->Inputs().HasTag("IMAGE_GPU")); - if (cc->Inputs().HasTag("IMAGE")) { - cc->Inputs().Tag("IMAGE").Set(); + RET_CHECK(cc->Inputs().HasTag(kImageFrameTag) ^ + cc->Inputs().HasTag(kGpuBufferTag)); + if (cc->Inputs().HasTag(kImageFrameTag)) { + cc->Inputs().Tag(kImageFrameTag).Set(); } #if !defined(MEDIAPIPE_DISABLE_GPU) - if (cc->Inputs().HasTag("IMAGE_GPU")) { - cc->Inputs().Tag("IMAGE_GPU").Set<::mediapipe::GpuBuffer>(); + if (cc->Inputs().HasTag(kGpuBufferTag)) { + cc->Inputs().Tag(kGpuBufferTag).Set<::mediapipe::GpuBuffer>(); } #endif // !MEDIAPIPE_DISABLE_GPU @@ -66,16 +72,17 @@ class ImagePropertiesCalculator : public CalculatorBase { int width; int height; - if (cc->Inputs().HasTag("IMAGE") && !cc->Inputs().Tag("IMAGE").IsEmpty()) { - const auto& image = cc->Inputs().Tag("IMAGE").Get(); + if (cc->Inputs().HasTag(kImageFrameTag) && + !cc->Inputs().Tag(kImageFrameTag).IsEmpty()) { + const auto& image = cc->Inputs().Tag(kImageFrameTag).Get(); width = image.Width(); height = image.Height(); } #if !defined(MEDIAPIPE_DISABLE_GPU) - if (cc->Inputs().HasTag("IMAGE_GPU") && - !cc->Inputs().Tag("IMAGE_GPU").IsEmpty()) { + if (cc->Inputs().HasTag(kGpuBufferTag) && + !cc->Inputs().Tag(kGpuBufferTag).IsEmpty()) { const auto& image = - cc->Inputs().Tag("IMAGE_GPU").Get(); + cc->Inputs().Tag(kGpuBufferTag).Get(); width = image.width(); height = image.height(); } diff --git a/mediapipe/calculators/image/image_transformation_calculator.cc b/mediapipe/calculators/image/image_transformation_calculator.cc index 34d61861..85960913 100644 --- a/mediapipe/calculators/image/image_transformation_calculator.cc +++ b/mediapipe/calculators/image/image_transformation_calculator.cc @@ -47,6 +47,9 @@ namespace mediapipe { #endif // !MEDIAPIPE_DISABLE_GPU namespace { +constexpr char kImageFrameTag[] = "IMAGE"; +constexpr char kGpuBufferTag[] = "IMAGE_GPU"; + int RotationModeToDegrees(mediapipe::RotationMode_Mode rotation) { switch (rotation) { case mediapipe::RotationMode_Mode_UNKNOWN: @@ -95,7 +98,7 @@ mediapipe::ScaleMode_Mode ParseScaleMode( // Scales, rotates, and flips images horizontally or vertically. // // Input: -// One of the following two tags: +// One of the following tags: // IMAGE: ImageFrame representing the input image. // IMAGE_GPU: GpuBuffer representing the input image. // @@ -113,7 +116,7 @@ mediapipe::ScaleMode_Mode ParseScaleMode( // corresponding field in the calculator options. // // Output: -// One of the following two tags: +// One of the following tags: // IMAGE - ImageFrame representing the output image. // IMAGE_GPU - GpuBuffer representing the output image. // @@ -152,7 +155,8 @@ mediapipe::ScaleMode_Mode ParseScaleMode( // Note: To enable horizontal or vertical flipping, specify them in the // calculator options. Flipping is applied after rotation. // -// Note: Only scale mode STRETCH is currently supported on CPU. +// Note: Input defines output, so only matchig types supported: +// IMAGE -> IMAGE or IMAGE_GPU -> IMAGE_GPU // class ImageTransformationCalculator : public CalculatorBase { public: @@ -186,7 +190,7 @@ class ImageTransformationCalculator : public CalculatorBase { bool use_gpu_ = false; #if !defined(MEDIAPIPE_DISABLE_GPU) - GlCalculatorHelper helper_; + GlCalculatorHelper gpu_helper_; std::unique_ptr rgb_renderer_; std::unique_ptr yuv_renderer_; std::unique_ptr ext_rgb_renderer_; @@ -197,21 +201,22 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); // static ::mediapipe::Status ImageTransformationCalculator::GetContract( CalculatorContract* cc) { - RET_CHECK(cc->Inputs().HasTag("IMAGE") ^ cc->Inputs().HasTag("IMAGE_GPU")); - RET_CHECK(cc->Outputs().HasTag("IMAGE") ^ cc->Outputs().HasTag("IMAGE_GPU")); + // Only one input can be set, and the output type must match. + RET_CHECK(cc->Inputs().HasTag(kImageFrameTag) ^ + cc->Inputs().HasTag(kGpuBufferTag)); bool use_gpu = false; - if (cc->Inputs().HasTag("IMAGE")) { - RET_CHECK(cc->Outputs().HasTag("IMAGE")); - cc->Inputs().Tag("IMAGE").Set(); - cc->Outputs().Tag("IMAGE").Set(); + if (cc->Inputs().HasTag(kImageFrameTag)) { + RET_CHECK(cc->Outputs().HasTag(kImageFrameTag)); + cc->Inputs().Tag(kImageFrameTag).Set(); + cc->Outputs().Tag(kImageFrameTag).Set(); } #if !defined(MEDIAPIPE_DISABLE_GPU) - if (cc->Inputs().HasTag("IMAGE_GPU")) { - RET_CHECK(cc->Outputs().HasTag("IMAGE_GPU")); - cc->Inputs().Tag("IMAGE_GPU").Set(); - cc->Outputs().Tag("IMAGE_GPU").Set(); + if (cc->Inputs().HasTag(kGpuBufferTag)) { + RET_CHECK(cc->Outputs().HasTag(kGpuBufferTag)); + cc->Inputs().Tag(kGpuBufferTag).Set(); + cc->Outputs().Tag(kGpuBufferTag).Set(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU @@ -259,7 +264,7 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); options_ = cc->Options(); - if (cc->Inputs().HasTag("IMAGE_GPU")) { + if (cc->Inputs().HasTag(kGpuBufferTag)) { use_gpu_ = true; } @@ -300,7 +305,7 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); if (use_gpu_) { #if !defined(MEDIAPIPE_DISABLE_GPU) // Let the helper access the GL context information. - MP_RETURN_IF_ERROR(helper_.Open(cc)); + MP_RETURN_IF_ERROR(gpu_helper_.Open(cc)); #else RET_CHECK_FAIL() << "GPU processing not enabled."; #endif // !MEDIAPIPE_DISABLE_GPU @@ -328,18 +333,14 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); if (use_gpu_) { #if !defined(MEDIAPIPE_DISABLE_GPU) - if (cc->Inputs().Tag("IMAGE_GPU").IsEmpty()) { - // Image is missing, hence no way to produce output image. (Timestamp - // bound will be updated automatically.) + if (cc->Inputs().Tag(kGpuBufferTag).IsEmpty()) { return ::mediapipe::OkStatus(); } - return helper_.RunInGlContext( + return gpu_helper_.RunInGlContext( [this, cc]() -> ::mediapipe::Status { return RenderGpu(cc); }); #endif // !MEDIAPIPE_DISABLE_GPU } else { - if (cc->Inputs().Tag("IMAGE").IsEmpty()) { - // Image is missing, hence no way to produce output image. (Timestamp - // bound will be updated automatically.) + if (cc->Inputs().Tag(kImageFrameTag).IsEmpty()) { return ::mediapipe::OkStatus(); } return RenderCpu(cc); @@ -354,7 +355,7 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); QuadRenderer* rgb_renderer = rgb_renderer_.release(); QuadRenderer* yuv_renderer = yuv_renderer_.release(); QuadRenderer* ext_rgb_renderer = ext_rgb_renderer_.release(); - helper_.RunInGlContext([rgb_renderer, yuv_renderer, ext_rgb_renderer] { + gpu_helper_.RunInGlContext([rgb_renderer, yuv_renderer, ext_rgb_renderer] { if (rgb_renderer) { rgb_renderer->GlTeardown(); delete rgb_renderer; @@ -376,17 +377,21 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); ::mediapipe::Status ImageTransformationCalculator::RenderCpu( CalculatorContext* cc) { - const auto& input_img = cc->Inputs().Tag("IMAGE").Get(); - cv::Mat input_mat = formats::MatView(&input_img); - cv::Mat scaled_mat; + cv::Mat input_mat; + mediapipe::ImageFormat::Format format; - const int input_width = input_img.Width(); - const int input_height = input_img.Height(); + const auto& input = cc->Inputs().Tag(kImageFrameTag).Get(); + input_mat = formats::MatView(&input); + format = input.Format(); + + const int input_width = input_mat.cols; + const int input_height = input_mat.rows; if (!output_height_ || !output_width_) { output_height_ = input_height; output_width_ = input_width; } + cv::Mat scaled_mat; if (scale_mode_ == mediapipe::ScaleMode_Mode_STRETCH) { cv::resize(input_mat, scaled_mat, cv::Size(output_width_, output_height_)); } else { @@ -443,10 +448,12 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); } std::unique_ptr output_frame( - new ImageFrame(input_img.Format(), output_width, output_height)); + new ImageFrame(format, output_width, output_height)); cv::Mat output_mat = formats::MatView(output_frame.get()); flipped_mat.copyTo(output_mat); - cc->Outputs().Tag("IMAGE").Add(output_frame.release(), cc->InputTimestamp()); + cc->Outputs() + .Tag(kImageFrameTag) + .Add(output_frame.release(), cc->InputTimestamp()); return ::mediapipe::OkStatus(); } @@ -454,7 +461,7 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); ::mediapipe::Status ImageTransformationCalculator::RenderGpu( CalculatorContext* cc) { #if !defined(MEDIAPIPE_DISABLE_GPU) - const auto& input = cc->Inputs().Tag("IMAGE_GPU").Get(); + const auto& input = cc->Inputs().Tag(kGpuBufferTag).Get(); const int input_width = input.width(); const int input_height = input.height(); @@ -485,11 +492,11 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); {"video_frame_y", "video_frame_uv"})); } renderer = yuv_renderer_.get(); - src1 = helper_.CreateSourceTexture(input, 0); + src1 = gpu_helper_.CreateSourceTexture(input, 0); } else // NOLINT(readability/braces) #endif // iOS { - src1 = helper_.CreateSourceTexture(input); + src1 = gpu_helper_.CreateSourceTexture(input); #if defined(TEXTURE_EXTERNAL_OES) if (src1.target() == GL_TEXTURE_EXTERNAL_OES) { if (!ext_rgb_renderer_) { @@ -515,10 +522,10 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); mediapipe::FrameRotation rotation = mediapipe::FrameRotationFromDegrees(RotationModeToDegrees(rotation_)); - auto dst = helper_.CreateDestinationTexture(output_width, output_height, - input.format()); + auto dst = gpu_helper_.CreateDestinationTexture(output_width, output_height, + input.format()); - helper_.BindFramebuffer(dst); // GL_TEXTURE0 + gpu_helper_.BindFramebuffer(dst); // GL_TEXTURE0 glActiveTexture(GL_TEXTURE1); glBindTexture(src1.target(), src1.name()); @@ -533,8 +540,8 @@ REGISTER_CALCULATOR(ImageTransformationCalculator); // Execute GL commands, before getting result. glFlush(); - auto output = dst.GetFrame(); - cc->Outputs().Tag("IMAGE_GPU").Add(output.release(), cc->InputTimestamp()); + auto output = dst.template GetFrame(); + cc->Outputs().Tag(kGpuBufferTag).Add(output.release(), cc->InputTimestamp()); #endif // !MEDIAPIPE_DISABLE_GPU diff --git a/mediapipe/calculators/image/recolor_calculator.cc b/mediapipe/calculators/image/recolor_calculator.cc index 07f347a1..c8d3d172 100644 --- a/mediapipe/calculators/image/recolor_calculator.cc +++ b/mediapipe/calculators/image/recolor_calculator.cc @@ -32,6 +32,11 @@ namespace { enum { ATTRIB_VERTEX, ATTRIB_TEXTURE_POSITION, NUM_ATTRIBUTES }; + +constexpr char kImageFrameTag[] = "IMAGE"; +constexpr char kMaskCpuTag[] = "MASK"; +constexpr char kGpuBufferTag[] = "IMAGE_GPU"; +constexpr char kMaskGpuTag[] = "MASK_GPU"; } // namespace namespace mediapipe { @@ -112,39 +117,41 @@ REGISTER_CALCULATOR(RecolorCalculator); bool use_gpu = false; #if !defined(MEDIAPIPE_DISABLE_GPU) - if (cc->Inputs().HasTag("IMAGE_GPU")) { - cc->Inputs().Tag("IMAGE_GPU").Set(); + if (cc->Inputs().HasTag(kGpuBufferTag)) { + cc->Inputs().Tag(kGpuBufferTag).Set(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU - if (cc->Inputs().HasTag("IMAGE")) { - cc->Inputs().Tag("IMAGE").Set(); + if (cc->Inputs().HasTag(kImageFrameTag)) { + cc->Inputs().Tag(kImageFrameTag).Set(); } #if !defined(MEDIAPIPE_DISABLE_GPU) - if (cc->Inputs().HasTag("MASK_GPU")) { - cc->Inputs().Tag("MASK_GPU").Set(); + if (cc->Inputs().HasTag(kMaskGpuTag)) { + cc->Inputs().Tag(kMaskGpuTag).Set(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU - if (cc->Inputs().HasTag("MASK")) { - cc->Inputs().Tag("MASK").Set(); + if (cc->Inputs().HasTag(kMaskCpuTag)) { + cc->Inputs().Tag(kMaskCpuTag).Set(); } #if !defined(MEDIAPIPE_DISABLE_GPU) - if (cc->Outputs().HasTag("IMAGE_GPU")) { - cc->Outputs().Tag("IMAGE_GPU").Set(); + if (cc->Outputs().HasTag(kGpuBufferTag)) { + cc->Outputs().Tag(kGpuBufferTag).Set(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU - if (cc->Outputs().HasTag("IMAGE")) { - cc->Outputs().Tag("IMAGE").Set(); + if (cc->Outputs().HasTag(kImageFrameTag)) { + cc->Outputs().Tag(kImageFrameTag).Set(); } // Confirm only one of the input streams is present. - RET_CHECK(cc->Inputs().HasTag("IMAGE") ^ cc->Inputs().HasTag("IMAGE_GPU")); + RET_CHECK(cc->Inputs().HasTag(kImageFrameTag) ^ + cc->Inputs().HasTag(kGpuBufferTag)); // Confirm only one of the output streams is present. - RET_CHECK(cc->Outputs().HasTag("IMAGE") ^ cc->Outputs().HasTag("IMAGE_GPU")); + RET_CHECK(cc->Outputs().HasTag(kImageFrameTag) ^ + cc->Outputs().HasTag(kGpuBufferTag)); if (use_gpu) { #if !defined(MEDIAPIPE_DISABLE_GPU) @@ -158,7 +165,7 @@ REGISTER_CALCULATOR(RecolorCalculator); ::mediapipe::Status RecolorCalculator::Open(CalculatorContext* cc) { cc->SetOffset(TimestampDiff(0)); - if (cc->Inputs().HasTag("IMAGE_GPU")) { + if (cc->Inputs().HasTag(kGpuBufferTag)) { use_gpu_ = true; #if !defined(MEDIAPIPE_DISABLE_GPU) MP_RETURN_IF_ERROR(gpu_helper_.Open(cc)); @@ -201,12 +208,12 @@ REGISTER_CALCULATOR(RecolorCalculator); } ::mediapipe::Status RecolorCalculator::RenderCpu(CalculatorContext* cc) { - if (cc->Inputs().Tag("MASK").IsEmpty()) { + if (cc->Inputs().Tag(kMaskCpuTag).IsEmpty()) { return ::mediapipe::OkStatus(); } // Get inputs and setup output. - const auto& input_img = cc->Inputs().Tag("IMAGE").Get(); - const auto& mask_img = cc->Inputs().Tag("MASK").Get(); + const auto& input_img = cc->Inputs().Tag(kImageFrameTag).Get(); + const auto& mask_img = cc->Inputs().Tag(kMaskCpuTag).Get(); cv::Mat input_mat = formats::MatView(&input_img); cv::Mat mask_mat = formats::MatView(&mask_img); @@ -254,19 +261,21 @@ REGISTER_CALCULATOR(RecolorCalculator); } } - cc->Outputs().Tag("IMAGE").Add(output_img.release(), cc->InputTimestamp()); + cc->Outputs() + .Tag(kImageFrameTag) + .Add(output_img.release(), cc->InputTimestamp()); return ::mediapipe::OkStatus(); } ::mediapipe::Status RecolorCalculator::RenderGpu(CalculatorContext* cc) { - if (cc->Inputs().Tag("MASK_GPU").IsEmpty()) { + if (cc->Inputs().Tag(kMaskGpuTag).IsEmpty()) { return ::mediapipe::OkStatus(); } #if !defined(MEDIAPIPE_DISABLE_GPU) // Get inputs and setup output. - const Packet& input_packet = cc->Inputs().Tag("IMAGE_GPU").Value(); - const Packet& mask_packet = cc->Inputs().Tag("MASK_GPU").Value(); + const Packet& input_packet = cc->Inputs().Tag(kGpuBufferTag).Value(); + const Packet& mask_packet = cc->Inputs().Tag(kMaskGpuTag).Value(); const auto& input_buffer = input_packet.Get(); const auto& mask_buffer = mask_packet.Get(); @@ -296,7 +305,7 @@ REGISTER_CALCULATOR(RecolorCalculator); // Send result image in GPU packet. auto output = dst_tex.GetFrame(); - cc->Outputs().Tag("IMAGE_GPU").Add(output.release(), cc->InputTimestamp()); + cc->Outputs().Tag(kGpuBufferTag).Add(output.release(), cc->InputTimestamp()); // Cleanup img_tex.Release(); diff --git a/mediapipe/calculators/tflite/BUILD b/mediapipe/calculators/tflite/BUILD index abc83f68..1932bbbf 100644 --- a/mediapipe/calculators/tflite/BUILD +++ b/mediapipe/calculators/tflite/BUILD @@ -243,6 +243,7 @@ cc_library( "@org_tensorflow//tensorflow/lite/delegates/gpu:metal_delegate_internal", ], "//conditions:default": [ + "//mediapipe/util/tflite:tflite_gpu_runner", "//mediapipe/gpu:gl_calculator_helper", "//mediapipe/gpu:gpu_buffer", "@org_tensorflow//tensorflow/lite/delegates/gpu/common:shape", diff --git a/mediapipe/calculators/tflite/tflite_converter_calculator.cc b/mediapipe/calculators/tflite/tflite_converter_calculator.cc index 8f3af1c3..3d5cdff8 100644 --- a/mediapipe/calculators/tflite/tflite_converter_calculator.cc +++ b/mediapipe/calculators/tflite/tflite_converter_calculator.cc @@ -63,6 +63,10 @@ typedef Eigen::Matrix typedef Eigen::Matrix ColMajorMatrixXf; +constexpr char kImageFrameTag[] = "IMAGE"; +constexpr char kGpuBufferTag[] = "IMAGE_GPU"; +constexpr char kTensorsTag[] = "TENSORS"; +constexpr char kTensorsGpuTag[] = "TENSORS_GPU"; } // namespace namespace mediapipe { @@ -124,6 +128,9 @@ struct GPUData { // GPU tensors are currently only supported on mobile platforms. // This calculator uses FixedSizeInputStreamHandler by default. // +// Note: Input defines output, so only these type sets are supported: +// IMAGE -> TENSORS | IMAGE_GPU -> TENSORS_GPU | MATRIX -> TENSORS +// class TfLiteConverterCalculator : public CalculatorBase { public: static ::mediapipe::Status GetContract(CalculatorContract* cc); @@ -138,9 +145,9 @@ class TfLiteConverterCalculator : public CalculatorBase { template ::mediapipe::Status NormalizeImage(const ImageFrame& image_frame, bool zero_center, bool flip_vertically, - float* tensor_buffer); + float* tensor_ptr); ::mediapipe::Status CopyMatrixToTensor(const Matrix& matrix, - float* tensor_buffer); + float* tensor_ptr); ::mediapipe::Status ProcessCPU(CalculatorContext* cc); ::mediapipe::Status ProcessGPU(CalculatorContext* cc); @@ -166,33 +173,35 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); ::mediapipe::Status TfLiteConverterCalculator::GetContract( CalculatorContract* cc) { - const bool has_image_tag = cc->Inputs().HasTag("IMAGE"); - const bool has_image_gpu_tag = cc->Inputs().HasTag("IMAGE_GPU"); - const bool has_matrix_tag = cc->Inputs().HasTag("MATRIX"); // Confirm only one of the input streams is present. - RET_CHECK(has_image_tag ^ has_image_gpu_tag ^ has_matrix_tag && - !(has_image_tag && has_image_gpu_tag && has_matrix_tag)); + RET_CHECK(cc->Inputs().HasTag(kImageFrameTag) ^ + cc->Inputs().HasTag(kGpuBufferTag) ^ cc->Inputs().HasTag("MATRIX")); // Confirm only one of the output streams is present. - RET_CHECK(cc->Outputs().HasTag("TENSORS") ^ - cc->Outputs().HasTag("TENSORS_GPU")); + RET_CHECK(cc->Outputs().HasTag(kTensorsTag) ^ + cc->Outputs().HasTag(kTensorsGpuTag)); bool use_gpu = false; - if (cc->Inputs().HasTag("IMAGE")) cc->Inputs().Tag("IMAGE").Set(); - if (cc->Inputs().HasTag("MATRIX")) cc->Inputs().Tag("MATRIX").Set(); + if (cc->Inputs().HasTag(kImageFrameTag)) { + cc->Inputs().Tag(kImageFrameTag).Set(); + } + if (cc->Inputs().HasTag("MATRIX")) { + cc->Inputs().Tag("MATRIX").Set(); + } #if !defined(MEDIAPIPE_DISABLE_GPU) && !defined(__EMSCRIPTEN__) - if (cc->Inputs().HasTag("IMAGE_GPU")) { - cc->Inputs().Tag("IMAGE_GPU").Set(); + if (cc->Inputs().HasTag(kGpuBufferTag)) { + cc->Inputs().Tag(kGpuBufferTag).Set(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU - if (cc->Outputs().HasTag("TENSORS")) - cc->Outputs().Tag("TENSORS").Set>(); + if (cc->Outputs().HasTag(kTensorsTag)) { + cc->Outputs().Tag(kTensorsTag).Set>(); + } #if !defined(MEDIAPIPE_DISABLE_GPU) && !defined(__EMSCRIPTEN__) - if (cc->Outputs().HasTag("TENSORS_GPU")) { - cc->Outputs().Tag("TENSORS_GPU").Set>(); + if (cc->Outputs().HasTag(kTensorsGpuTag)) { + cc->Outputs().Tag(kTensorsGpuTag).Set>(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU @@ -216,8 +225,8 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); MP_RETURN_IF_ERROR(LoadOptions(cc)); - if (cc->Inputs().HasTag("IMAGE_GPU") || - cc->Outputs().HasTag("IMAGE_OUT_GPU")) { + if (cc->Inputs().HasTag(kGpuBufferTag) || + cc->Outputs().HasTag(kGpuBufferTag)) { #if !defined(MEDIAPIPE_DISABLE_GPU) && !defined(__EMSCRIPTEN__) use_gpu_ = true; #else @@ -227,8 +236,8 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); if (use_gpu_) { // Cannot mix CPU/GPU streams. - RET_CHECK(cc->Inputs().HasTag("IMAGE_GPU") && - cc->Outputs().HasTag("TENSORS_GPU")); + RET_CHECK(cc->Inputs().HasTag(kGpuBufferTag) && + cc->Outputs().HasTag(kTensorsGpuTag)); // Cannot use quantization. use_quantized_tensors_ = false; #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) @@ -248,7 +257,6 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); ::mediapipe::Status TfLiteConverterCalculator::Process(CalculatorContext* cc) { if (use_gpu_) { - // GpuBuffer to tflite::gpu::GlBuffer conversion. if (!initialized_) { MP_RETURN_IF_ERROR(InitGpu(cc)); initialized_ = true; @@ -259,7 +267,6 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); // Convert to CPU tensors or Matrix type. MP_RETURN_IF_ERROR(ProcessCPU(cc)); } - return ::mediapipe::OkStatus(); } @@ -275,24 +282,26 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); ::mediapipe::Status TfLiteConverterCalculator::ProcessCPU( CalculatorContext* cc) { - if (cc->Inputs().HasTag("IMAGE")) { + if (cc->Inputs().HasTag(kImageFrameTag)) { // CPU ImageFrame to TfLiteTensor conversion. - const auto& image_frame = cc->Inputs().Tag("IMAGE").Get(); + const auto& image_frame = + cc->Inputs().Tag(kImageFrameTag).Get(); const int height = image_frame.Height(); const int width = image_frame.Width(); const int channels = image_frame.NumberOfChannels(); const int channels_preserved = std::min(channels, max_num_channels_); + const mediapipe::ImageFormat::Format format = image_frame.Format(); if (!initialized_) { - if (!(image_frame.Format() == mediapipe::ImageFormat::SRGBA || - image_frame.Format() == mediapipe::ImageFormat::SRGB || - image_frame.Format() == mediapipe::ImageFormat::GRAY8 || - image_frame.Format() == mediapipe::ImageFormat::VEC32F1)) + if (!(format == mediapipe::ImageFormat::SRGBA || + format == mediapipe::ImageFormat::SRGB || + format == mediapipe::ImageFormat::GRAY8 || + format == mediapipe::ImageFormat::VEC32F1)) RET_CHECK_FAIL() << "Unsupported CPU input format."; TfLiteQuantization quant; if (use_quantized_tensors_) { - RET_CHECK(image_frame.Format() != mediapipe::ImageFormat::VEC32F1) + RET_CHECK(format != mediapipe::ImageFormat::VEC32F1) << "Only 8-bit input images are supported for quantization."; quant.type = kTfLiteAffineQuantization; quant.params = nullptr; @@ -349,8 +358,9 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); auto output_tensors = absl::make_unique>(); output_tensors->emplace_back(*tensor); - cc->Outputs().Tag("TENSORS").Add(output_tensors.release(), - cc->InputTimestamp()); + cc->Outputs() + .Tag(kTensorsTag) + .Add(output_tensors.release(), cc->InputTimestamp()); } else if (cc->Inputs().HasTag("MATRIX")) { // CPU Matrix to TfLiteTensor conversion. @@ -371,15 +381,16 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); interpreter_->ResizeInputTensor(tensor_idx, {height, width, channels}); interpreter_->AllocateTensors(); - float* tensor_buffer = tensor->data.f; - RET_CHECK(tensor_buffer); + float* tensor_ptr = tensor->data.f; + RET_CHECK(tensor_ptr); - MP_RETURN_IF_ERROR(CopyMatrixToTensor(matrix, tensor_buffer)); + MP_RETURN_IF_ERROR(CopyMatrixToTensor(matrix, tensor_ptr)); auto output_tensors = absl::make_unique>(); output_tensors->emplace_back(*tensor); - cc->Outputs().Tag("TENSORS").Add(output_tensors.release(), - cc->InputTimestamp()); + cc->Outputs() + .Tag(kTensorsTag) + .Add(output_tensors.release(), cc->InputTimestamp()); } return ::mediapipe::OkStatus(); @@ -389,7 +400,8 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); CalculatorContext* cc) { #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) // GpuBuffer to tflite::gpu::GlBuffer conversion. - const auto& input = cc->Inputs().Tag("IMAGE_GPU").Get(); + const auto& input = + cc->Inputs().Tag(kGpuBufferTag).Get(); MP_RETURN_IF_ERROR( gpu_helper_.RunInGlContext([this, &input]() -> ::mediapipe::Status { // Convert GL texture into TfLite GlBuffer (SSBO). @@ -421,11 +433,12 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); return ::mediapipe::OkStatus(); })); cc->Outputs() - .Tag("TENSORS_GPU") + .Tag(kTensorsGpuTag) .Add(output_tensors.release(), cc->InputTimestamp()); #elif defined(MEDIAPIPE_IOS) // GpuBuffer to id conversion. - const auto& input = cc->Inputs().Tag("IMAGE_GPU").Get(); + const auto& input = + cc->Inputs().Tag(kGpuBufferTag).Get(); id command_buffer = [gpu_helper_ commandBuffer]; id src_texture = [gpu_helper_ metalTextureWithGpuBuffer:input]; @@ -457,7 +470,7 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); commandBuffer:command_buffer]; cc->Outputs() - .Tag("TENSORS_GPU") + .Tag(kTensorsGpuTag) .Add(output_tensors.release(), cc->InputTimestamp()); #else RET_CHECK_FAIL() << "GPU processing is not enabled."; @@ -469,7 +482,8 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); ::mediapipe::Status TfLiteConverterCalculator::InitGpu(CalculatorContext* cc) { #if !defined(MEDIAPIPE_DISABLE_GPU) && !defined(__EMSCRIPTEN__) // Get input image sizes. - const auto& input = cc->Inputs().Tag("IMAGE_GPU").Get(); + const auto& input = + cc->Inputs().Tag(kGpuBufferTag).Get(); mediapipe::ImageFormat::Format format = mediapipe::ImageFormatForGpuBufferFormat(input.format()); gpu_data_out_ = absl::make_unique(); @@ -612,7 +626,7 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); CHECK_LE(max_num_channels_, 4); CHECK_NE(max_num_channels_, 2); #if defined(MEDIAPIPE_IOS) - if (cc->Inputs().HasTag("IMAGE_GPU")) + if (cc->Inputs().HasTag(kGpuBufferTag)) // Currently on iOS, tflite gpu input tensor must be 4 channels, // so input image must be 4 channels also (checked in InitGpu). max_num_channels_ = 4; @@ -627,7 +641,7 @@ REGISTER_CALCULATOR(TfLiteConverterCalculator); template ::mediapipe::Status TfLiteConverterCalculator::NormalizeImage( const ImageFrame& image_frame, bool zero_center, bool flip_vertically, - float* tensor_buffer) { + float* tensor_ptr) { const int height = image_frame.Height(); const int width = image_frame.Width(); const int channels = image_frame.NumberOfChannels(); @@ -651,7 +665,7 @@ template (flip_vertically ? height - 1 - i : i) * image_frame.WidthStep()); for (int j = 0; j < width; ++j) { for (int c = 0; c < channels_preserved; ++c) { - *tensor_buffer++ = *image_ptr++ / div - sub; + *tensor_ptr++ = *image_ptr++ / div - sub; } image_ptr += channels_ignored; } @@ -661,14 +675,14 @@ template } ::mediapipe::Status TfLiteConverterCalculator::CopyMatrixToTensor( - const Matrix& matrix, float* tensor_buffer) { + const Matrix& matrix, float* tensor_ptr) { if (row_major_matrix_) { - auto matrix_map = Eigen::Map(tensor_buffer, matrix.rows(), - matrix.cols()); + auto matrix_map = + Eigen::Map(tensor_ptr, matrix.rows(), matrix.cols()); matrix_map = matrix; } else { - auto matrix_map = Eigen::Map(tensor_buffer, matrix.rows(), - matrix.cols()); + auto matrix_map = + Eigen::Map(tensor_ptr, matrix.rows(), matrix.cols()); matrix_map = matrix; } diff --git a/mediapipe/calculators/tflite/tflite_inference_calculator.cc b/mediapipe/calculators/tflite/tflite_inference_calculator.cc index b0eb3c50..765ab4dd 100644 --- a/mediapipe/calculators/tflite/tflite_inference_calculator.cc +++ b/mediapipe/calculators/tflite/tflite_inference_calculator.cc @@ -36,6 +36,7 @@ #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) #include "mediapipe/gpu/gl_calculator_helper.h" #include "mediapipe/gpu/gpu_buffer.h" +#include "mediapipe/util/tflite/tflite_gpu_runner.h" #include "tensorflow/lite/delegates/gpu/common/shape.h" #include "tensorflow/lite/delegates/gpu/gl/gl_buffer.h" #include "tensorflow/lite/delegates/gpu/gl/gl_program.h" @@ -75,6 +76,9 @@ typedef id GpuTensor; // Round up n to next multiple of m. size_t RoundUp(size_t n, size_t m) { return ((n + m - 1) / m) * m; } // NOLINT + +constexpr char kTensorsTag[] = "TENSORS"; +constexpr char kTensorsGpuTag[] = "TENSORS_GPU"; } // namespace #if defined(MEDIAPIPE_EDGE_TPU) @@ -219,6 +223,7 @@ class TfLiteInferenceCalculator : public CalculatorBase { ::mediapipe::Status LoadModel(CalculatorContext* cc); ::mediapipe::StatusOr GetModelAsPacket(const CalculatorContext& cc); ::mediapipe::Status LoadDelegate(CalculatorContext* cc); + ::mediapipe::Status InitTFLiteGPURunner(); Packet model_packet_; std::unique_ptr interpreter_; @@ -228,6 +233,7 @@ class TfLiteInferenceCalculator : public CalculatorBase { mediapipe::GlCalculatorHelper gpu_helper_; std::vector> gpu_data_in_; std::vector> gpu_data_out_; + std::unique_ptr tflite_gpu_runner_; #elif defined(MEDIAPIPE_IOS) MPPMetalHelper* gpu_helper_ = nullptr; std::vector> gpu_data_in_; @@ -245,6 +251,8 @@ class TfLiteInferenceCalculator : public CalculatorBase { bool gpu_input_ = false; bool gpu_output_ = false; bool use_quantized_tensors_ = false; + + bool use_advanced_gpu_api_ = false; }; REGISTER_CALCULATOR(TfLiteInferenceCalculator); @@ -252,10 +260,10 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); ::mediapipe::Status TfLiteInferenceCalculator::GetContract( CalculatorContract* cc) { - RET_CHECK(cc->Inputs().HasTag("TENSORS") ^ - cc->Inputs().HasTag("TENSORS_GPU")); - RET_CHECK(cc->Outputs().HasTag("TENSORS") ^ - cc->Outputs().HasTag("TENSORS_GPU")); + RET_CHECK(cc->Inputs().HasTag(kTensorsTag) ^ + cc->Inputs().HasTag(kTensorsGpuTag)); + RET_CHECK(cc->Outputs().HasTag(kTensorsTag) ^ + cc->Outputs().HasTag(kTensorsGpuTag)); const auto& options = cc->Options<::mediapipe::TfLiteInferenceCalculatorOptions>(); @@ -266,26 +274,26 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); bool use_gpu = options.has_delegate() ? options.delegate().has_gpu() : options.use_gpu(); - if (cc->Inputs().HasTag("TENSORS")) - cc->Inputs().Tag("TENSORS").Set>(); + if (cc->Inputs().HasTag(kTensorsTag)) + cc->Inputs().Tag(kTensorsTag).Set>(); #if !defined(MEDIAPIPE_DISABLE_GPU) && !defined(__EMSCRIPTEN__) - if (cc->Inputs().HasTag("TENSORS_GPU")) { + if (cc->Inputs().HasTag(kTensorsGpuTag)) { RET_CHECK(!options.has_delegate() || options.delegate().has_gpu()) << "GPU input is compatible with GPU delegate only."; - cc->Inputs().Tag("TENSORS_GPU").Set>(); + cc->Inputs().Tag(kTensorsGpuTag).Set>(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU - if (cc->Outputs().HasTag("TENSORS")) - cc->Outputs().Tag("TENSORS").Set>(); + if (cc->Outputs().HasTag(kTensorsTag)) + cc->Outputs().Tag(kTensorsTag).Set>(); #if !defined(MEDIAPIPE_DISABLE_GPU) && !defined(__EMSCRIPTEN__) - if (cc->Outputs().HasTag("TENSORS_GPU")) { + if (cc->Outputs().HasTag(kTensorsGpuTag)) { RET_CHECK(!options.has_delegate() || options.delegate().has_gpu()) << "GPU output is compatible with GPU delegate only."; - cc->Outputs().Tag("TENSORS_GPU").Set>(); + cc->Outputs().Tag(kTensorsGpuTag).Set>(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU @@ -320,27 +328,31 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); cc->Options<::mediapipe::TfLiteInferenceCalculatorOptions>(); gpu_inference_ = options.use_gpu(); - if (cc->Inputs().HasTag("TENSORS_GPU")) { + if (cc->Inputs().HasTag(kTensorsGpuTag)) { #if !defined(MEDIAPIPE_DISABLE_GPU) && !defined(__EMSCRIPTEN__) gpu_input_ = true; gpu_inference_ = true; // Inference must be on GPU also. #else - RET_CHECK(!cc->Inputs().HasTag("TENSORS_GPU")) + RET_CHECK(!cc->Inputs().HasTag(kTensorsGpuTag)) << "GPU processing not enabled."; #endif // !MEDIAPIPE_DISABLE_GPU } - if (cc->Outputs().HasTag("TENSORS_GPU")) { + if (cc->Outputs().HasTag(kTensorsGpuTag)) { #if !defined(MEDIAPIPE_DISABLE_GPU) && !defined(__EMSCRIPTEN__) gpu_output_ = true; - RET_CHECK(cc->Inputs().HasTag("TENSORS_GPU")) + RET_CHECK(cc->Inputs().HasTag(kTensorsGpuTag)) << "GPU output must also have GPU Input."; #else - RET_CHECK(!cc->Inputs().HasTag("TENSORS_GPU")) + RET_CHECK(!cc->Inputs().HasTag(kTensorsGpuTag)) << "GPU processing not enabled."; #endif // !MEDIAPIPE_DISABLE_GPU } + const auto& calculator_opts = + cc->Options(); + use_advanced_gpu_api_ = false; + MP_RETURN_IF_ERROR(LoadModel(cc)); if (gpu_inference_) { @@ -352,8 +364,12 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); #endif #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) - MP_RETURN_IF_ERROR(gpu_helper_.RunInGlContext( - [this, &cc]() -> ::mediapipe::Status { return LoadDelegate(cc); })); + MP_RETURN_IF_ERROR( + gpu_helper_.RunInGlContext([this, &cc]() -> ::mediapipe::Status { + return use_advanced_gpu_api_ ? InitTFLiteGPURunner() + : LoadDelegate(cc); + })); + if (use_advanced_gpu_api_) return ::mediapipe::OkStatus(); #else MP_RETURN_IF_ERROR(LoadDelegate(cc)); #endif @@ -365,13 +381,51 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); return ::mediapipe::OkStatus(); } +::mediapipe::Status TfLiteInferenceCalculator::InitTFLiteGPURunner() { +#if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) + // Create and bind OpenGL buffers for outputs. + // These buffers are created onve and later their ids are jut passed to the + // calculator outputs. + + gpu_data_out_.resize(tflite_gpu_runner_->outputs_size()); + for (int i = 0; i < tflite_gpu_runner_->outputs_size(); ++i) { + gpu_data_out_[i] = absl::make_unique(); + ASSIGN_OR_RETURN(gpu_data_out_[i]->elements, + tflite_gpu_runner_->GetOutputElements(i)); + // Create and bind input buffer. + RET_CHECK_CALL(::tflite::gpu::gl::CreateReadWriteShaderStorageBuffer( + gpu_data_out_[i]->elements, &gpu_data_out_[i]->buffer)); + } + RET_CHECK_CALL(tflite_gpu_runner_->Build()); +#endif + return ::mediapipe::OkStatus(); +} + ::mediapipe::Status TfLiteInferenceCalculator::Process(CalculatorContext* cc) { // 1. Receive pre-processed tensor inputs. - if (gpu_input_) { - // Read GPU input into SSBO. + if (use_advanced_gpu_api_) { #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) const auto& input_tensors = cc->Inputs().Tag("TENSORS_GPU").Get>(); + RET_CHECK(input_tensors.empty()); + MP_RETURN_IF_ERROR(gpu_helper_.RunInGlContext( + [this, &input_tensors]() -> ::mediapipe::Status { + for (int i = 0; i < input_tensors.size(); ++i) { + MP_RETURN_IF_ERROR(tflite_gpu_runner_->BindSSBOToInputTensor( + input_tensors[i].id(), i)); + } + for (int i = 0; i < gpu_data_out_.size(); ++i) { + MP_RETURN_IF_ERROR(tflite_gpu_runner_->BindSSBOToOutputTensor( + gpu_data_out_[i]->buffer.id(), i)); + } + return ::mediapipe::OkStatus(); + })); +#endif + } else if (gpu_input_) { + // Read GPU input into SSBO. +#if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) + const auto& input_tensors = + cc->Inputs().Tag(kTensorsGpuTag).Get>(); RET_CHECK_GT(input_tensors.size(), 0); MP_RETURN_IF_ERROR(gpu_helper_.RunInGlContext( [this, &input_tensors]() -> ::mediapipe::Status { @@ -386,7 +440,7 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); })); #elif defined(MEDIAPIPE_IOS) const auto& input_tensors = - cc->Inputs().Tag("TENSORS_GPU").Get>(); + cc->Inputs().Tag(kTensorsGpuTag).Get>(); RET_CHECK_GT(input_tensors.size(), 0); // Explicit copy input with conversion float 32 bits to 16 bits. gpu_data_in_.resize(input_tensors.size()); @@ -413,7 +467,7 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); } else { // Read CPU input into tensors. const auto& input_tensors = - cc->Inputs().Tag("TENSORS").Get>(); + cc->Inputs().Tag(kTensorsTag).Get>(); RET_CHECK_GT(input_tensors.size(), 0); for (int i = 0; i < input_tensors.size(); ++i) { const TfLiteTensor* input_tensor = &input_tensors[i]; @@ -437,7 +491,11 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) MP_RETURN_IF_ERROR( gpu_helper_.RunInGlContext([this]() -> ::mediapipe::Status { - RET_CHECK_EQ(interpreter_->Invoke(), kTfLiteOk); + if (use_advanced_gpu_api_) { + RET_CHECK(tflite_gpu_runner_->Invoke().ok()); + } else { + RET_CHECK_EQ(interpreter_->Invoke(), kTfLiteOk); + } return ::mediapipe::OkStatus(); })); #elif defined(MEDIAPIPE_IOS) @@ -448,7 +506,18 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); } // 3. Output processed tensors. - if (gpu_output_) { + if (use_advanced_gpu_api_) { +#if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) + auto output_tensors = absl::make_unique>(); + output_tensors->resize(gpu_data_out_.size()); + for (int i = 0; i < gpu_data_out_.size(); ++i) { + output_tensors->at(i) = gpu_data_out_[0]->buffer.MakeRef(); + } + cc->Outputs() + .Tag("TENSORS_GPU") + .Add(output_tensors.release(), cc->InputTimestamp()); +#endif + } else if (gpu_output_) { #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) // Output result tensors (GPU). auto output_tensors = absl::make_unique>(); @@ -464,7 +533,7 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); return ::mediapipe::OkStatus(); })); cc->Outputs() - .Tag("TENSORS_GPU") + .Tag(kTensorsGpuTag) .Add(output_tensors.release(), cc->InputTimestamp()); #elif defined(MEDIAPIPE_IOS) // Output result tensors (GPU). @@ -488,7 +557,7 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); [convert_command endEncoding]; [command_buffer commit]; cc->Outputs() - .Tag("TENSORS_GPU") + .Tag(kTensorsGpuTag) .Add(output_tensors.release(), cc->InputTimestamp()); #else RET_CHECK_FAIL() << "GPU processing not enabled."; @@ -501,8 +570,9 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); TfLiteTensor* tensor = interpreter_->tensor(tensor_indexes[i]); output_tensors->emplace_back(*tensor); } - cc->Outputs().Tag("TENSORS").Add(output_tensors.release(), - cc->InputTimestamp()); + cc->Outputs() + .Tag(kTensorsTag) + .Add(output_tensors.release(), cc->InputTimestamp()); } return ::mediapipe::OkStatus(); @@ -557,6 +627,20 @@ REGISTER_CALCULATOR(TfLiteInferenceCalculator); .Tag("CUSTOM_OP_RESOLVER") .Get(); } + +#if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) + if (use_advanced_gpu_api_) { + tflite::gpu::InferenceOptions options; + options.priority1 = tflite::gpu::InferencePriority::MIN_LATENCY; + options.priority2 = tflite::gpu::InferencePriority::AUTO; + options.priority3 = tflite::gpu::InferencePriority::AUTO; + options.usage = tflite::gpu::InferenceUsage::SUSTAINED_SPEED; + tflite_gpu_runner_ = + std::make_unique(options); + return tflite_gpu_runner_->InitializeWithModel(model); + } +#endif + #if defined(MEDIAPIPE_EDGE_TPU) interpreter_ = BuildEdgeTpuInterpreter(model, &op_resolver, edgetpu_context_.get()); diff --git a/mediapipe/calculators/tflite/tflite_inference_calculator.proto b/mediapipe/calculators/tflite/tflite_inference_calculator.proto index a764e89f..d784dc2d 100644 --- a/mediapipe/calculators/tflite/tflite_inference_calculator.proto +++ b/mediapipe/calculators/tflite/tflite_inference_calculator.proto @@ -42,7 +42,11 @@ message TfLiteInferenceCalculatorOptions { message TfLite {} // Delegate to run GPU inference depending on the device. // (Can use OpenGl, OpenCl, Metal depending on the device.) - message Gpu {} + message Gpu { + // Experimental, Android/Linux only. Use TFLite GPU delegate API2 for + // the NN inference. + optional bool use_advanced_gpu_api = 1 [default = false]; + } // Android only. message Nnapi {} message Xnnpack { diff --git a/mediapipe/calculators/tflite/tflite_tensors_to_detections_calculator.cc b/mediapipe/calculators/tflite/tflite_tensors_to_detections_calculator.cc index 732adb26..be679643 100644 --- a/mediapipe/calculators/tflite/tflite_tensors_to_detections_calculator.cc +++ b/mediapipe/calculators/tflite/tflite_tensors_to_detections_calculator.cc @@ -47,10 +47,11 @@ #endif // iOS namespace { - constexpr int kNumInputTensorsWithAnchors = 3; constexpr int kNumCoordsPerBox = 4; +constexpr char kTensorsTag[] = "TENSORS"; +constexpr char kTensorsGpuTag[] = "TENSORS_GPU"; } // namespace namespace mediapipe { @@ -200,13 +201,13 @@ REGISTER_CALCULATOR(TfLiteTensorsToDetectionsCalculator); bool use_gpu = false; - if (cc->Inputs().HasTag("TENSORS")) { - cc->Inputs().Tag("TENSORS").Set>(); + if (cc->Inputs().HasTag(kTensorsTag)) { + cc->Inputs().Tag(kTensorsTag).Set>(); } #if !defined(MEDIAPIPE_DISABLE_GPU) && !defined(__EMSCRIPTEN__) - if (cc->Inputs().HasTag("TENSORS_GPU")) { - cc->Inputs().Tag("TENSORS_GPU").Set>(); + if (cc->Inputs().HasTag(kTensorsGpuTag)) { + cc->Inputs().Tag(kTensorsGpuTag).Set>(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU @@ -236,7 +237,7 @@ REGISTER_CALCULATOR(TfLiteTensorsToDetectionsCalculator); CalculatorContext* cc) { cc->SetOffset(TimestampDiff(0)); - if (cc->Inputs().HasTag("TENSORS_GPU")) { + if (cc->Inputs().HasTag(kTensorsGpuTag)) { gpu_input_ = true; #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) MP_RETURN_IF_ERROR(gpu_helper_.Open(cc)); @@ -258,8 +259,8 @@ REGISTER_CALCULATOR(TfLiteTensorsToDetectionsCalculator); ::mediapipe::Status TfLiteTensorsToDetectionsCalculator::Process( CalculatorContext* cc) { - if ((!gpu_input_ && cc->Inputs().Tag("TENSORS").IsEmpty()) || - (gpu_input_ && cc->Inputs().Tag("TENSORS_GPU").IsEmpty())) { + if ((!gpu_input_ && cc->Inputs().Tag(kTensorsTag).IsEmpty()) || + (gpu_input_ && cc->Inputs().Tag(kTensorsGpuTag).IsEmpty())) { return ::mediapipe::OkStatus(); } @@ -284,7 +285,7 @@ REGISTER_CALCULATOR(TfLiteTensorsToDetectionsCalculator); ::mediapipe::Status TfLiteTensorsToDetectionsCalculator::ProcessCPU( CalculatorContext* cc, std::vector* output_detections) { const auto& input_tensors = - cc->Inputs().Tag("TENSORS").Get>(); + cc->Inputs().Tag(kTensorsTag).Get>(); if (input_tensors.size() == 2 || input_tensors.size() == kNumInputTensorsWithAnchors) { @@ -402,7 +403,7 @@ REGISTER_CALCULATOR(TfLiteTensorsToDetectionsCalculator); CalculatorContext* cc, std::vector* output_detections) { #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) const auto& input_tensors = - cc->Inputs().Tag("TENSORS_GPU").Get>(); + cc->Inputs().Tag(kTensorsGpuTag).Get>(); RET_CHECK_GE(input_tensors.size(), 2); MP_RETURN_IF_ERROR(gpu_helper_.RunInGlContext([this, &input_tensors, &cc, @@ -466,7 +467,7 @@ REGISTER_CALCULATOR(TfLiteTensorsToDetectionsCalculator); #elif defined(MEDIAPIPE_IOS) const auto& input_tensors = - cc->Inputs().Tag("TENSORS_GPU").Get>(); + cc->Inputs().Tag(kTensorsGpuTag).Get>(); RET_CHECK_GE(input_tensors.size(), 2); // Copy inputs. diff --git a/mediapipe/calculators/tflite/tflite_tensors_to_segmentation_calculator.cc b/mediapipe/calculators/tflite/tflite_tensors_to_segmentation_calculator.cc index 7fde0322..a1193cdf 100644 --- a/mediapipe/calculators/tflite/tflite_tensors_to_segmentation_calculator.cc +++ b/mediapipe/calculators/tflite/tflite_tensors_to_segmentation_calculator.cc @@ -49,6 +49,16 @@ int NumGroups(const int size, const int group_size) { // NOLINT float Clamp(float val, float min, float max) { return std::min(std::max(val, min), max); } + +constexpr char kTensorsTag[] = "TENSORS"; +constexpr char kTensorsGpuTag[] = "TENSORS_GPU"; +constexpr char kSizeImageTag[] = "REFERENCE_IMAGE"; +constexpr char kSizeImageGpuTag[] = "REFERENCE_IMAGE_GPU"; +constexpr char kMaskTag[] = "MASK"; +constexpr char kMaskGpuTag[] = "MASK_GPU"; +constexpr char kPrevMaskTag[] = "PREV_MASK"; +constexpr char kPrevMaskGpuTag[] = "PREV_MASK_GPU"; + } // namespace namespace mediapipe { @@ -148,39 +158,39 @@ REGISTER_CALCULATOR(TfLiteTensorsToSegmentationCalculator); bool use_gpu = false; // Inputs CPU. - if (cc->Inputs().HasTag("TENSORS")) { - cc->Inputs().Tag("TENSORS").Set>(); + if (cc->Inputs().HasTag(kTensorsTag)) { + cc->Inputs().Tag(kTensorsTag).Set>(); } - if (cc->Inputs().HasTag("PREV_MASK")) { - cc->Inputs().Tag("PREV_MASK").Set(); + if (cc->Inputs().HasTag(kPrevMaskTag)) { + cc->Inputs().Tag(kPrevMaskTag).Set(); } - if (cc->Inputs().HasTag("REFERENCE_IMAGE")) { - cc->Inputs().Tag("REFERENCE_IMAGE").Set(); + if (cc->Inputs().HasTag(kSizeImageTag)) { + cc->Inputs().Tag(kSizeImageTag).Set(); } // Inputs GPU. #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) - if (cc->Inputs().HasTag("TENSORS_GPU")) { - cc->Inputs().Tag("TENSORS_GPU").Set>(); + if (cc->Inputs().HasTag(kTensorsGpuTag)) { + cc->Inputs().Tag(kTensorsGpuTag).Set>(); use_gpu |= true; } - if (cc->Inputs().HasTag("PREV_MASK_GPU")) { - cc->Inputs().Tag("PREV_MASK_GPU").Set(); + if (cc->Inputs().HasTag(kPrevMaskGpuTag)) { + cc->Inputs().Tag(kPrevMaskGpuTag).Set(); use_gpu |= true; } - if (cc->Inputs().HasTag("REFERENCE_IMAGE_GPU")) { - cc->Inputs().Tag("REFERENCE_IMAGE_GPU").Set(); + if (cc->Inputs().HasTag(kSizeImageGpuTag)) { + cc->Inputs().Tag(kSizeImageGpuTag).Set(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU // Outputs. - if (cc->Outputs().HasTag("MASK")) { - cc->Outputs().Tag("MASK").Set(); + if (cc->Outputs().HasTag(kMaskTag)) { + cc->Outputs().Tag(kMaskTag).Set(); } #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) - if (cc->Outputs().HasTag("MASK_GPU")) { - cc->Outputs().Tag("MASK_GPU").Set(); + if (cc->Outputs().HasTag(kMaskGpuTag)) { + cc->Outputs().Tag(kMaskGpuTag).Set(); use_gpu |= true; } #endif // !MEDIAPIPE_DISABLE_GPU @@ -197,7 +207,7 @@ REGISTER_CALCULATOR(TfLiteTensorsToSegmentationCalculator); CalculatorContext* cc) { cc->SetOffset(TimestampDiff(0)); - if (cc->Inputs().HasTag("TENSORS_GPU")) { + if (cc->Inputs().HasTag(kTensorsGpuTag)) { use_gpu_ = true; #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) MP_RETURN_IF_ERROR(gpu_helper_.Open(cc)); @@ -255,23 +265,22 @@ REGISTER_CALCULATOR(TfLiteTensorsToSegmentationCalculator); ::mediapipe::Status TfLiteTensorsToSegmentationCalculator::ProcessCpu( CalculatorContext* cc) { - if (cc->Inputs().Tag("TENSORS").IsEmpty()) { + if (cc->Inputs().Tag(kTensorsTag).IsEmpty()) { return ::mediapipe::OkStatus(); } // Get input streams. const auto& input_tensors = - cc->Inputs().Tag("TENSORS").Get>(); - const bool has_prev_mask = cc->Inputs().HasTag("PREV_MASK") && - !cc->Inputs().Tag("PREV_MASK").IsEmpty(); + cc->Inputs().Tag(kTensorsTag).Get>(); + const bool has_prev_mask = cc->Inputs().HasTag(kPrevMaskTag) && + !cc->Inputs().Tag(kPrevMaskTag).IsEmpty(); const ImageFrame placeholder; - const auto& input_mask = has_prev_mask - ? cc->Inputs().Tag("PREV_MASK").Get() - : placeholder; + const auto& input_mask = + has_prev_mask ? cc->Inputs().Tag(kPrevMaskTag).Get() + : placeholder; int output_width = tensor_width_, output_height = tensor_height_; - if (cc->Inputs().HasTag("REFERENCE_IMAGE")) { - const auto& input_image = - cc->Inputs().Tag("REFERENCE_IMAGE").Get(); + if (cc->Inputs().HasTag(kSizeImageTag)) { + const auto& input_image = cc->Inputs().Tag(kSizeImageTag).Get(); output_width = input_image.Width(); output_height = input_image.Height(); } @@ -353,7 +362,7 @@ REGISTER_CALCULATOR(TfLiteTensorsToSegmentationCalculator); ImageFormat::SRGBA, output_width, output_height); cv::Mat output_mat = formats::MatView(output_mask.get()); large_mask_mat.copyTo(output_mat); - cc->Outputs().Tag("MASK").Add(output_mask.release(), cc->InputTimestamp()); + cc->Outputs().Tag(kMaskTag).Add(output_mask.release(), cc->InputTimestamp()); return ::mediapipe::OkStatus(); } @@ -364,23 +373,23 @@ REGISTER_CALCULATOR(TfLiteTensorsToSegmentationCalculator); // 3. upsample small mask into output mask to be same size as input image ::mediapipe::Status TfLiteTensorsToSegmentationCalculator::ProcessGpu( CalculatorContext* cc) { - if (cc->Inputs().Tag("TENSORS_GPU").IsEmpty()) { + if (cc->Inputs().Tag(kTensorsGpuTag).IsEmpty()) { return ::mediapipe::OkStatus(); } #if !defined(MEDIAPIPE_DISABLE_GL_COMPUTE) // Get input streams. const auto& input_tensors = - cc->Inputs().Tag("TENSORS_GPU").Get>(); - const bool has_prev_mask = cc->Inputs().HasTag("PREV_MASK_GPU") && - !cc->Inputs().Tag("PREV_MASK_GPU").IsEmpty(); + cc->Inputs().Tag(kTensorsGpuTag).Get>(); + const bool has_prev_mask = cc->Inputs().HasTag(kPrevMaskGpuTag) && + !cc->Inputs().Tag(kPrevMaskGpuTag).IsEmpty(); const auto& input_mask = has_prev_mask - ? cc->Inputs().Tag("PREV_MASK_GPU").Get() + ? cc->Inputs().Tag(kPrevMaskGpuTag).Get() : mediapipe::GpuBuffer(); int output_width = tensor_width_, output_height = tensor_height_; - if (cc->Inputs().HasTag("REFERENCE_IMAGE_GPU")) { + if (cc->Inputs().HasTag(kSizeImageGpuTag)) { const auto& input_image = - cc->Inputs().Tag("REFERENCE_IMAGE_GPU").Get(); + cc->Inputs().Tag(kSizeImageGpuTag).Get(); output_width = input_image.width(); output_height = input_image.height(); } @@ -441,7 +450,7 @@ REGISTER_CALCULATOR(TfLiteTensorsToSegmentationCalculator); // Send out image as GPU packet. auto output_image = output_texture.GetFrame(); cc->Outputs() - .Tag("MASK_GPU") + .Tag(kMaskGpuTag) .Add(output_image.release(), cc->InputTimestamp()); // Cleanup diff --git a/mediapipe/docs/examples.md b/mediapipe/docs/examples.md index 39ce4b06..d3a004d6 100644 --- a/mediapipe/docs/examples.md +++ b/mediapipe/docs/examples.md @@ -121,6 +121,14 @@ and model details are described in the * [Android](./hair_segmentation_mobile_gpu.md) +### Template Matching using KNIFT with CPU + +[Template Matching using KNIFT on Mobile](./template_matching_mobile_cpu.md) +shows how to use MediaPipe with TFLite model for template matching using Knift +on mobile using CPU. + +* [Android](./template_matching_mobile_cpu.md) + ## Desktop ### Hello World for C++ @@ -171,7 +179,6 @@ on desktop with webcam input. * [Desktop GPU](./face_mesh_desktop.md) * [Desktop CPU](./face_mesh_desktop.md) - ### Hand Tracking on Desktop with Webcam [Hand Tracking on Desktop with Webcam](./hand_tracking_desktop.md) shows how to @@ -198,7 +205,7 @@ GPU with live video from a webcam. * [Desktop GPU](./hair_segmentation_desktop.md) -## Google Coral (machine learning acceleration with Google EdgeTPU) +## Google Coral (ML acceleration with Google EdgeTPU) Below are code samples on how to run MediaPipe on Google Coral Dev Board. diff --git a/mediapipe/docs/images/mobile/template_matching_android_cpu.gif b/mediapipe/docs/images/mobile/template_matching_android_cpu.gif new file mode 100644 index 00000000..9aa0229e Binary files /dev/null and b/mediapipe/docs/images/mobile/template_matching_android_cpu.gif differ diff --git a/mediapipe/docs/images/mobile/template_matching_android_cpu_small.gif b/mediapipe/docs/images/mobile/template_matching_android_cpu_small.gif new file mode 100644 index 00000000..68f64aea Binary files /dev/null and b/mediapipe/docs/images/mobile/template_matching_android_cpu_small.gif differ diff --git a/mediapipe/docs/images/mobile/template_matching_mobile_graph.png b/mediapipe/docs/images/mobile/template_matching_mobile_graph.png new file mode 100644 index 00000000..3e8c2b5d Binary files /dev/null and b/mediapipe/docs/images/mobile/template_matching_mobile_graph.png differ diff --git a/mediapipe/docs/images/mobile/template_matching_mobile_template.jpg b/mediapipe/docs/images/mobile/template_matching_mobile_template.jpg new file mode 100644 index 00000000..2843efdf Binary files /dev/null and b/mediapipe/docs/images/mobile/template_matching_mobile_template.jpg differ diff --git a/mediapipe/docs/template_matching_desktop_cpu.md b/mediapipe/docs/template_matching_desktop_cpu.md new file mode 100644 index 00000000..97b2cb3f --- /dev/null +++ b/mediapipe/docs/template_matching_desktop_cpu.md @@ -0,0 +1,31 @@ +# Template Matching using KNIFT on Desktop + +This doc focuses on the +[example graph](https://github.com/google/mediapipe/tree/master/mediapipe/graphs/template_matching/template_matching_desktop.pbtxt) +that performs template matching with KNIFT (Keypoint Neural Invariant Feature +Transform) on desktop CPU. + +If you are interested in more detail about KNIFT or running the example on +mobile, please see +[Template Matching using KNIFT on Mobile (CPU)](template_matching_mobile_cpu.md). + +To build the desktop app, run: + +```bash +$ bazel build -c opt --define MEDIAPIPE_DISABLE_GPU=1 \ + mediapipe/examples/desktop/template_matching:template_matching_tflite +``` + +To run the desktop app, please specify a template index file +([example](https://github.com/google/mediapipe/tree/master/mediapipe/models/knift_index.pb)) and a +video to be matched. For how to build your own index file, please see +[here](template_matching_mobile_cpu.md#build-index-file). + +```bash +$ GLOG_logtostderr=1 bazel-bin/mediapipe/examples/desktop/template_matching/template_matching_tflite \ + --calculator_graph_config_file=mediapipe/graphs/template_matching/template_matching_desktop.pbtxt --input_side_packets="input_video_path=,output_video_path=" +``` + +## Graph + +[Source pbtxt file](https://github.com/google/mediapipe/tree/master/mediapipe/graphs/template_matching/template_matching_desktop.pbtxt) diff --git a/mediapipe/docs/template_matching_mobile_cpu.md b/mediapipe/docs/template_matching_mobile_cpu.md new file mode 100644 index 00000000..d5b87fdf --- /dev/null +++ b/mediapipe/docs/template_matching_mobile_cpu.md @@ -0,0 +1,94 @@ +# Template Matching using KNIFT on Mobile (CPU) + +This doc focuses on the +[example graph](https://github.com/google/mediapipe/tree/master/mediapipe/graphs/template_matching/template_matching_mobile_cpu.pbtxt) +that performs template matching with KNIFT (Keypoint Neural Invariant Feature +Transform) on mobile CPU. + +![template_matching_mobile_cpu.gif](images/mobile/template_matching_android_cpu.gif) + +In the visualization above, the green dots represent detected keypoints on each +frame and the red box represents the targets matched by templates using KNIFT +features (see also [model card](https://mediapipe.page.link/knift-mc)). For more +information, please see +[Google Developers Blog](https://mediapipe.page.link/knift-blog). + +## Build Index Files + +In MediaPipe, we've already provided a file in +[knift_index.pb](https://github.com/google/mediapipe/tree/master/mediapipe/models/knift_index.pb), +pre-computed from the 3 template images (of USD bills) shown below. If you'd +like to use your own template images, please follow the steps below, or +otherwise you can jump directly to [Android](#android). + +![template_matching_mobile_template.jpg](images/mobile/template_matching_mobile_template.jpg) + +### Step 1: + +Put all template images in a single directory. + +### Step 2: + +To build the index file for all templates in the directory, run: + +```bash +$ bazel build -c opt --define MEDIAPIPE_DISABLE_GPU=1 \ + mediapipe/examples/desktop/template_matching:template_matching_tflite +$ bazel-bin/mediapipe/examples/desktop/template_matching/template_matching_tflite \ + --calculator_graph_config_file=mediapipe/graphs/template_matching/index_building.pbtxt \ + --input_side_packets="file_directory=