Project import generated by Copybara.
GitOrigin-RevId: 72ff4ae24943c2ccf9905bc9e516042b0aa3dd86
This commit is contained in:
@@ -0,0 +1,11 @@
|
||||
# Modules
|
||||
|
||||
Each module (represented as a subfolder) provides subgraphs and corresponding resources (e.g. tflite models) to perform domain-specific tasks (e.g. detect faces, detect face landmarks).
|
||||
|
||||
*Modules listed below are already used in some of `mediapipe/graphs` and more graphs are being migrated to use existing and upcoming modules.*
|
||||
|
||||
| Module | Description |
|
||||
| :--- | :--- |
|
||||
| [`face_detection`](face_detection/README.md) | Subgraphs to detect faces. |
|
||||
| [`face_landmark`](face_landmark/README.md) | Subgraphs to detect and track face landmarks. |
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
# Copyright 2019 The MediaPipe Authors.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
load(
|
||||
"//mediapipe/framework/tool:mediapipe_graph.bzl",
|
||||
"mediapipe_simple_subgraph",
|
||||
)
|
||||
|
||||
licenses(["notice"]) # Apache 2.0
|
||||
|
||||
package(default_visibility = ["//visibility:public"])
|
||||
|
||||
mediapipe_simple_subgraph(
|
||||
name = "face_detection_front_cpu",
|
||||
graph = "face_detection_front_cpu.pbtxt",
|
||||
register_as = "FaceDetectionFrontCpu",
|
||||
deps = [
|
||||
"//mediapipe/calculators/image:image_transformation_calculator",
|
||||
"//mediapipe/calculators/tflite:ssd_anchors_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_converter_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_inference_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_tensors_to_detections_calculator",
|
||||
"//mediapipe/calculators/util:detection_letterbox_removal_calculator",
|
||||
"//mediapipe/calculators/util:non_max_suppression_calculator",
|
||||
],
|
||||
)
|
||||
|
||||
mediapipe_simple_subgraph(
|
||||
name = "face_detection_front_gpu",
|
||||
graph = "face_detection_front_gpu.pbtxt",
|
||||
register_as = "FaceDetectionFrontGpu",
|
||||
deps = [
|
||||
"//mediapipe/calculators/image:image_transformation_calculator",
|
||||
"//mediapipe/calculators/tflite:ssd_anchors_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_converter_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_inference_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_tensors_to_detections_calculator",
|
||||
"//mediapipe/calculators/util:detection_letterbox_removal_calculator",
|
||||
"//mediapipe/calculators/util:non_max_suppression_calculator",
|
||||
],
|
||||
)
|
||||
|
||||
exports_files(
|
||||
srcs = [
|
||||
"face_detection_front.tflite",
|
||||
],
|
||||
)
|
||||
@@ -0,0 +1,7 @@
|
||||
# face_detection
|
||||
|
||||
Subgraphs|Details
|
||||
:--- | :---
|
||||
[`FaceDetectionFrontCpu`](https://github.com/google/mediapipe/tree/master/mediapipe/modules/face_detection/face_detection_front_cpu.pbtxt)| Detects faces. Works best for images from front-facing cameras (i.e. selfie images). (CPU input, and inference is executed on CPU.)
|
||||
[`FaceDetectionFrontGpu`](https://github.com/google/mediapipe/tree/master/mediapipe/modules/face_detection/face_detection_front_gpu.pbtxt)| Detects faces. Works best for images from front-facing cameras (i.e. selfie images). (GPU input, and inference is executed on GPU.)
|
||||
|
||||
Binary file not shown.
@@ -0,0 +1,143 @@
|
||||
# MediaPipe graph to detect faces. (CPU input, and inference is executed on
|
||||
# CPU.)
|
||||
#
|
||||
# It is required that "face_detection_front.tflite" is available at
|
||||
# "mediapipe/modules/face_detection/face_detection_front.tflite"
|
||||
# path during execution.
|
||||
#
|
||||
# EXAMPLE:
|
||||
# node {
|
||||
# calculator: "FaceDetectionFrontCpu"
|
||||
# input_stream: "IMAGE:image"
|
||||
# output_stream: "DETECTIONS:face_detections"
|
||||
# }
|
||||
|
||||
type: "FaceDetectionFrontCpu"
|
||||
|
||||
# CPU image. (ImageFrame)
|
||||
input_stream: "IMAGE:image"
|
||||
|
||||
# Detected faces. (std::vector<Detection>)
|
||||
# NOTE: there will not be an output packet in the DETECTIONS stream for this
|
||||
# particular timestamp if none of faces detected. However, the MediaPipe
|
||||
# framework will internally inform the downstream calculators of the absence of
|
||||
# this packet so that they don't wait for it unnecessarily.
|
||||
output_stream: "DETECTIONS:detections"
|
||||
|
||||
# Transforms the input image on CPU to a 128x128 image. To scale the input
|
||||
# image, the scale_mode option is set to FIT to preserve the aspect ratio
|
||||
# (what is expected by the corresponding face detection model), resulting in
|
||||
# potential letterboxing in the transformed image.
|
||||
node: {
|
||||
calculator: "ImageTransformationCalculator"
|
||||
input_stream: "IMAGE:image"
|
||||
output_stream: "IMAGE:transformed_image"
|
||||
output_stream: "LETTERBOX_PADDING:letterbox_padding"
|
||||
options: {
|
||||
[mediapipe.ImageTransformationCalculatorOptions.ext] {
|
||||
output_width: 128
|
||||
output_height: 128
|
||||
scale_mode: FIT
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Converts the transformed input image on CPU into an image tensor stored as a
|
||||
# TfLiteTensor.
|
||||
node {
|
||||
calculator: "TfLiteConverterCalculator"
|
||||
input_stream: "IMAGE:transformed_image"
|
||||
output_stream: "TENSORS:input_tensors"
|
||||
}
|
||||
|
||||
# Runs a TensorFlow Lite model on CPU that takes an image tensor and outputs a
|
||||
# vector of tensors representing, for instance, detection boxes/keypoints and
|
||||
# scores.
|
||||
node {
|
||||
calculator: "TfLiteInferenceCalculator"
|
||||
input_stream: "TENSORS:input_tensors"
|
||||
output_stream: "TENSORS:detection_tensors"
|
||||
options: {
|
||||
[mediapipe.TfLiteInferenceCalculatorOptions.ext] {
|
||||
model_path: "mediapipe/modules/face_detection/face_detection_front.tflite"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Generates a single side packet containing a vector of SSD anchors based on
|
||||
# the specification in the options.
|
||||
node {
|
||||
calculator: "SsdAnchorsCalculator"
|
||||
output_side_packet: "anchors"
|
||||
options: {
|
||||
[mediapipe.SsdAnchorsCalculatorOptions.ext] {
|
||||
num_layers: 4
|
||||
min_scale: 0.1484375
|
||||
max_scale: 0.75
|
||||
input_size_height: 128
|
||||
input_size_width: 128
|
||||
anchor_offset_x: 0.5
|
||||
anchor_offset_y: 0.5
|
||||
strides: 8
|
||||
strides: 16
|
||||
strides: 16
|
||||
strides: 16
|
||||
aspect_ratios: 1.0
|
||||
fixed_anchor_size: true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Decodes the detection tensors generated by the TensorFlow Lite model, based on
|
||||
# the SSD anchors and the specification in the options, into a vector of
|
||||
# detections. Each detection describes a detected object.
|
||||
node {
|
||||
calculator: "TfLiteTensorsToDetectionsCalculator"
|
||||
input_stream: "TENSORS:detection_tensors"
|
||||
input_side_packet: "ANCHORS:anchors"
|
||||
output_stream: "DETECTIONS:unfiltered_detections"
|
||||
options: {
|
||||
[mediapipe.TfLiteTensorsToDetectionsCalculatorOptions.ext] {
|
||||
num_classes: 1
|
||||
num_boxes: 896
|
||||
num_coords: 16
|
||||
box_coord_offset: 0
|
||||
keypoint_coord_offset: 4
|
||||
num_keypoints: 6
|
||||
num_values_per_keypoint: 2
|
||||
sigmoid_score: true
|
||||
score_clipping_thresh: 100.0
|
||||
reverse_output_order: true
|
||||
x_scale: 128.0
|
||||
y_scale: 128.0
|
||||
h_scale: 128.0
|
||||
w_scale: 128.0
|
||||
min_score_thresh: 0.75
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Performs non-max suppression to remove excessive detections.
|
||||
node {
|
||||
calculator: "NonMaxSuppressionCalculator"
|
||||
input_stream: "unfiltered_detections"
|
||||
output_stream: "filtered_detections"
|
||||
options: {
|
||||
[mediapipe.NonMaxSuppressionCalculatorOptions.ext] {
|
||||
min_suppression_threshold: 0.3
|
||||
overlap_type: INTERSECTION_OVER_UNION
|
||||
algorithm: WEIGHTED
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Adjusts detection locations (already normalized to [0.f, 1.f]) on the
|
||||
# letterboxed image (after image transformation with the FIT scale mode) to the
|
||||
# corresponding locations on the same image with the letterbox removed (the
|
||||
# input image to the graph before image transformation).
|
||||
node {
|
||||
calculator: "DetectionLetterboxRemovalCalculator"
|
||||
input_stream: "DETECTIONS:filtered_detections"
|
||||
input_stream: "LETTERBOX_PADDING:letterbox_padding"
|
||||
output_stream: "DETECTIONS:detections"
|
||||
}
|
||||
@@ -0,0 +1,143 @@
|
||||
# MediaPipe graph to detect faces. (GPU input, and inference is executed on
|
||||
# GPU.)
|
||||
#
|
||||
# It is required that "face_detection_front.tflite" is available at
|
||||
# "mediapipe/modules/face_detection/face_detection_front.tflite"
|
||||
# path during execution.
|
||||
#
|
||||
# EXAMPLE:
|
||||
# node {
|
||||
# calculator: "FaceDetectionFrontGpu"
|
||||
# input_stream: "IMAGE:image"
|
||||
# output_stream: "DETECTIONS:face_detections"
|
||||
# }
|
||||
|
||||
type: "FaceDetectionFrontGpu"
|
||||
|
||||
# GPU image. (GpuBuffer)
|
||||
input_stream: "IMAGE:image"
|
||||
|
||||
# Detected faces. (std::vector<Detection>)
|
||||
# NOTE: there will not be an output packet in the DETECTIONS stream for this
|
||||
# particular timestamp if none of faces detected. However, the MediaPipe
|
||||
# framework will internally inform the downstream calculators of the absence of
|
||||
# this packet so that they don't wait for it unnecessarily.
|
||||
output_stream: "DETECTIONS:detections"
|
||||
|
||||
# Transforms the input image on GPU to a 128x128 image. To scale the input
|
||||
# image, the scale_mode option is set to FIT to preserve the aspect ratio
|
||||
# (what is expected by the corresponding face detection model), resulting in
|
||||
# potential letterboxing in the transformed image.
|
||||
node: {
|
||||
calculator: "ImageTransformationCalculator"
|
||||
input_stream: "IMAGE_GPU:image"
|
||||
output_stream: "IMAGE_GPU:transformed_image"
|
||||
output_stream: "LETTERBOX_PADDING:letterbox_padding"
|
||||
options: {
|
||||
[mediapipe.ImageTransformationCalculatorOptions.ext] {
|
||||
output_width: 128
|
||||
output_height: 128
|
||||
scale_mode: FIT
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Converts the transformed input image on GPU into an image tensor stored as a
|
||||
# TfLiteTensor.
|
||||
node {
|
||||
calculator: "TfLiteConverterCalculator"
|
||||
input_stream: "IMAGE_GPU:transformed_image"
|
||||
output_stream: "TENSORS_GPU:input_tensors"
|
||||
}
|
||||
|
||||
# Runs a TensorFlow Lite model on GPU that takes an image tensor and outputs a
|
||||
# vector of tensors representing, for instance, detection boxes/keypoints and
|
||||
# scores.
|
||||
node {
|
||||
calculator: "TfLiteInferenceCalculator"
|
||||
input_stream: "TENSORS_GPU:input_tensors"
|
||||
output_stream: "TENSORS_GPU:detection_tensors"
|
||||
options: {
|
||||
[mediapipe.TfLiteInferenceCalculatorOptions.ext] {
|
||||
model_path: "mediapipe/modules/face_detection/face_detection_front.tflite"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Generates a single side packet containing a vector of SSD anchors based on
|
||||
# the specification in the options.
|
||||
node {
|
||||
calculator: "SsdAnchorsCalculator"
|
||||
output_side_packet: "anchors"
|
||||
options: {
|
||||
[mediapipe.SsdAnchorsCalculatorOptions.ext] {
|
||||
num_layers: 4
|
||||
min_scale: 0.1484375
|
||||
max_scale: 0.75
|
||||
input_size_height: 128
|
||||
input_size_width: 128
|
||||
anchor_offset_x: 0.5
|
||||
anchor_offset_y: 0.5
|
||||
strides: 8
|
||||
strides: 16
|
||||
strides: 16
|
||||
strides: 16
|
||||
aspect_ratios: 1.0
|
||||
fixed_anchor_size: true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Decodes the detection tensors generated by the TensorFlow Lite model, based on
|
||||
# the SSD anchors and the specification in the options, into a vector of
|
||||
# detections. Each detection describes a detected object.
|
||||
node {
|
||||
calculator: "TfLiteTensorsToDetectionsCalculator"
|
||||
input_stream: "TENSORS_GPU:detection_tensors"
|
||||
input_side_packet: "ANCHORS:anchors"
|
||||
output_stream: "DETECTIONS:unfiltered_detections"
|
||||
options: {
|
||||
[mediapipe.TfLiteTensorsToDetectionsCalculatorOptions.ext] {
|
||||
num_classes: 1
|
||||
num_boxes: 896
|
||||
num_coords: 16
|
||||
box_coord_offset: 0
|
||||
keypoint_coord_offset: 4
|
||||
num_keypoints: 6
|
||||
num_values_per_keypoint: 2
|
||||
sigmoid_score: true
|
||||
score_clipping_thresh: 100.0
|
||||
reverse_output_order: true
|
||||
x_scale: 128.0
|
||||
y_scale: 128.0
|
||||
h_scale: 128.0
|
||||
w_scale: 128.0
|
||||
min_score_thresh: 0.75
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Performs non-max suppression to remove excessive detections.
|
||||
node {
|
||||
calculator: "NonMaxSuppressionCalculator"
|
||||
input_stream: "unfiltered_detections"
|
||||
output_stream: "filtered_detections"
|
||||
options: {
|
||||
[mediapipe.NonMaxSuppressionCalculatorOptions.ext] {
|
||||
min_suppression_threshold: 0.3
|
||||
overlap_type: INTERSECTION_OVER_UNION
|
||||
algorithm: WEIGHTED
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Adjusts detection locations (already normalized to [0.f, 1.f]) on the
|
||||
# letterboxed image (after image transformation with the FIT scale mode) to the
|
||||
# corresponding locations on the same image with the letterbox removed (the
|
||||
# input image to the graph before image transformation).
|
||||
node {
|
||||
calculator: "DetectionLetterboxRemovalCalculator"
|
||||
input_stream: "DETECTIONS:filtered_detections"
|
||||
input_stream: "LETTERBOX_PADDING:letterbox_padding"
|
||||
output_stream: "DETECTIONS:detections"
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
# Copyright 2019 The MediaPipe Authors.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
load(
|
||||
"//mediapipe/framework/tool:mediapipe_graph.bzl",
|
||||
"mediapipe_simple_subgraph",
|
||||
)
|
||||
|
||||
licenses(["notice"]) # Apache 2.0
|
||||
|
||||
package(default_visibility = ["//visibility:public"])
|
||||
|
||||
mediapipe_simple_subgraph(
|
||||
name = "face_landmark_cpu",
|
||||
graph = "face_landmark_cpu.pbtxt",
|
||||
register_as = "FaceLandmarkCpu",
|
||||
deps = [
|
||||
"//mediapipe/calculators/core:gate_calculator",
|
||||
"//mediapipe/calculators/core:split_vector_calculator",
|
||||
"//mediapipe/calculators/image:image_cropping_calculator",
|
||||
"//mediapipe/calculators/image:image_transformation_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_converter_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_inference_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_tensors_to_floats_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_tensors_to_landmarks_calculator",
|
||||
"//mediapipe/calculators/util:landmark_projection_calculator",
|
||||
"//mediapipe/calculators/util:thresholding_calculator",
|
||||
],
|
||||
)
|
||||
|
||||
mediapipe_simple_subgraph(
|
||||
name = "face_landmark_gpu",
|
||||
graph = "face_landmark_gpu.pbtxt",
|
||||
register_as = "FaceLandmarkGpu",
|
||||
deps = [
|
||||
"//mediapipe/calculators/core:gate_calculator",
|
||||
"//mediapipe/calculators/core:split_vector_calculator",
|
||||
"//mediapipe/calculators/image:image_cropping_calculator",
|
||||
"//mediapipe/calculators/image:image_transformation_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_converter_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_inference_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_tensors_to_floats_calculator",
|
||||
"//mediapipe/calculators/tflite:tflite_tensors_to_landmarks_calculator",
|
||||
"//mediapipe/calculators/util:landmark_projection_calculator",
|
||||
"//mediapipe/calculators/util:thresholding_calculator",
|
||||
],
|
||||
)
|
||||
|
||||
mediapipe_simple_subgraph(
|
||||
name = "face_landmark_front_cpu",
|
||||
graph = "face_landmark_front_cpu.pbtxt",
|
||||
register_as = "FaceLandmarkFrontCpu",
|
||||
deps = [
|
||||
":face_detection_front_detection_to_roi",
|
||||
":face_landmark_cpu",
|
||||
":face_landmark_landmarks_to_roi",
|
||||
"//mediapipe/calculators/core:begin_loop_calculator",
|
||||
"//mediapipe/calculators/core:clip_vector_size_calculator",
|
||||
"//mediapipe/calculators/core:end_loop_calculator",
|
||||
"//mediapipe/calculators/core:gate_calculator",
|
||||
"//mediapipe/calculators/core:merge_calculator",
|
||||
"//mediapipe/calculators/core:previous_loopback_calculator",
|
||||
"//mediapipe/calculators/image:image_properties_calculator",
|
||||
"//mediapipe/calculators/util:association_norm_rect_calculator",
|
||||
"//mediapipe/calculators/util:collection_has_min_size_calculator",
|
||||
"//mediapipe/modules/face_detection:face_detection_front_cpu",
|
||||
],
|
||||
)
|
||||
|
||||
mediapipe_simple_subgraph(
|
||||
name = "face_landmark_front_gpu",
|
||||
graph = "face_landmark_front_gpu.pbtxt",
|
||||
register_as = "FaceLandmarkFrontGpu",
|
||||
deps = [
|
||||
":face_detection_front_detection_to_roi",
|
||||
":face_landmark_gpu",
|
||||
":face_landmark_landmarks_to_roi",
|
||||
"//mediapipe/calculators/core:begin_loop_calculator",
|
||||
"//mediapipe/calculators/core:clip_vector_size_calculator",
|
||||
"//mediapipe/calculators/core:end_loop_calculator",
|
||||
"//mediapipe/calculators/core:gate_calculator",
|
||||
"//mediapipe/calculators/core:merge_calculator",
|
||||
"//mediapipe/calculators/core:previous_loopback_calculator",
|
||||
"//mediapipe/calculators/image:image_properties_calculator",
|
||||
"//mediapipe/calculators/util:association_norm_rect_calculator",
|
||||
"//mediapipe/calculators/util:collection_has_min_size_calculator",
|
||||
"//mediapipe/modules/face_detection:face_detection_front_gpu",
|
||||
],
|
||||
)
|
||||
|
||||
exports_files(
|
||||
srcs = [
|
||||
"face_landmark.tflite",
|
||||
],
|
||||
)
|
||||
|
||||
mediapipe_simple_subgraph(
|
||||
name = "face_detection_front_detection_to_roi",
|
||||
graph = "face_detection_front_detection_to_roi.pbtxt",
|
||||
register_as = "FaceDetectionFrontDetectionToRoi",
|
||||
deps = [
|
||||
"//mediapipe/calculators/util:detections_to_rects_calculator",
|
||||
"//mediapipe/calculators/util:rect_transformation_calculator",
|
||||
],
|
||||
)
|
||||
|
||||
mediapipe_simple_subgraph(
|
||||
name = "face_landmark_landmarks_to_roi",
|
||||
graph = "face_landmark_landmarks_to_roi.pbtxt",
|
||||
register_as = "FaceLandmarkLandmarksToRoi",
|
||||
deps = [
|
||||
"//mediapipe/calculators/util:detections_to_rects_calculator",
|
||||
"//mediapipe/calculators/util:landmarks_to_detection_calculator",
|
||||
"//mediapipe/calculators/util:rect_transformation_calculator",
|
||||
],
|
||||
)
|
||||
@@ -0,0 +1,9 @@
|
||||
# face_landmark
|
||||
|
||||
Subgraphs|Details
|
||||
:--- | :---
|
||||
[`FaceLandmarkCpu`](https://github.com/google/mediapipe/tree/master/mediapipe/modules/face_landmark/face_landmark_cpu.pbtxt)| Detects landmarks on a single face. (CPU input, and inference is executed on CPU.)
|
||||
[`FaceLandmarkGpu`](https://github.com/google/mediapipe/tree/master/mediapipe/modules/face_landmark/face_landmark_gpu.pbtxt)| Detects landmarks on a single face. (GPU input, and inference is executed on GPU)
|
||||
[`FaceLandmarkFrontCpu`](https://github.com/google/mediapipe/tree/master/mediapipe/modules/face_landmark/face_landmark_front_cpu.pbtxt)| Detects and tracks landmarks on multiple faces. (CPU input, and inference is executed on CPU)
|
||||
[`FaceLandmarkFrontGpu`](https://github.com/google/mediapipe/tree/master/mediapipe/modules/face_landmark/face_landmark_front_gpu.pbtxt)| Detects and tracks landmarks on multiple faces. (GPU input, and inference is executed on GPU.)
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
# MediaPipe graph to calculate face region of interest (ROI) from the very
|
||||
# first face detection in the vector of detections provided by
|
||||
# "FaceDetectionFrontCpu" or "FaceDetectionFrontGpu"
|
||||
#
|
||||
# NOTE: this graph is subject to change and should not be used directly.
|
||||
|
||||
type: "FaceDetectionFrontDetectionToRoi"
|
||||
|
||||
# Face detection. (Detection)
|
||||
input_stream: "DETECTION:detection"
|
||||
# Frame size (width and height). (std::pair<int, int>)
|
||||
input_stream: "IMAGE_SIZE:image_size"
|
||||
# ROI according to the first detection of input detections. (NormalizedRect)
|
||||
output_stream: "ROI:roi"
|
||||
|
||||
# Converts results of face detection into a rectangle (normalized by image size)
|
||||
# that encloses the face and is rotated such that the line connecting left eye
|
||||
# and right eye is aligned with the X-axis of the rectangle.
|
||||
node {
|
||||
calculator: "DetectionsToRectsCalculator"
|
||||
input_stream: "DETECTION:detection"
|
||||
input_stream: "IMAGE_SIZE:image_size"
|
||||
output_stream: "NORM_RECT:initial_roi"
|
||||
options: {
|
||||
[mediapipe.DetectionsToRectsCalculatorOptions.ext] {
|
||||
rotation_vector_start_keypoint_index: 0 # Left eye.
|
||||
rotation_vector_end_keypoint_index: 1 # Right eye.
|
||||
rotation_vector_target_angle_degrees: 0
|
||||
output_zero_rect_for_empty_detections: true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Expands and shifts the rectangle that contains the face so that it's likely
|
||||
# to cover the entire face.
|
||||
node {
|
||||
calculator: "RectTransformationCalculator"
|
||||
input_stream: "NORM_RECT:initial_roi"
|
||||
input_stream: "IMAGE_SIZE:image_size"
|
||||
output_stream: "roi"
|
||||
options: {
|
||||
[mediapipe.RectTransformationCalculatorOptions.ext] {
|
||||
scale_x: 1.5
|
||||
scale_y: 1.5
|
||||
square_long: true
|
||||
}
|
||||
}
|
||||
}
|
||||
Binary file not shown.
@@ -0,0 +1,146 @@
|
||||
# MediaPipe graph to detect/predict face landmarks. (CPU input, and inference is
|
||||
# executed on CPU.)
|
||||
#
|
||||
# It is required that "face_landmark.tflite" is available at
|
||||
# "mediapipe/modules/face_landmark/face_landmark.tflite"
|
||||
# path during execution.
|
||||
#
|
||||
# EXAMPLE:
|
||||
# node {
|
||||
# calculator: "FaceLandmarkCpu"
|
||||
# input_stream: "IMAGE:image"
|
||||
# input_stream: "ROI:face_roi"
|
||||
# output_stream: "LANDMARKS:face_landmarks"
|
||||
# }
|
||||
|
||||
type: "FaceLandmarkCpu"
|
||||
|
||||
# CPU image. (ImageFrame)
|
||||
input_stream: "IMAGE:image"
|
||||
# ROI (region of interest) within the given image where a face is located.
|
||||
# (NormalizedRect)
|
||||
input_stream: "ROI:roi"
|
||||
|
||||
# 468 face landmarks within the given ROI. (NormalizedLandmarkList)
|
||||
# NOTE: if a face is not present within the given ROI, for this particular
|
||||
# timestamp there will not be an output packet in the LANDMARKS stream. However,
|
||||
# the MediaPipe framework will internally inform the downstream calculators of
|
||||
# the absence of this packet so that they don't wait for it unnecessarily.
|
||||
output_stream: "LANDMARKS:face_landmarks"
|
||||
|
||||
# Crops the input image to the region of interest.
|
||||
node {
|
||||
calculator: "ImageCroppingCalculator"
|
||||
input_stream: "IMAGE:image"
|
||||
input_stream: "NORM_RECT:roi"
|
||||
output_stream: "IMAGE:face_region"
|
||||
options: {
|
||||
[mediapipe.ImageCroppingCalculatorOptions.ext] {
|
||||
border_mode: BORDER_REPLICATE
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Transforms the input image on CPU to a 192x192 image. To scale the input
|
||||
# image, the scale_mode option is set to FIT to preserve the aspect ratio,
|
||||
# resulting in potential letterboxing in the transformed image.
|
||||
node: {
|
||||
calculator: "ImageTransformationCalculator"
|
||||
input_stream: "IMAGE:face_region"
|
||||
output_stream: "IMAGE:transformed_face_region"
|
||||
options: {
|
||||
[mediapipe.ImageTransformationCalculatorOptions.ext] {
|
||||
output_width: 192
|
||||
output_height: 192
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Converts the transformed input image on CPU into an image tensor stored as a
|
||||
# TfLiteTensor.
|
||||
node {
|
||||
calculator: "TfLiteConverterCalculator"
|
||||
input_stream: "IMAGE:transformed_face_region"
|
||||
output_stream: "TENSORS:input_tensor"
|
||||
}
|
||||
|
||||
# Runs a TensorFlow Lite model on CPU that takes an image tensor and outputs a
|
||||
# vector of tensors representing, for instance, detection boxes/keypoints and
|
||||
# scores.
|
||||
node {
|
||||
calculator: "TfLiteInferenceCalculator"
|
||||
input_stream: "TENSORS:input_tensor"
|
||||
output_stream: "TENSORS:output_tensors"
|
||||
options: {
|
||||
[mediapipe.TfLiteInferenceCalculatorOptions.ext] {
|
||||
model_path: "mediapipe/modules/face_landmark/face_landmark.tflite"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Splits a vector of tensors into multiple vectors.
|
||||
node {
|
||||
calculator: "SplitTfLiteTensorVectorCalculator"
|
||||
input_stream: "output_tensors"
|
||||
output_stream: "landmark_tensors"
|
||||
output_stream: "face_flag_tensor"
|
||||
options: {
|
||||
[mediapipe.SplitVectorCalculatorOptions.ext] {
|
||||
ranges: { begin: 0 end: 1 }
|
||||
ranges: { begin: 1 end: 2 }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Converts the face-flag tensor into a float that represents the confidence
|
||||
# score of face presence.
|
||||
node {
|
||||
calculator: "TfLiteTensorsToFloatsCalculator"
|
||||
input_stream: "TENSORS:face_flag_tensor"
|
||||
output_stream: "FLOAT:face_presence_score"
|
||||
}
|
||||
|
||||
# Applies a threshold to the confidence score to determine whether a face is
|
||||
# present.
|
||||
node {
|
||||
calculator: "ThresholdingCalculator"
|
||||
input_stream: "FLOAT:face_presence_score"
|
||||
output_stream: "FLAG:face_presence"
|
||||
options: {
|
||||
[mediapipe.ThresholdingCalculatorOptions.ext] {
|
||||
threshold: 0.1
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Drop landmarks tensors if face is not present.
|
||||
node {
|
||||
calculator: "GateCalculator"
|
||||
input_stream: "landmark_tensors"
|
||||
input_stream: "ALLOW:face_presence"
|
||||
output_stream: "ensured_landmark_tensors"
|
||||
}
|
||||
|
||||
# Decodes the landmark tensors into a vector of lanmarks, where the landmark
|
||||
# coordinates are normalized by the size of the input image to the model.
|
||||
node {
|
||||
calculator: "TfLiteTensorsToLandmarksCalculator"
|
||||
input_stream: "TENSORS:ensured_landmark_tensors"
|
||||
output_stream: "NORM_LANDMARKS:landmarks"
|
||||
options: {
|
||||
[mediapipe.TfLiteTensorsToLandmarksCalculatorOptions.ext] {
|
||||
num_landmarks: 468
|
||||
input_image_width: 192
|
||||
input_image_height: 192
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Projects the landmarks from the cropped face image to the corresponding
|
||||
# locations on the full image before cropping (input to the graph).
|
||||
node {
|
||||
calculator: "LandmarkProjectionCalculator"
|
||||
input_stream: "NORM_LANDMARKS:landmarks"
|
||||
input_stream: "NORM_RECT:roi"
|
||||
output_stream: "NORM_LANDMARKS:face_landmarks"
|
||||
}
|
||||
@@ -0,0 +1,216 @@
|
||||
# MediaPipe graph to detect/predict face landmarks. (CPU input, and inference is
|
||||
# executed on CPU.) This graph tries to skip face detection as much as possible
|
||||
# by using previously detected/predicted landmarks for new images.
|
||||
#
|
||||
# It is required that "face_detection_front.tflite" is available at
|
||||
# "mediapipe/modules/face_detection/face_detection_front.tflite"
|
||||
# path during execution.
|
||||
#
|
||||
# It is required that "face_landmark.tflite" is available at
|
||||
# "mediapipe/modules/face_landmark/face_landmark.tflite"
|
||||
# path during execution.
|
||||
#
|
||||
# EXAMPLE:
|
||||
# node {
|
||||
# calculator: "FaceLandmarkFrontCpu"
|
||||
# input_stream: "IMAGE:image"
|
||||
# input_side_packet: "NUM_FACES:num_faces"
|
||||
# output_stream: "LANDMARKS:multi_face_landmarks"
|
||||
# }
|
||||
|
||||
type: "FaceLandmarkFrontCpu"
|
||||
|
||||
# CPU image. (ImageFrame)
|
||||
input_stream: "IMAGE:image"
|
||||
|
||||
# Max number of faces to detect/track. (int)
|
||||
input_side_packet: "NUM_FACES:num_faces"
|
||||
|
||||
# Collection of detected/predicted faces, each represented as a list of 468 face
|
||||
# landmarks. (std::vector<NormalizedLandmarkList>)
|
||||
# NOTE: there will not be an output packet in the LANDMARKS stream for this
|
||||
# particular timestamp if none of faces detected. However, the MediaPipe
|
||||
# framework will internally inform the downstream calculators of the absence of
|
||||
# this packet so that they don't wait for it unnecessarily.
|
||||
output_stream: "LANDMARKS:multi_face_landmarks"
|
||||
|
||||
# Extra outputs (for debugging, for instance).
|
||||
# Detected faces. (std::vector<Detection>)
|
||||
output_stream: "DETECTIONS:face_detections"
|
||||
# Regions of interest calculated based on landmarks.
|
||||
# (std::vector<NormalizedRect>)
|
||||
output_stream: "ROIS_FROM_LANDMARKS:face_rects_from_landmarks"
|
||||
# Regions of interest calculated based on face detections.
|
||||
# (std::vector<NormalizedRect>)
|
||||
output_stream: "ROIS_FROM_DETECTIONS:face_rects_from_detections"
|
||||
|
||||
# Determines if an input vector of NormalizedRect has a size greater than or
|
||||
# equal to the provided num_faces.
|
||||
node {
|
||||
calculator: "NormalizedRectVectorHasMinSizeCalculator"
|
||||
input_stream: "ITERABLE:prev_face_rects_from_landmarks"
|
||||
input_side_packet: "num_faces"
|
||||
output_stream: "prev_has_enough_faces"
|
||||
}
|
||||
|
||||
# Drops the incoming image if FaceLandmarkCpu was able to identify face presence
|
||||
# in the previous image. Otherwise, passes the incoming image through to trigger
|
||||
# a new round of face detection in FaceDetectionFrontCpu.
|
||||
node {
|
||||
calculator: "GateCalculator"
|
||||
input_stream: "image"
|
||||
input_stream: "DISALLOW:prev_has_enough_faces"
|
||||
output_stream: "gated_image"
|
||||
options: {
|
||||
[mediapipe.GateCalculatorOptions.ext] {
|
||||
empty_packets_as_allow: true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Detects faces.
|
||||
node {
|
||||
calculator: "FaceDetectionFrontCpu"
|
||||
input_stream: "IMAGE:gated_image"
|
||||
output_stream: "DETECTIONS:all_face_detections"
|
||||
}
|
||||
|
||||
# Makes sure there are no more detections than the provided num_faces.
|
||||
node {
|
||||
calculator: "ClipDetectionVectorSizeCalculator"
|
||||
input_stream: "all_face_detections"
|
||||
output_stream: "face_detections"
|
||||
input_side_packet: "num_faces"
|
||||
}
|
||||
|
||||
# Calculate size of the image.
|
||||
node {
|
||||
calculator: "ImagePropertiesCalculator"
|
||||
input_stream: "IMAGE:gated_image"
|
||||
output_stream: "SIZE:gated_image_size"
|
||||
}
|
||||
|
||||
# Outputs each element of face_detections at a fake timestamp for the rest of
|
||||
# the graph to process. Clones the image size packet for each face_detection at
|
||||
# the fake timestamp. At the end of the loop, outputs the BATCH_END timestamp
|
||||
# for downstream calculators to inform them that all elements in the vector have
|
||||
# been processed.
|
||||
node {
|
||||
calculator: "BeginLoopDetectionCalculator"
|
||||
input_stream: "ITERABLE:face_detections"
|
||||
input_stream: "CLONE:gated_image_size"
|
||||
output_stream: "ITEM:face_detection"
|
||||
output_stream: "CLONE:detections_loop_image_size"
|
||||
output_stream: "BATCH_END:detections_loop_end_timestamp"
|
||||
}
|
||||
|
||||
# Calculates region of interest based on face detections, so that can be used
|
||||
# to detect landmarks.
|
||||
node {
|
||||
calculator: "FaceDetectionFrontDetectionToRoi"
|
||||
input_stream: "DETECTION:face_detection"
|
||||
input_stream: "IMAGE_SIZE:detections_loop_image_size"
|
||||
output_stream: "ROI:face_rect_from_detection"
|
||||
}
|
||||
|
||||
# Collects a NormalizedRect for each face into a vector. Upon receiving the
|
||||
# BATCH_END timestamp, outputs the vector of NormalizedRect at the BATCH_END
|
||||
# timestamp.
|
||||
node {
|
||||
calculator: "EndLoopNormalizedRectCalculator"
|
||||
input_stream: "ITEM:face_rect_from_detection"
|
||||
input_stream: "BATCH_END:detections_loop_end_timestamp"
|
||||
output_stream: "ITERABLE:face_rects_from_detections"
|
||||
}
|
||||
|
||||
# Performs association between NormalizedRect vector elements from previous
|
||||
# image and rects based on face detections from the current image. This
|
||||
# calculator ensures that the output face_rects vector doesn't contain
|
||||
# overlapping regions based on the specified min_similarity_threshold.
|
||||
node {
|
||||
calculator: "AssociationNormRectCalculator"
|
||||
input_stream: "prev_face_rects_from_landmarks"
|
||||
input_stream: "face_rects_from_detections"
|
||||
output_stream: "face_rects"
|
||||
options: {
|
||||
[mediapipe.AssociationCalculatorOptions.ext] {
|
||||
min_similarity_threshold: 0.5
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Calculate size of the image.
|
||||
node {
|
||||
calculator: "ImagePropertiesCalculator"
|
||||
input_stream: "IMAGE:image"
|
||||
output_stream: "SIZE:image_size"
|
||||
}
|
||||
|
||||
# Outputs each element of face_rects at a fake timestamp for the rest of the
|
||||
# graph to process. Clones image and image size packets for each
|
||||
# single_face_rect at the fake timestamp. At the end of the loop, outputs the
|
||||
# BATCH_END timestamp for downstream calculators to inform them that all
|
||||
# elements in the vector have been processed.
|
||||
node {
|
||||
calculator: "BeginLoopNormalizedRectCalculator"
|
||||
input_stream: "ITERABLE:face_rects"
|
||||
input_stream: "CLONE:0:image"
|
||||
input_stream: "CLONE:1:image_size"
|
||||
output_stream: "ITEM:face_rect"
|
||||
output_stream: "CLONE:0:landmarks_loop_image"
|
||||
output_stream: "CLONE:1:landmarks_loop_image_size"
|
||||
output_stream: "BATCH_END:landmarks_loop_end_timestamp"
|
||||
}
|
||||
|
||||
# Detects face landmarks within specified region of interest of the image.
|
||||
node {
|
||||
calculator: "FaceLandmarkCpu"
|
||||
input_stream: "IMAGE:landmarks_loop_image"
|
||||
input_stream: "ROI:face_rect"
|
||||
output_stream: "LANDMARKS:face_landmarks"
|
||||
}
|
||||
|
||||
# Calculates region of interest based on face landmarks, so that can be reused
|
||||
# for subsequent image.
|
||||
node {
|
||||
calculator: "FaceLandmarkLandmarksToRoi"
|
||||
input_stream: "LANDMARKS:face_landmarks"
|
||||
input_stream: "IMAGE_SIZE:landmarks_loop_image_size"
|
||||
output_stream: "ROI:face_rect_from_landmarks"
|
||||
}
|
||||
|
||||
# Collects a set of landmarks for each face into a vector. Upon receiving the
|
||||
# BATCH_END timestamp, outputs the vector of landmarks at the BATCH_END
|
||||
# timestamp.
|
||||
node {
|
||||
calculator: "EndLoopNormalizedLandmarkListVectorCalculator"
|
||||
input_stream: "ITEM:face_landmarks"
|
||||
input_stream: "BATCH_END:landmarks_loop_end_timestamp"
|
||||
output_stream: "ITERABLE:multi_face_landmarks"
|
||||
}
|
||||
|
||||
# Collects a NormalizedRect for each face into a vector. Upon receiving the
|
||||
# BATCH_END timestamp, outputs the vector of NormalizedRect at the BATCH_END
|
||||
# timestamp.
|
||||
node {
|
||||
calculator: "EndLoopNormalizedRectCalculator"
|
||||
input_stream: "ITEM:face_rect_from_landmarks"
|
||||
input_stream: "BATCH_END:landmarks_loop_end_timestamp"
|
||||
output_stream: "ITERABLE:face_rects_from_landmarks"
|
||||
}
|
||||
|
||||
# Caches face rects calculated from landmarks, and upon the arrival of the next
|
||||
# input image, sends out the cached rects with timestamps replaced by that of
|
||||
# the input image, essentially generating a packet that carries the previous
|
||||
# face rects. Note that upon the arrival of the very first input image, a
|
||||
# timestamp bound update occurs to jump start the feedback loop.
|
||||
node {
|
||||
calculator: "PreviousLoopbackCalculator"
|
||||
input_stream: "MAIN:image"
|
||||
input_stream: "LOOP:face_rects_from_landmarks"
|
||||
input_stream_info: {
|
||||
tag_index: "LOOP"
|
||||
back_edge: true
|
||||
}
|
||||
output_stream: "PREV_LOOP:prev_face_rects_from_landmarks"
|
||||
}
|
||||
@@ -0,0 +1,216 @@
|
||||
# MediaPipe graph to detect/predict face landmarks. (GPU input, and inference is
|
||||
# executed on GPU.) This graph tries to skip face detection as much as possible
|
||||
# by using previously detected/predicted landmarks for new images.
|
||||
#
|
||||
# It is required that "face_detection_front.tflite" is available at
|
||||
# "mediapipe/modules/face_detection/face_detection_front.tflite"
|
||||
# path during execution.
|
||||
#
|
||||
# It is required that "face_landmark.tflite" is available at
|
||||
# "mediapipe/modules/face_landmark/face_landmark.tflite"
|
||||
# path during execution.
|
||||
#
|
||||
# EXAMPLE:
|
||||
# node {
|
||||
# calculator: "FaceLandmarkFrontGpu"
|
||||
# input_stream: "IMAGE:image"
|
||||
# input_side_packet: "NUM_FACES:num_faces"
|
||||
# output_stream: "LANDMARKS:multi_face_landmarks"
|
||||
# }
|
||||
|
||||
type: "FaceLandmarkFrontGpu"
|
||||
|
||||
# GPU image. (GpuBuffer)
|
||||
input_stream: "IMAGE:image"
|
||||
|
||||
# Max number of faces to detect/track. (int)
|
||||
input_side_packet: "NUM_FACES:num_faces"
|
||||
|
||||
# Collection of detected/predicted faces, each represented as a list of 468 face
|
||||
# landmarks. (std::vector<NormalizedLandmarkList>)
|
||||
# NOTE: there will not be an output packet in the LANDMARKS stream for this
|
||||
# particular timestamp if none of faces detected. However, the MediaPipe
|
||||
# framework will internally inform the downstream calculators of the absence of
|
||||
# this packet so that they don't wait for it unnecessarily.
|
||||
output_stream: "LANDMARKS:multi_face_landmarks"
|
||||
|
||||
# Extra outputs (for debugging, for instance).
|
||||
# Detected faces. (std::vector<Detection>)
|
||||
output_stream: "DETECTIONS:face_detections"
|
||||
# Regions of interest calculated based on landmarks.
|
||||
# (std::vector<NormalizedRect>)
|
||||
output_stream: "ROIS_FROM_LANDMARKS:face_rects_from_landmarks"
|
||||
# Regions of interest calculated based on face detections.
|
||||
# (std::vector<NormalizedRect>)
|
||||
output_stream: "ROIS_FROM_DETECTIONS:face_rects_from_detections"
|
||||
|
||||
# Determines if an input vector of NormalizedRect has a size greater than or
|
||||
# equal to the provided num_faces.
|
||||
node {
|
||||
calculator: "NormalizedRectVectorHasMinSizeCalculator"
|
||||
input_stream: "ITERABLE:prev_face_rects_from_landmarks"
|
||||
input_side_packet: "num_faces"
|
||||
output_stream: "prev_has_enough_faces"
|
||||
}
|
||||
|
||||
# Drops the incoming image if FaceLandmarkGpu was able to identify face presence
|
||||
# in the previous image. Otherwise, passes the incoming image through to trigger
|
||||
# a new round of face detection in FaceDetectionFrontGpu.
|
||||
node {
|
||||
calculator: "GateCalculator"
|
||||
input_stream: "image"
|
||||
input_stream: "DISALLOW:prev_has_enough_faces"
|
||||
output_stream: "gated_image"
|
||||
options: {
|
||||
[mediapipe.GateCalculatorOptions.ext] {
|
||||
empty_packets_as_allow: true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Detects faces.
|
||||
node {
|
||||
calculator: "FaceDetectionFrontGpu"
|
||||
input_stream: "IMAGE:gated_image"
|
||||
output_stream: "DETECTIONS:all_face_detections"
|
||||
}
|
||||
|
||||
# Makes sure there are no more detections than the provided num_faces.
|
||||
node {
|
||||
calculator: "ClipDetectionVectorSizeCalculator"
|
||||
input_stream: "all_face_detections"
|
||||
output_stream: "face_detections"
|
||||
input_side_packet: "num_faces"
|
||||
}
|
||||
|
||||
# Calculate size of the image.
|
||||
node {
|
||||
calculator: "ImagePropertiesCalculator"
|
||||
input_stream: "IMAGE_GPU:gated_image"
|
||||
output_stream: "SIZE:gated_image_size"
|
||||
}
|
||||
|
||||
# Outputs each element of face_detections at a fake timestamp for the rest of
|
||||
# the graph to process. Clones the image size packet for each face_detection at
|
||||
# the fake timestamp. At the end of the loop, outputs the BATCH_END timestamp
|
||||
# for downstream calculators to inform them that all elements in the vector have
|
||||
# been processed.
|
||||
node {
|
||||
calculator: "BeginLoopDetectionCalculator"
|
||||
input_stream: "ITERABLE:face_detections"
|
||||
input_stream: "CLONE:gated_image_size"
|
||||
output_stream: "ITEM:face_detection"
|
||||
output_stream: "CLONE:detections_loop_image_size"
|
||||
output_stream: "BATCH_END:detections_loop_end_timestamp"
|
||||
}
|
||||
|
||||
# Calculates region of interest based on face detections, so that can be used
|
||||
# to detect landmarks.
|
||||
node {
|
||||
calculator: "FaceDetectionFrontDetectionToRoi"
|
||||
input_stream: "DETECTION:face_detection"
|
||||
input_stream: "IMAGE_SIZE:detections_loop_image_size"
|
||||
output_stream: "ROI:face_rect_from_detection"
|
||||
}
|
||||
|
||||
# Collects a NormalizedRect for each face into a vector. Upon receiving the
|
||||
# BATCH_END timestamp, outputs the vector of NormalizedRect at the BATCH_END
|
||||
# timestamp.
|
||||
node {
|
||||
calculator: "EndLoopNormalizedRectCalculator"
|
||||
input_stream: "ITEM:face_rect_from_detection"
|
||||
input_stream: "BATCH_END:detections_loop_end_timestamp"
|
||||
output_stream: "ITERABLE:face_rects_from_detections"
|
||||
}
|
||||
|
||||
# Performs association between NormalizedRect vector elements from previous
|
||||
# image and rects based on face detections from the current image. This
|
||||
# calculator ensures that the output face_rects vector doesn't contain
|
||||
# overlapping regions based on the specified min_similarity_threshold.
|
||||
node {
|
||||
calculator: "AssociationNormRectCalculator"
|
||||
input_stream: "prev_face_rects_from_landmarks"
|
||||
input_stream: "face_rects_from_detections"
|
||||
output_stream: "face_rects"
|
||||
options: {
|
||||
[mediapipe.AssociationCalculatorOptions.ext] {
|
||||
min_similarity_threshold: 0.5
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Calculate size of the image.
|
||||
node {
|
||||
calculator: "ImagePropertiesCalculator"
|
||||
input_stream: "IMAGE_GPU:image"
|
||||
output_stream: "SIZE:image_size"
|
||||
}
|
||||
|
||||
# Outputs each element of face_rects at a fake timestamp for the rest of the
|
||||
# graph to process. Clones image and image size packets for each
|
||||
# single_face_rect at the fake timestamp. At the end of the loop, outputs the
|
||||
# BATCH_END timestamp for downstream calculators to inform them that all
|
||||
# elements in the vector have been processed.
|
||||
node {
|
||||
calculator: "BeginLoopNormalizedRectCalculator"
|
||||
input_stream: "ITERABLE:face_rects"
|
||||
input_stream: "CLONE:0:image"
|
||||
input_stream: "CLONE:1:image_size"
|
||||
output_stream: "ITEM:face_rect"
|
||||
output_stream: "CLONE:0:landmarks_loop_image"
|
||||
output_stream: "CLONE:1:landmarks_loop_image_size"
|
||||
output_stream: "BATCH_END:landmarks_loop_end_timestamp"
|
||||
}
|
||||
|
||||
# Detects face landmarks within specified region of interest of the image.
|
||||
node {
|
||||
calculator: "FaceLandmarkGpu"
|
||||
input_stream: "IMAGE:landmarks_loop_image"
|
||||
input_stream: "ROI:face_rect"
|
||||
output_stream: "LANDMARKS:face_landmarks"
|
||||
}
|
||||
|
||||
# Calculates region of interest based on face landmarks, so that can be reused
|
||||
# for subsequent image.
|
||||
node {
|
||||
calculator: "FaceLandmarkLandmarksToRoi"
|
||||
input_stream: "LANDMARKS:face_landmarks"
|
||||
input_stream: "IMAGE_SIZE:landmarks_loop_image_size"
|
||||
output_stream: "ROI:face_rect_from_landmarks"
|
||||
}
|
||||
|
||||
# Collects a set of landmarks for each face into a vector. Upon receiving the
|
||||
# BATCH_END timestamp, outputs the vector of landmarks at the BATCH_END
|
||||
# timestamp.
|
||||
node {
|
||||
calculator: "EndLoopNormalizedLandmarkListVectorCalculator"
|
||||
input_stream: "ITEM:face_landmarks"
|
||||
input_stream: "BATCH_END:landmarks_loop_end_timestamp"
|
||||
output_stream: "ITERABLE:multi_face_landmarks"
|
||||
}
|
||||
|
||||
# Collects a NormalizedRect for each face into a vector. Upon receiving the
|
||||
# BATCH_END timestamp, outputs the vector of NormalizedRect at the BATCH_END
|
||||
# timestamp.
|
||||
node {
|
||||
calculator: "EndLoopNormalizedRectCalculator"
|
||||
input_stream: "ITEM:face_rect_from_landmarks"
|
||||
input_stream: "BATCH_END:landmarks_loop_end_timestamp"
|
||||
output_stream: "ITERABLE:face_rects_from_landmarks"
|
||||
}
|
||||
|
||||
# Caches face rects calculated from landmarks, and upon the arrival of the next
|
||||
# input image, sends out the cached rects with timestamps replaced by that of
|
||||
# the input image, essentially generating a packet that carries the previous
|
||||
# face rects. Note that upon the arrival of the very first input image, a
|
||||
# timestamp bound update occurs to jump start the feedback loop.
|
||||
node {
|
||||
calculator: "PreviousLoopbackCalculator"
|
||||
input_stream: "MAIN:image"
|
||||
input_stream: "LOOP:face_rects_from_landmarks"
|
||||
input_stream_info: {
|
||||
tag_index: "LOOP"
|
||||
back_edge: true
|
||||
}
|
||||
output_stream: "PREV_LOOP:prev_face_rects_from_landmarks"
|
||||
}
|
||||
@@ -0,0 +1,146 @@
|
||||
# MediaPipe graph to detect/predict face landmarks. (GPU input, and inference is
|
||||
# executed on GPU.)
|
||||
#
|
||||
# It is required that "face_landmark.tflite" is available at
|
||||
# "mediapipe/modules/face_landmark/face_landmark.tflite"
|
||||
# path during execution.
|
||||
#
|
||||
# EXAMPLE:
|
||||
# node {
|
||||
# calculator: "FaceLandmarkGpu"
|
||||
# input_stream: "IMAGE:image"
|
||||
# input_stream: "ROI:face_roi"
|
||||
# output_stream: "LANDMARKS:face_landmarks"
|
||||
# }
|
||||
|
||||
type: "FaceLandmarkGpu"
|
||||
|
||||
# GPU image. (GpuBuffer)
|
||||
input_stream: "IMAGE:image"
|
||||
# ROI (region of interest) within the given image where a face is located.
|
||||
# (NormalizedRect)
|
||||
input_stream: "ROI:roi"
|
||||
|
||||
# 468 face landmarks within the given ROI. (NormalizedLandmarkList)
|
||||
# NOTE: if a face is not present within the given ROI, for this particular
|
||||
# timestamp there will not be an output packet in the LANDMARKS stream. However,
|
||||
# the MediaPipe framework will internally inform the downstream calculators of
|
||||
# the absence of this packet so that they don't wait for it unnecessarily.
|
||||
output_stream: "LANDMARKS:face_landmarks"
|
||||
|
||||
# Crops the input image to the given region of interest.
|
||||
node {
|
||||
calculator: "ImageCroppingCalculator"
|
||||
input_stream: "IMAGE_GPU:image"
|
||||
input_stream: "NORM_RECT:roi"
|
||||
output_stream: "IMAGE_GPU:face_region"
|
||||
options: {
|
||||
[mediapipe.ImageCroppingCalculatorOptions.ext] {
|
||||
border_mode: BORDER_REPLICATE
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Transforms the input image on GPU to a 192x192 image. To scale the input
|
||||
# image, the scale_mode option is set to FIT to preserve the aspect ratio,
|
||||
# resulting in potential letterboxing in the transformed image.
|
||||
node: {
|
||||
calculator: "ImageTransformationCalculator"
|
||||
input_stream: "IMAGE_GPU:face_region"
|
||||
output_stream: "IMAGE_GPU:transformed_face_region"
|
||||
options: {
|
||||
[mediapipe.ImageTransformationCalculatorOptions.ext] {
|
||||
output_width: 192
|
||||
output_height: 192
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Converts the transformed input image on GPU into an image tensor stored as a
|
||||
# TfLiteTensor.
|
||||
node {
|
||||
calculator: "TfLiteConverterCalculator"
|
||||
input_stream: "IMAGE_GPU:transformed_face_region"
|
||||
output_stream: "TENSORS_GPU:input_tensor"
|
||||
}
|
||||
|
||||
# Runs a TensorFlow Lite model on GPU that takes an image tensor and outputs a
|
||||
# vector of GPU tensors representing, for instance, detection boxes/keypoints
|
||||
# and scores.
|
||||
node {
|
||||
calculator: "TfLiteInferenceCalculator"
|
||||
input_stream: "TENSORS_GPU:input_tensor"
|
||||
output_stream: "TENSORS:output_tensors"
|
||||
options: {
|
||||
[mediapipe.TfLiteInferenceCalculatorOptions.ext] {
|
||||
model_path: "mediapipe/modules/face_landmark/face_landmark.tflite"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Splits a vector of tensors into multiple vectors.
|
||||
node {
|
||||
calculator: "SplitTfLiteTensorVectorCalculator"
|
||||
input_stream: "output_tensors"
|
||||
output_stream: "landmark_tensors"
|
||||
output_stream: "face_flag_tensor"
|
||||
options: {
|
||||
[mediapipe.SplitVectorCalculatorOptions.ext] {
|
||||
ranges: { begin: 0 end: 1 }
|
||||
ranges: { begin: 1 end: 2 }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Converts the face-flag tensor into a float that represents the confidence
|
||||
# score of face presence.
|
||||
node {
|
||||
calculator: "TfLiteTensorsToFloatsCalculator"
|
||||
input_stream: "TENSORS:face_flag_tensor"
|
||||
output_stream: "FLOAT:face_presence_score"
|
||||
}
|
||||
|
||||
# Applies a threshold to the confidence score to determine whether a face is
|
||||
# present.
|
||||
node {
|
||||
calculator: "ThresholdingCalculator"
|
||||
input_stream: "FLOAT:face_presence_score"
|
||||
output_stream: "FLAG:face_presence"
|
||||
options: {
|
||||
[mediapipe.ThresholdingCalculatorOptions.ext] {
|
||||
threshold: 0.1
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Drop landmarks tensors if face is not present.
|
||||
node {
|
||||
calculator: "GateCalculator"
|
||||
input_stream: "landmark_tensors"
|
||||
input_stream: "ALLOW:face_presence"
|
||||
output_stream: "ensured_landmark_tensors"
|
||||
}
|
||||
|
||||
# Decodes the landmark tensors into a vector of lanmarks, where the landmark
|
||||
# coordinates are normalized by the size of the input image to the model.
|
||||
node {
|
||||
calculator: "TfLiteTensorsToLandmarksCalculator"
|
||||
input_stream: "TENSORS:ensured_landmark_tensors"
|
||||
output_stream: "NORM_LANDMARKS:landmarks"
|
||||
options: {
|
||||
[mediapipe.TfLiteTensorsToLandmarksCalculatorOptions.ext] {
|
||||
num_landmarks: 468
|
||||
input_image_width: 192
|
||||
input_image_height: 192
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Projects the landmarks from the cropped face image to the corresponding
|
||||
# locations on the full image before cropping (input to the graph).
|
||||
node {
|
||||
calculator: "LandmarkProjectionCalculator"
|
||||
input_stream: "NORM_LANDMARKS:landmarks"
|
||||
input_stream: "NORM_RECT:roi"
|
||||
output_stream: "NORM_LANDMARKS:face_landmarks"
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
# MediaPipe graph to calculate face region of interest (ROI) from landmarks
|
||||
# detected by "FaceLandmarkCpu" or "FaceLandmarkGpu".
|
||||
#
|
||||
# NOTE: this graph is subject to change and should not be used directly.
|
||||
|
||||
type: "FaceLandmarkLandmarksToRoi"
|
||||
|
||||
# Normalized landmarks. (NormalizedLandmarkList)
|
||||
input_stream: "LANDMARKS:landmarks"
|
||||
# Frame size (width & height). (std::pair<int, int>)
|
||||
input_stream: "IMAGE_SIZE:image_size"
|
||||
# ROI according to landmarks. (NormalizedRect)
|
||||
output_stream: "ROI:roi"
|
||||
|
||||
# Converts face landmarks to a detection that tightly encloses all landmarks.
|
||||
node {
|
||||
calculator: "LandmarksToDetectionCalculator"
|
||||
input_stream: "NORM_LANDMARKS:landmarks"
|
||||
output_stream: "DETECTION:face_detection"
|
||||
}
|
||||
|
||||
# Converts the face detection into a rectangle (normalized by image size)
|
||||
# that encloses the face and is rotated such that the line connecting left side
|
||||
# of the left eye and right side of the right eye is aligned with the X-axis of
|
||||
# the rectangle.
|
||||
node {
|
||||
calculator: "DetectionsToRectsCalculator"
|
||||
input_stream: "DETECTION:face_detection"
|
||||
input_stream: "IMAGE_SIZE:image_size"
|
||||
output_stream: "NORM_RECT:face_rect_from_landmarks"
|
||||
options: {
|
||||
[mediapipe.DetectionsToRectsCalculatorOptions.ext] {
|
||||
rotation_vector_start_keypoint_index: 33 # Left side of left eye.
|
||||
rotation_vector_end_keypoint_index: 133 # Right side of right eye.
|
||||
rotation_vector_target_angle_degrees: 0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Expands the face rectangle so that in the next video image it's likely to
|
||||
# still contain the face even with some motion.
|
||||
node {
|
||||
calculator: "RectTransformationCalculator"
|
||||
input_stream: "NORM_RECT:face_rect_from_landmarks"
|
||||
input_stream: "IMAGE_SIZE:image_size"
|
||||
output_stream: "roi"
|
||||
options: {
|
||||
[mediapipe.RectTransformationCalculatorOptions.ext] {
|
||||
scale_x: 1.5
|
||||
scale_y: 1.5
|
||||
square_long: true
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user