From 02c2561be4d553bcc55104a2618ef503c0535783 Mon Sep 17 00:00:00 2001 From: Alex Hultman Date: Sat, 10 Oct 2020 17:54:32 +0200 Subject: [PATCH] Add initial multipart parser --- src/MessageParser.h | 64 ++++++++++++++++++++ src/Multipart.h | 142 ++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 206 insertions(+) create mode 100644 src/MessageParser.h create mode 100644 src/Multipart.h diff --git a/src/MessageParser.h b/src/MessageParser.h new file mode 100644 index 0000000..aa8d455 --- /dev/null +++ b/src/MessageParser.h @@ -0,0 +1,64 @@ +/* + * Authored by Alex Hultman, 2018-2020. + * Intellectual property of third-party. + + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + + * http://www.apache.org/licenses/LICENSE-2.0 + + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* Implements the common parser (RFC 822) used in both HTTP and Multipart parsing */ + +#ifndef UWS_MESSAGE_PARSER_H +#define UWS_MESSAGE_PARSER_H + +#include +#include +#include + +/* For now we have this one here */ +#define MAX_HEADERS 10 + +namespace uWS { + + // should be templated on whether it needs at lest one header (http), or not (multipart) + static inline unsigned int getHeaders(char *postPaddedBuffer, char *end, std::pair *headers) { + char *preliminaryKey, *preliminaryValue, *start = postPaddedBuffer; + + for (unsigned int i = 0; i < MAX_HEADERS; i++) { + for (preliminaryKey = postPaddedBuffer; (*postPaddedBuffer != ':') & (*postPaddedBuffer > 32); *(postPaddedBuffer++) |= 32); + if (*postPaddedBuffer == '\r') { + if ((postPaddedBuffer != end) & (postPaddedBuffer[1] == '\n') /* & (i > 0) */) { // multipart does not require any headers like http does + headers->first = std::string_view(nullptr, 0); + return (unsigned int) ((postPaddedBuffer + 2) - start); + } else { + return 0; + } + } else { + headers->first = std::string_view(preliminaryKey, (size_t) (postPaddedBuffer - preliminaryKey)); + for (postPaddedBuffer++; (*postPaddedBuffer == ':' || *postPaddedBuffer < 33) && *postPaddedBuffer != '\r'; postPaddedBuffer++); + preliminaryValue = postPaddedBuffer; + postPaddedBuffer = (char *) memchr(postPaddedBuffer, '\r', end - postPaddedBuffer); + if (postPaddedBuffer && postPaddedBuffer[1] == '\n') { + headers->second = std::string_view(preliminaryValue, (size_t) (postPaddedBuffer - preliminaryValue)); + postPaddedBuffer += 2; + headers++; + } else { + return 0; + } + } + } + return 0; + } + +} + +#endif \ No newline at end of file diff --git a/src/Multipart.h b/src/Multipart.h new file mode 100644 index 0000000..9135246 --- /dev/null +++ b/src/Multipart.h @@ -0,0 +1,142 @@ +/* + * Authored by Alex Hultman, 2018-2020. + * Intellectual property of third-party. + + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + + * http://www.apache.org/licenses/LICENSE-2.0 + + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* Implements the multipart protocol. Builds atop parts of our common http parser (not yet refactored that way). */ +/* https://www.w3.org/Protocols/rfc1341/7_2_Multipart.html */ + +#ifndef UWS_MULTIPART_H +#define UWS_MULTIPART_H + +#include "MessageParser.h" + +#include +#include +#include + +namespace uWS { + + struct MultipartParser { + + /* 2 chars of hyphen + 1 - 70 chars of boundary */ + char prependedBoundaryBuffer[72]; + std::string_view prependedBoundary; + std::string_view remainingBody; + bool first = true; + + /* I think it is more than sane to limit this to 10 per part */ + //static const int MAX_HEADERS = 10; + + /* Construct the parser based on contentType (reads boundary) */ + MultipartParser(std::string_view contentType) { + + /* We expect the form "multipart/something;somethingboundary=something" */ + if (!contentType.starts_with("multipart/")) { + return; + } + + /* For now we simply guess boundary will lie between = and end. This is not entirely + * standards compliant as boundary may be expressed with or without " and spaces */ + auto equalToken = contentType.find('=', 10); + if (equalToken != std::string_view::npos) { + + /* Boundary must be less than or equal to 70 chars yet 1 char or longer */ + std::string_view boundary = contentType.substr(equalToken + 1); + if (!boundary.length() || boundary.length() > 70) { + /* Invalid size */ + return; + } + + /* Prepend it with two hyphens */ + prependedBoundaryBuffer[0] = prependedBoundaryBuffer[1] = '-'; + memcpy(&prependedBoundaryBuffer[2], boundary.data(), boundary.length()); + + prependedBoundary = {prependedBoundaryBuffer, boundary.length() + 2}; + } + } + + /* Is this even a valid multipart request? */ + bool isValid() { + return prependedBoundary.length() != 0; + } + + /* Set the body once, before getting any parts */ + void setBody(std::string_view body) { + remainingBody = body; + } + + /* Parse out the next part's data, filling the headers. Returns empty part on end or error */ + std::string_view getNextPart(std::pair *headers) { + + /* The remaining two hyphens should be shorter than the boundary */ + if (remainingBody.length() < prependedBoundary.length()) { + /* We are done now */ + return {}; + } + + if (first) { + auto nextBoundary = remainingBody.find(prependedBoundary); + if (nextBoundary == std::string_view::npos) { + /* Cannot parse */ + return {}; + } + + /* Toss away boundary and anything before it */ + remainingBody.remove_prefix(nextBoundary + prependedBoundary.length()); + first = false; + } + + auto nextEndBoundary = remainingBody.find(prependedBoundary); + if (nextEndBoundary == std::string_view::npos) { + /* Cannot parse (or simply done) */ + return {}; + } + + std::string_view part = remainingBody.substr(0, nextEndBoundary); + remainingBody.remove_prefix(nextEndBoundary + prependedBoundary.length()); + + /* We are allowed to post pad like this because we know the boundary is at least 2 bytes */ + /* This makes parsing a second pass invalid, so you can only iterate over parts once */ + memset((char *) part.data() + part.length(), '\r', 1); + + /* For this to be a valid part, we need to consume at least 4 bytes (\r\n\r\n) */ + int consumed = getHeaders((char *) part.data(), (char *) part.data() + part.length(), headers); + + if (!consumed) { + /* This is an invalid part */ + return {}; + } + + /* This is to be done by the user, not the parser */ + /*for (int i = 0; i < MAX_HEADERS; i++) { + if (headers[i].first.length()) { + std::cout << "<" << headers[i].first << "> = <" << headers[i].second << ">" << std::endl; + } else { + break; + } + }*/ + + /* Strip away the headers from the part body data */ + part.remove_prefix(consumed); + + /* Now pass whatever is remaining of the part */ + return part; + } + }; + +} + +#endif \ No newline at end of file