Source
stdlib/parsers/url/url.hpp
1
// Copyright (c) 2026 BigBrain LLC. MIT-licensed (see LICENSE).2
// Original work; see ACKNOWLEDGMENTS.md for the open-source ideas we build upon.3
#pragma once5
// cheatah::parsers::url — a from-scratch parser for the http(s) URL subset the requests module6
// speaks: `scheme://host[:port][/path][?query]`. No allocation beyond the component strings, no7
// regex, no dependencies. Userinfo (`user@`) and fragments (`#...`) are not supported — the first8
// is an obsolete security hazard in http URLs, the second is never sent to the server anyway.9
//10
// Laid out as a cheatah stdlib module: from .purr this is `import parsers.url.Parser as Parser`,11
// mirroring `import parsers.json.Parser as Parser` — each parsers submodule exposes a Parser.13
#include <string>14
#include <string_view>16
namespace cheatah::parsers::url {18
/**19
* @brief One parsed http(s) URL. @c target is the HTTP request-target — the path plus the original20
* query, always beginning with '/' (an empty path becomes "/").21
*/22
struct Url {23
std::string scheme; ///< the lowercased scheme: "http" or "https".24
std::string host; ///< the host (name or IP); never empty on success.25
long long port = 0; ///< the explicit port, or the scheme default (80 for http, 443 for https).26
std::string target; ///< the HTTP request-target "/path?query" (always begins with '/').27
};29
/**30
* @brief The URL parser. Stateless and reusable; a class (not a free function) so the module31
* surface is symmetric with parsers::json::Parser and imports the same way from cheatah.32
*/33
class Parser {34
public:35
/**36
* Parse @p text into @p out. Accepts `scheme://host[:port][/path][?query]` with scheme http or37
* https (case-insensitive). Rejects empty hosts, non-numeric or out-of-range ports, userinfo,38
* and fragments. On failure @p out is left unspecified.39
*40
* @param text the URL text to parse.41
* @param out receives the parsed components on success.42
* @return true iff @p text is a valid accepted URL.43
* @complexity O(|text|)44
* @alloc the component strings in @p out45
* @test CheatahParsers.UrlParserComponents46
* @test CheatahParsers.UrlParserRejects47
* @crtest ParsersCompileRun.UrlParserImport48
*/49
[[nodiscard]] bool parse(std::string_view text, Url& out) const { // NOLINT(readability-convert-member-functions-to-static): callers hold a Parser instance (the .purr API shape)50
const std::size_t scheme_end = text.find("://");51
if (scheme_end == std::string_view::npos || scheme_end == 0) {52
return false;53
}54
out.scheme.clear();55
for (const char ch : text.substr(0, scheme_end)) { // lowercase the scheme as we copy56
out.scheme.push_back(ch >= 'A' && ch <= 'Z' ? static_cast<char>(ch - 'A' + 'a') : ch);57
}58
if (out.scheme != "http" && out.scheme != "https") {59
return false;60
}62
std::string_view rest = text.substr(scheme_end + 3);63
const std::size_t path_start = rest.find('/');64
const std::size_t query_start = rest.find('?');65
const std::size_t authority_end = std::min(path_start, query_start);66
const std::string_view authority = rest.substr(0, authority_end);67
if (authority.empty() || authority.find('@') != std::string_view::npos ||68
rest.find('#') != std::string_view::npos) {69
return false; // empty host, userinfo, and fragments are all rejected70
}72
const std::size_t colon = authority.rfind(':');73
if (colon == std::string_view::npos) {74
out.host = std::string(authority);75
out.port = (out.scheme == "https") ? 443 : 80;76
} else {77
out.host = std::string(authority.substr(0, colon));78
const std::string_view digits = authority.substr(colon + 1);79
if (out.host.empty() || digits.empty() || digits.size() > 5) {80
return false;81
}82
long long port = 0;83
for (const char ch : digits) {84
if (ch < '0' || ch > '9') {85
return false;86
}87
port = port * 10 + (ch - '0');88
}89
if (port < 1 || port > 65535) {90
return false;91
}92
out.port = port;93
}95
// The request-target: everything from the first '/' on (or "/" when the path is absent,96
// including the bare-query form "host?x=1" -> "/?x=1").97
if (path_start != std::string_view::npos) {98
out.target = std::string(rest.substr(path_start));99
} else if (query_start != std::string_view::npos) {100
out.target = "/" + std::string(rest.substr(query_start));101
} else {102
out.target = "/";103
}104
return true;105
}106
};108
} // namespace cheatah::parsers::url