Sourcemeta Core 0.0.0
Loading...
Searching...
No Matches
html_escape.h
1#ifndef SOURCEMETA_CORE_HTML_ESCAPE_H_
2#define SOURCEMETA_CORE_HTML_ESCAPE_H_
3
4#ifndef SOURCEMETA_CORE_HTML_EXPORT
5#include <sourcemeta/core/html_export.h>
6#endif
7
8#include <sourcemeta/core/html_buffer.h>
9#include <sourcemeta/core/preprocessor.h>
10
11#include <array> // std::array
12#include <concepts> // std::same_as
13#include <cstddef> // std::size_t
14#include <cstdint> // std::uint8_t, std::uint64_t
15#include <cstring> // std::memcpy
16#include <string> // std::string
17#include <string_view> // std::string_view
18
19namespace sourcemeta::core {
20
39SOURCEMETA_CORE_HTML_EXPORT
40auto html_escape(std::string &text) -> void;
41
55template <typename Output>
56 requires std::same_as<Output, std::string> || std::same_as<Output, HTMLBuffer>
57inline auto html_escape_append(Output &output, const std::string_view input)
58 -> void {
59 // The bytes that escaping may replace, which are the quotation mark, the
60 // ampersand, the apostrophe, the angle brackets, and the first byte of the
61 // UTF-8 encoding of the no-break space
62 static constexpr std::array<std::uint8_t, 256> SPECIAL_BYTES{{
63 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x00
64 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x10
65 0, 0, 1, 0, 0, 0, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, // 0x20
66 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 1, 0, // 0x30
67 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x40
68 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x50
69 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x60
70 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x70
71 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x80
72 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x90
73 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xA0
74 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xB0
75 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xC0
76 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xD0
77 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xE0
78 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0 // 0xF0
79 }};
80 constexpr std::uint64_t LOW_BITS{0x0101010101010101ULL};
81 constexpr std::uint64_t HIGH_BITS{0x8080808080808080ULL};
82
83 const auto size{input.size()};
84 // Whatever stays as it is goes out in runs, so that text with nothing to
85 // escape takes a single append
86 std::size_t run_start{0};
87 std::size_t position{0};
88 while (position < size) {
89 auto end{size};
90 if (size - position >= 8) {
91 std::uint64_t word{0};
92 std::memcpy(&word, input.data() + position, 8);
93 // A byte of a word equals another when subtracting from their difference
94 // borrows, which also flags the byte right after a genuine match, and
95 // that is harmless when only asking whether the word has a match at all.
96 // Setting the lowest bit of every byte makes the ampersand match the
97 // apostrophe, and setting the second lowest bit makes the less-than sign
98 // match the greater-than sign, so four comparisons cover the six bytes
99 const auto quotations{word ^ (LOW_BITS * '"')};
100 const auto apostrophes{(word | LOW_BITS) ^ (LOW_BITS * '\'')};
101 const auto angles{(word | (LOW_BITS * 2U)) ^ (LOW_BITS * '>')};
102 const auto leads{word ^ (LOW_BITS * 0xC2)};
103 const auto matches{((quotations - LOW_BITS) & ~quotations) |
104 ((apostrophes - LOW_BITS) & ~apostrophes) |
105 ((angles - LOW_BITS) & ~angles) |
106 ((leads - LOW_BITS) & ~leads)};
107 if ((matches & HIGH_BITS) == 0) {
108 position += 8;
109 continue;
110 }
111
112 end = position + 8;
113 }
114
115 while (position < end) {
116 if (SPECIAL_BYTES[static_cast<unsigned char>(input[position])] == 0) {
117 position += 1;
118 continue;
119 }
120
121 std::string_view replacement;
122 std::size_t consumed{1};
123 switch (input[position]) {
124 case '&':
125 replacement = "&amp;";
126 break;
127 case '<':
128 replacement = "&lt;";
129 break;
130 case '>':
131 replacement = "&gt;";
132 break;
133 case '"':
134 replacement = "&quot;";
135 break;
136 case '\'':
137 replacement = "&#39;";
138 break;
139 default:
140 // The no-break space is replaced by its named entity (HTML Living
141 // Standard "escaping a string" step 2)
142 if (static_cast<unsigned char>(input[position]) == 0xC2 &&
143 position + 1 < size &&
144 static_cast<unsigned char>(input[position + 1]) == 0xA0) {
145 replacement = "&nbsp;";
146 consumed = 2;
147 }
148
149 break;
150 }
151
152 if (replacement.empty()) {
153 position += 1;
154 continue;
155 }
156
157 output.append(input.substr(run_start, position - run_start));
158 output.append(replacement);
159 position += consumed;
160 run_start = position;
161 }
162 }
163
164 output.append(input.substr(run_start));
165}
166
167} // namespace sourcemeta::core
168
169#endif
SOURCEMETA_CORE_HTML_EXPORT auto html_escape(std::string &text) -> void
auto html_escape_append(Output &output, const std::string_view input) -> void
Definition html_escape.h:57
SOURCEMETA_CORE_REGEX_EXPORT auto matches(const Regex &regex, const std::string_view value) -> bool