Qt
Internal/Contributor docs for the Qt SDK. Note: These are NOT official API docs; those are found at https://doc.qt.io/
Loading...
Searching...
No Matches
qurlrecode.cpp
Go to the documentation of this file.
1// Copyright (C) 2016 Intel Corporation.
2// SPDX-License-Identifier: LicenseRef-Qt-Commercial OR LGPL-3.0-only OR GPL-2.0-only OR GPL-3.0-only
3// Qt-Security score:critical reason:data-parser
4
5#include "qurl.h"
6#include "private/qurl_p.h"
7#include "private/qstringconverter_p.h"
8#include "private/qtools_p.h"
9#include "private/qsimd_p.h"
10
12
13// ### move to qurl_p.h
19
20// From RFC 3896, Appendix A Collected ABNF for URI
21// unreserved = ALPHA / DIGIT / "-" / "." / "_" / "~"
22// reserved = gen-delims / sub-delims
23// gen-delims = ":" / "/" / "?" / "#" / "[" / "]" / "@"
24// sub-delims = "!" / "$" / "&" / "'" / "(" / ")"
25// / "*" / "+" / "," / ";" / "="
26static const uchar defaultActionTable[96] = {
27 0, // space
28 1, // '!' (sub-delim)
29 2, // '"'
30 1, // '#' (gen-delim)
31 1, // '$' (gen-delim)
32 2, // '%' (percent)
33 1, // '&' (gen-delim)
34 1, // "'" (sub-delim)
35 1, // '(' (sub-delim)
36 1, // ')' (sub-delim)
37 1, // '*' (sub-delim)
38 1, // '+' (sub-delim)
39 1, // ',' (sub-delim)
40 0, // '-' (unreserved)
41 0, // '.' (unreserved)
42 1, // '/' (gen-delim)
43
44 0, 0, 0, 0, 0, // '0' to '4' (unreserved)
45 0, 0, 0, 0, 0, // '5' to '9' (unreserved)
46 1, // ':' (gen-delim)
47 1, // ';' (sub-delim)
48 2, // '<'
49 1, // '=' (sub-delim)
50 2, // '>'
51 1, // '?' (gen-delim)
52
53 1, // '@' (gen-delim)
54 0, 0, 0, 0, 0, // 'A' to 'E' (unreserved)
55 0, 0, 0, 0, 0, // 'F' to 'J' (unreserved)
56 0, 0, 0, 0, 0, // 'K' to 'O' (unreserved)
57 0, 0, 0, 0, 0, // 'P' to 'T' (unreserved)
58 0, 0, 0, 0, 0, 0, // 'U' to 'Z' (unreserved)
59 1, // '[' (gen-delim)
60 2, // '\'
61 1, // ']' (gen-delim)
62 2, // '^'
63 0, // '_' (unreserved)
64
65 2, // '`'
66 0, 0, 0, 0, 0, // 'a' to 'e' (unreserved)
67 0, 0, 0, 0, 0, // 'f' to 'j' (unreserved)
68 0, 0, 0, 0, 0, // 'k' to 'o' (unreserved)
69 0, 0, 0, 0, 0, // 'p' to 't' (unreserved)
70 0, 0, 0, 0, 0, 0, // 'u' to 'z' (unreserved)
71 2, // '{'
72 2, // '|'
73 2, // '}'
74 0, // '~' (unreserved)
75
76 2 // BSKP
77};
78
79// mask tables, in negative polarity
80// 0x00 if it belongs to this category
81// 0xff if it doesn't
82
83static const uchar reservedMask[96] = {
84 0xff, // space
85 0xff, // '!' (sub-delim)
86 0x00, // '"'
87 0xff, // '#' (gen-delim)
88 0xff, // '$' (gen-delim)
89 0xff, // '%' (percent)
90 0xff, // '&' (gen-delim)
91 0xff, // "'" (sub-delim)
92 0xff, // '(' (sub-delim)
93 0xff, // ')' (sub-delim)
94 0xff, // '*' (sub-delim)
95 0xff, // '+' (sub-delim)
96 0xff, // ',' (sub-delim)
97 0xff, // '-' (unreserved)
98 0xff, // '.' (unreserved)
99 0xff, // '/' (gen-delim)
100
101 0xff, 0xff, 0xff, 0xff, 0xff, // '0' to '4' (unreserved)
102 0xff, 0xff, 0xff, 0xff, 0xff, // '5' to '9' (unreserved)
103 0xff, // ':' (gen-delim)
104 0xff, // ';' (sub-delim)
105 0x00, // '<'
106 0xff, // '=' (sub-delim)
107 0x00, // '>'
108 0xff, // '?' (gen-delim)
109
110 0xff, // '@' (gen-delim)
111 0xff, 0xff, 0xff, 0xff, 0xff, // 'A' to 'E' (unreserved)
112 0xff, 0xff, 0xff, 0xff, 0xff, // 'F' to 'J' (unreserved)
113 0xff, 0xff, 0xff, 0xff, 0xff, // 'K' to 'O' (unreserved)
114 0xff, 0xff, 0xff, 0xff, 0xff, // 'P' to 'T' (unreserved)
115 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, // 'U' to 'Z' (unreserved)
116 0xff, // '[' (gen-delim)
117 0x00, // '\'
118 0xff, // ']' (gen-delim)
119 0x00, // '^'
120 0xff, // '_' (unreserved)
121
122 0x00, // '`'
123 0xff, 0xff, 0xff, 0xff, 0xff, // 'a' to 'e' (unreserved)
124 0xff, 0xff, 0xff, 0xff, 0xff, // 'f' to 'j' (unreserved)
125 0xff, 0xff, 0xff, 0xff, 0xff, // 'k' to 'o' (unreserved)
126 0xff, 0xff, 0xff, 0xff, 0xff, // 'p' to 't' (unreserved)
127 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, // 'u' to 'z' (unreserved)
128 0x00, // '{'
129 0x00, // '|'
130 0x00, // '}'
131 0xff, // '~' (unreserved)
132
133 0xff // BSKP
134};
135
136static inline bool isHex(char16_t c)
137{
138 return (c >= u'a' && c <= u'f') || (c >= u'A' && c <= u'F') || (c >= u'0' && c <= u'9');
139}
140
141static inline bool isUpperHex(char16_t c)
142{
143 // undefined behaviour if c isn't an hex char!
144 return c < 0x60;
145}
146
147static inline char16_t toUpperHex(char16_t c)
148{
149 return isUpperHex(c) ? c : c - 0x20;
150}
151
152static inline ushort decodeNibble(char16_t c)
153{
154 return c >= u'a' ? c - u'a' + 0xA : c >= u'A' ? c - u'A' + 0xA : c - u'0';
155}
156
157// if the sequence at input is 2*HEXDIG, returns its decoding
158// returns -1 if it isn't.
159// assumes that the range has been checked already
160static inline char16_t decodePercentEncoding(const char16_t *input)
161{
162 char16_t c1 = input[1];
163 char16_t c2 = input[2];
164 if (!isHex(c1) || !isHex(c2))
165 return char16_t(-1);
166 return decodeNibble(c1) << 4 | decodeNibble(c2);
167}
168
169static inline char16_t encodeNibble(ushort c)
170{
171 return QtMiscUtils::toHexUpper(c);
172}
173
174static void ensureDetached(QString &result, char16_t *&output, const char16_t *begin, const char16_t *input, const char16_t *end,
175 int add = 0)
176{
177 if (!output) {
178 // now detach
179 // create enough space if the rest of the string needed to be percent-encoded
180 int charsProcessed = input - begin;
181 int charsRemaining = end - input;
182 int spaceNeeded = end - begin + 2 * charsRemaining + add;
183 int origSize = result.size();
184 result.resize(origSize + spaceNeeded);
185
186 // we know that resize() above detached, so we bypass the reference count check
187 output = const_cast<char16_t *>(reinterpret_cast<const char16_t *>(result.constData()))
188 + origSize;
189
190 // copy the chars we've already processed
191 int i;
192 for (i = 0; i < charsProcessed; ++i)
193 output[i] = begin[i];
194 output += i;
195 }
196}
197
198namespace {
199struct QUrlUtf8Traits : public QUtf8BaseTraitsNoAscii
200{
201 // From RFC 3987:
202 // iunreserved = ALPHA / DIGIT / "-" / "." / "_" / "~" / ucschar
203 //
204 // ucschar = %xA0-D7FF / %xF900-FDCF / %xFDF0-FFEF
205 // / %x10000-1FFFD / %x20000-2FFFD / %x30000-3FFFD
206 // / %x40000-4FFFD / %x50000-5FFFD / %x60000-6FFFD
207 // / %x70000-7FFFD / %x80000-8FFFD / %x90000-9FFFD
208 // / %xA0000-AFFFD / %xB0000-BFFFD / %xC0000-CFFFD
209 // / %xD0000-DFFFD / %xE1000-EFFFD
210 //
211 // iprivate = %xE000-F8FF / %xF0000-FFFFD / %x100000-10FFFD
212 //
213 // That RFC allows iprivate only as part of iquery, but we don't know here
214 // whether we're looking at a query or another part of an URI, so we accept
215 // them too. The definition above excludes U+FFF0 to U+FFFD from appearing
216 // unencoded, but we see no reason for its exclusion, so we allow them to
217 // be decoded (and we need U+FFFD the replacement character to indicate
218 // failure to decode).
219 //
220 // That means we must disallow:
221 // * unpaired surrogates (QUtf8Functions takes care of that for us)
222 // * non-characters
223 static const bool allowNonCharacters = false;
224
225 // override: our "bytes" are three percent-encoded UTF-16 characters
226 static void appendByte(char16_t *&ptr, uchar b)
227 {
228 // b >= 0x80, by construction, so percent-encode
229 *ptr++ = '%';
230 *ptr++ = encodeNibble(b >> 4);
231 *ptr++ = encodeNibble(b & 0xf);
232 }
233
234 static uchar peekByte(const char16_t *ptr, qsizetype n = 0)
235 {
236 // decodePercentEncoding returns char16_t(-1) if it can't decode,
237 // which means we return 0xff, which is not a valid continuation byte.
238 // If ptr[i * 3] is not '%', we'll multiply by zero and return 0,
239 // also not a valid continuation byte (if it's '%', we multiply by 1).
240 return uchar(decodePercentEncoding(ptr + n * 3))
241 * uchar(ptr[n * 3] == '%');
242 }
243
244 static qptrdiff availableBytes(const char16_t *ptr, const char16_t *end)
245 {
246 return (end - ptr) / 3;
247 }
248
249 static void advanceByte(const char16_t *&ptr, int n = 1)
250 {
251 ptr += n * 3;
252 }
253};
254}
255
256// returns true if we performed an UTF-8 decoding
257static bool encodedUtf8ToUtf16(QString &result, char16_t *&output, const char16_t *begin,
258 const char16_t *&input, const char16_t *end, char16_t decoded)
259{
260 char32_t buffer[1];
261 char32_t &ucs4 = buffer[0];
262 char32_t *dst = buffer;
263 const char16_t *src = input + 3;// skip the %XX that yielded \a decoded
264 int charsNeeded = QUtf8Functions::fromUtf8<QUrlUtf8Traits>(decoded, dst, src, end);
265 if (charsNeeded < 0)
266 return false;
267
268 if (!QChar::requiresSurrogates(ucs4)) {
269 // UTF-8 decoded and no surrogates are required
270 // detach if necessary
271 // possibilities are: 6 chars (%XX%XX) -> one char; 9 chars (%XX%XX%XX) -> one char
272 ensureDetached(result, output, begin, input, end, -3 * charsNeeded + 1);
273 *output++ = ucs4;
274 } else {
275 // UTF-8 decoded to something that requires a surrogate pair
276 // compressing from %XX%XX%XX%XX (12 chars) to two
277 ensureDetached(result, output, begin, input, end, -10);
278 *output++ = QChar::highSurrogate(ucs4);
279 *output++ = QChar::lowSurrogate(ucs4);
280 }
281
282 input = src - 1;
283 return true;
284}
285
286static void unicodeToEncodedUtf8(QString &result, char16_t *&output, const char16_t *begin,
287 const char16_t *&input, const char16_t *end, char16_t decoded)
288{
289 // calculate the utf8 length and ensure enough space is available
290 int utf8len = QChar::isHighSurrogate(decoded) ? 4 : decoded >= 0x800 ? 3 : 2;
291
292 // detach
293 if (!output) {
294 // we need 3 * utf8len for the encoded UTF-8 sequence
295 // but ensureDetached already adds 3 for the char we're processing
296 ensureDetached(result, output, begin, input, end, 3*utf8len - 3);
297 } else {
298 // verify that there's enough space or expand
299 int charsRemaining = end - input - 1; // not including this one
300 int pos = output - reinterpret_cast<const char16_t *>(result.constData());
301 int spaceRemaining = result.size() - pos;
302 if (spaceRemaining < 3*charsRemaining + 3*utf8len) {
303 // must resize
304 result.resize(result.size() + 3*utf8len);
305
306 // we know that resize() above detached, so we bypass the reference count check
307 output = const_cast<char16_t *>(reinterpret_cast<const char16_t *>(result.constData()));
308 output += pos;
309 }
310 }
311
312 ++input;
313 int res = QUtf8Functions::toUtf8<QUrlUtf8Traits>(decoded, output, input, end);
314 --input;
315 if (res < 0) {
316 // bad surrogate pair sequence
317 // we will encode bad UTF-16 to UTF-8
318 // but they don't get decoded back
319
320 // first of three bytes
321 uchar c = 0xe0 | uchar(decoded >> 12);
322 *output++ = '%';
323 *output++ = 'E';
324 *output++ = encodeNibble(c & 0xf);
325
326 // second byte
327 c = 0x80 | (uchar(decoded >> 6) & 0x3f);
328 *output++ = '%';
329 *output++ = encodeNibble(c >> 4);
330 *output++ = encodeNibble(c & 0xf);
331
332 // third byte
333 c = 0x80 | (decoded & 0x3f);
334 *output++ = '%';
335 *output++ = encodeNibble(c >> 4);
336 *output++ = encodeNibble(c & 0xf);
337 }
338}
339
340static int recode(QString &result, const char16_t *begin, const char16_t *end,
341 QUrl::ComponentFormattingOptions encoding, const uchar *actionTable,
342 bool retryBadEncoding)
343{
344 const int origSize = result.size();
345 const char16_t *input = begin;
346 char16_t *output = nullptr;
347
349 for ( ; input != end; ++input) {
350 char16_t c;
351 // try a run where no change is necessary
352 for ( ; input != end; ++input) {
353 c = *input;
354 if (c < 0x20U)
355 action = EncodeCharacter;
356 if (c < 0x20U || c >= 0x80U) // also: (c - 0x20 < 0x60U)
357 goto non_trivial;
358 action = EncodingAction(actionTable[c - ' ']);
359 if (action == EncodeCharacter)
360 goto non_trivial;
361 if (output)
362 *output++ = c;
363 }
364 break;
365
366non_trivial:
367 char16_t decoded;
368 if (c == '%' && retryBadEncoding) {
369 // always write "%25"
370 ensureDetached(result, output, begin, input, end);
371 *output++ = '%';
372 *output++ = '2';
373 *output++ = '5';
374 continue;
375 } else if (c == '%') {
376 // check if the input is valid
377 if (input + 2 >= end || (decoded = decodePercentEncoding(input)) == char16_t(-1)) {
378 // not valid, retry
379 result.resize(origSize);
380 return recode(result, begin, end, encoding, actionTable, true);
381 }
382
383 if (decoded >= 0x80) {
384 // decode the UTF-8 sequence
385 if (!(encoding & QUrl::EncodeUnicode) &&
386 encodedUtf8ToUtf16(result, output, begin, input, end, decoded))
387 continue;
388
389 // decoding the encoded UTF-8 failed
390 action = LeaveCharacter;
391 } else if (decoded >= 0x20) {
392 action = EncodingAction(actionTable[decoded - ' ']);
393 }
394 } else {
395 decoded = c;
396 if (decoded >= 0x80 && encoding & QUrl::EncodeUnicode) {
397 // encode the UTF-8 sequence
398 unicodeToEncodedUtf8(result, output, begin, input, end, decoded);
399 continue;
400 } else if (decoded >= 0x80) {
401 if (output)
402 *output++ = c;
403 continue;
404 }
405 }
406
407 // there are six possibilities:
408 // current \ action | DecodeCharacter | LeaveCharacter | EncodeCharacter
409 // decoded | 1:leave | 2:leave | 3:encode
410 // encoded | 4:decode | 5:leave | 6:leave
411 // cases 1 and 2 were handled before this section
412
413 if (c == '%' && action != DecodeCharacter) {
414 // cases 5 and 6: it's encoded and we're leaving it as it is
415 // except we're pedantic and we'll uppercase the hex
416 if (output || !isUpperHex(input[1]) || !isUpperHex(input[2])) {
417 ensureDetached(result, output, begin, input, end);
418 *output++ = '%';
419 *output++ = toUpperHex(*++input);
420 *output++ = toUpperHex(*++input);
421 }
422 } else if (c == '%' && action == DecodeCharacter) {
423 // case 4: we need to decode
424 ensureDetached(result, output, begin, input, end);
425 *output++ = decoded;
426 input += 2;
427 } else {
428 // must be case 3: we need to encode
429 ensureDetached(result, output, begin, input, end);
430 *output++ = '%';
431 *output++ = encodeNibble(c >> 4);
432 *output++ = encodeNibble(c & 0xf);
433 }
434 }
435
436 if (output) {
437 int len = output - reinterpret_cast<const char16_t *>(result.constData());
438 result.truncate(len);
439 return len - origSize;
440 }
441 return 0;
442}
443
444/*
445 * Returns true if the input it checked (if it checked anything) is not
446 * encoded. A return of false indicates there's a percent at \a input that
447 * needs to be decoded.
448 */
449#ifdef __SSE2__
450static bool simdCheckNonEncoded(QChar *&output, const char16_t *&input, const char16_t *end)
451{
452# ifdef __AVX2__
453 const __m256i percents256 = _mm256_broadcastw_epi16(_mm_cvtsi32_si128('%'));
454 const __m128i percents = _mm256_castsi256_si128(percents256);
455# else
456 const __m128i percents = _mm_set1_epi16('%');
457# endif
458
459 uint idx = 0;
460 quint32 mask = 0;
461 if (input + 16 <= end) {
462 qptrdiff offset = 0;
463 for ( ; input + offset + 16 <= end; offset += 16) {
464# ifdef __AVX2__
465 // do 32 bytes at a time using AVX2
466 __m256i data = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(input + offset));
467 __m256i comparison = _mm256_cmpeq_epi16(data, percents256);
468 mask = _mm256_movemask_epi8(comparison);
469 _mm256_storeu_si256(reinterpret_cast<__m256i *>(output + offset), data);
470# else
471 // do 32 bytes at a time using unrolled SSE2
472 __m128i data1 = _mm_loadu_si128(reinterpret_cast<const __m128i *>(input + offset));
473 __m128i data2 = _mm_loadu_si128(reinterpret_cast<const __m128i *>(input + offset + 8));
474 __m128i comparison1 = _mm_cmpeq_epi16(data1, percents);
475 __m128i comparison2 = _mm_cmpeq_epi16(data2, percents);
476 uint mask1 = _mm_movemask_epi8(comparison1);
477 uint mask2 = _mm_movemask_epi8(comparison2);
478
479 _mm_storeu_si128(reinterpret_cast<__m128i *>(output + offset), data1);
480 if (!mask1)
481 _mm_storeu_si128(reinterpret_cast<__m128i *>(output + offset + 8), data2);
482 mask = mask1 | (mask2 << 16);
483# endif
484
485 if (mask) {
486 idx = qCountTrailingZeroBits(mask) / 2;
487 break;
488 }
489 }
490
491 input += offset;
492 if (output)
493 output += offset;
494 } else if (input + 8 <= end) {
495 // do 16 bytes at a time
496 __m128i data = _mm_loadu_si128(reinterpret_cast<const __m128i *>(input));
497 __m128i comparison = _mm_cmpeq_epi16(data, percents);
498 mask = _mm_movemask_epi8(comparison);
499 _mm_storeu_si128(reinterpret_cast<__m128i *>(output), data);
500 idx = qCountTrailingZeroBits(quint16(mask)) / 2;
501 } else if (input + 4 <= end) {
502 // do 8 bytes only
503 __m128i data = _mm_loadl_epi64(reinterpret_cast<const __m128i *>(input));
504 __m128i comparison = _mm_cmpeq_epi16(data, percents);
505 mask = _mm_movemask_epi8(comparison) & 0xffu;
506 _mm_storel_epi64(reinterpret_cast<__m128i *>(output), data);
507 idx = qCountTrailingZeroBits(quint8(mask)) / 2;
508 } else {
509 // no percents found (because we didn't check)
510 return true;
511 }
512
513 // advance to the next non-encoded
514 input += idx;
515 output += idx;
516
517 return !mask;
518}
519#else
520static bool simdCheckNonEncoded(...)
521{
522 return true;
523}
524#endif
525
526// Returns whether the percent-decoded octet \a byte (0x00 to 0xFF) is unsafe
527// to appear in a local-file path. Valid UTF-8 has already been folded to
528// UTF-16 by the time we get here, so any byte >= 0x80 comes from an invalid
529// sequence.
530static bool isUnsafeForLocalFile(uchar byte)
531{
532 if (byte == 0 || byte >= 0x80 || byte == '/')
533 return true;
534#ifdef Q_OS_WIN
535 if (byte == '\\')
536 return true;
537#endif
538 return false;
539}
540
541/*!
542 \since 5.0
543 \internal
544
545 This function decodes a percent-encoded string located in \a in
546 by appending each character to \a appendTo. It returns the number of
547 characters appended. When the \a encoding mode is QUrl::FullyDecoded (the
548 mode used for most functions), each percent-encoded sequence is decoded as
549 follows:
550
551 \list
552 \li from %00 to %7F: the exact decoded value is appended;
553 \li from %80 to %FF: QChar::ReplacementCharacter is appended;
554 \li bad encoding: original input is copied to the output, undecoded.
555 \endlist
556
557 Given the above, it's important for the input to already have all UTF-8
558 percent sequences decoded by qt_urlRecode (that is, the input should not
559 have been processed with QUrl::EncodeUnicode).
560
561 The input should also be a valid percent-encoded sequence (the output of
562 qt_urlRecode is always valid).
563
564 With QUrlDecodeForLocalFile in \a encoding, the function instead returns -1
565 when it would decode a byte that is unsafe in a local-file path. The \a
566 appendTo string is truncated back to its original size, but always returns
567 null if the original was empty.
568
569 If no decoding error happened, this function returns the number of
570 characters added to \a appendTo. That might be 0 if there was nothing to
571 decode (see qt_urlRecode() below for the reason).
572*/
573static qsizetype decode(QString &appendTo, QStringView in,
574 QUrl::ComponentFormattingOptions encoding)
575{
576 const bool forLocalFile = encoding.testFlags(QUrlDecodeForLocalFile);
577 const char16_t *begin = in.utf16();
578 const char16_t *end = begin + in.size();
579
580#ifdef Q_OS_WIN
581 // precheck for backslashes on Windows (they are stored in decoded form)
582 if (forLocalFile && in.contains(u'\\')) {
583 if (appendTo.isEmpty())
584 appendTo = QString(); // make null
585 return -1;
586 }
587#endif
588
589 // fast check whether there's anything to be decoded in the first place
590 const char16_t *input = QtPrivate::qustrchr(in, '%');
591
592 if (Q_LIKELY(input == end))
593 return 0; // nothing to do, it was already decoded!
594
595 // detach
596 const int origSize = appendTo.size();
597 appendTo.resize(origSize + (end - begin));
598 QChar *output = appendTo.data() + origSize;
599 memcpy(static_cast<void *>(output), static_cast<const void *>(begin), (input - begin) * sizeof(QChar));
600 output += input - begin;
601
602 while (input != end) {
603 // something was encoded
604 Q_ASSERT(*input == '%');
605
606 if (Q_UNLIKELY(end - input < 3 || !isHex(input[1]) || !isHex(input[2]))) {
607 // badly-encoded data
608 appendTo.resize(origSize + (end - begin));
609 memcpy(static_cast<void *>(appendTo.begin() + origSize),
610 static_cast<const void *>(begin), (end - begin) * sizeof(*end));
611 return end - begin;
612 }
613
614 ++input;
615 uchar byte = decodeNibble(input[0]) << 4 | decodeNibble(input[1]);
616 *output++ = QChar::fromUcs2(byte);
617 if (forLocalFile && Q_UNLIKELY(isUnsafeForLocalFile(byte))) {
618 if (origSize == 0)
619 appendTo = QString(); // make null
620 else
621 appendTo.truncate(origSize);
622 return -1;
623 }
624 if (byte >= 0x80)
625 output[-1] = QChar::ReplacementCharacter;
626 input += 2;
627
628 // search for the next percent, copying from input to output
629 if (simdCheckNonEncoded(output, input, end)) {
630 while (input != end) {
631 const char16_t uc = *input;
632 if (uc == '%')
633 break;
634 *output++ = uc;
635 ++input;
636 }
637 }
638 }
639
640 const qsizetype len = output - appendTo.begin();
641 appendTo.truncate(len);
642 return len - origSize;
643}
644
645template <size_t N>
646static void maskTable(uchar (&table)[N], const uchar (&mask)[N])
647{
648 for (size_t i = 0; i < N; ++i)
649 table[i] &= mask[i];
650}
651
652/*!
653 \internal
654
655 Recodes the string from \a begin to \a end. If any transformations are
656 done, append them to \a appendTo and return the number of characters added.
657 If no transformations were required, return 0.
658
659 The \a encoding option modifies the default behaviour:
660 \list
661 \li QUrl::DecodeReserved: if set, reserved characters will be decoded;
662 if unset, reserved characters will be encoded
663 \li QUrl::EncodeSpaces: if set, spaces will be encoded to "%20"; if unset, they will be " "
664 \li QUrl::EncodeUnicode: if set, characters above U+0080 will be encoded to their UTF-8
665 percent-encoded form; if unset, they will be decoded to UTF-16
666 \li QUrl::FullyDecoded: if set, this function will decode all percent-encoded sequences,
667 including that of the percent character. The resulting string
668 will not be percent-encoded anymore. Use with caution!
669 In this mode, the behaviour is undefined if the input string
670 contains any percent-encoding sequences above %80.
671 Also, the function will not correct bad % sequences.
672 \li QUrlDecodeForLocalFile: if set, this performs the FullyDecoded mode with error checking
673 suitable for local file paths. See decode() above for details.
674 \endlist
675
676 Other flags are ignored (including QUrl::EncodeReserved).
677
678 The \a tableModifications argument can be used to supply extra
679 modifications to the tables, to be applied after the flags above are
680 handled. It consists of a sequence of 16-bit values, where the low 8 bits
681 indicate the character in question and the high 8 bits are either \c
682 EncodeCharacter, \c LeaveCharacter or \c DecodeCharacter.
683
684 This function corrects percent-encoded errors by interpreting every '%' as
685 meaning "%25" (all percents in the same content), except in Decoded modes.
686
687 On success, this function returns the number of characters added to \a
688 appendTo, or -1 on error (errors are currently only detected if \a encoding
689 is QUrlDecodeForLocalFile). A return value of 0 means this function
690 had no percent-encoded transformations to perform, so the caller must
691 append the \a in input to \a appendTo. This lets the caller return a
692 shallow copy of the original QString (avoiding allocation) and keep control
693 of its null-vs-empty state.
694 */
695
696Q_AUTOTEST_EXPORT qsizetype
697qt_urlRecode(QString &appendTo, QStringView in,
698 QUrl::ComponentFormattingOptions encoding, const ushort *tableModifications)
699{
700 uchar actionTable[sizeof defaultActionTable];
701 if ((encoding & QUrl::FullyDecoded) == QUrl::FullyDecoded) {
702 return decode(appendTo, in, encoding);
703 }
704
705 memcpy(actionTable, defaultActionTable, sizeof actionTable);
706 if (encoding & QUrl::DecodeReserved)
707 maskTable(actionTable, reservedMask);
708 if (encoding & QUrl::EncodeSpaces)
709 actionTable[0] = EncodeCharacter;
710
711 if (tableModifications) {
712 for (const ushort *p = tableModifications; *p; ++p)
713 actionTable[uchar(*p) - ' '] = *p >> 8;
714 }
715
716 return recode(appendTo, reinterpret_cast<const char16_t *>(in.begin()),
717 reinterpret_cast<const char16_t *>(in.end()), encoding, actionTable, false);
718}
719
720qsizetype qt_encodeFromUser(QString &appendTo, const QString &in, const ushort *tableModifications)
721{
722 uchar actionTable[sizeof defaultActionTable];
723 memcpy(actionTable, defaultActionTable, sizeof actionTable);
724
725 // Different defaults to the regular encoded-to-encoded recoding
726 actionTable['[' - ' '] = EncodeCharacter;
727 actionTable[']' - ' '] = EncodeCharacter;
728
729 if (tableModifications) {
730 for (const ushort *p = tableModifications; *p; ++p)
731 actionTable[uchar(*p) - ' '] = *p >> 8;
732 }
733
734 return recode(appendTo, reinterpret_cast<const char16_t *>(in.begin()),
735 reinterpret_cast<const char16_t *>(in.end()), {}, actionTable, true);
736}
737
738QT_END_NAMESPACE
Combined button and popup list for selecting options.
qsizetype qt_encodeFromUser(QString &appendTo, const QString &input, const ushort *tableModifications)
static char16_t decodePercentEncoding(const char16_t *input)
static bool encodedUtf8ToUtf16(QString &result, char16_t *&output, const char16_t *begin, const char16_t *&input, const char16_t *end, char16_t decoded)
static qsizetype decode(QString &appendTo, QStringView in, QUrl::ComponentFormattingOptions encoding)
static int recode(QString &result, const char16_t *begin, const char16_t *end, QUrl::ComponentFormattingOptions encoding, const uchar *actionTable, bool retryBadEncoding)
static ushort decodeNibble(char16_t c)
static bool isUnsafeForLocalFile(uchar byte)
static char16_t encodeNibble(ushort c)
static const uchar reservedMask[96]
EncodingAction
@ DecodeCharacter
@ EncodeCharacter
@ LeaveCharacter
static char16_t toUpperHex(char16_t c)
static bool isUpperHex(char16_t c)
static void maskTable(uchar(&table)[N], const uchar(&mask)[N])
static void unicodeToEncodedUtf8(QString &result, char16_t *&output, const char16_t *begin, const char16_t *&input, const char16_t *end, char16_t decoded)
static const uchar defaultActionTable[96]
static void ensureDetached(QString &result, char16_t *&output, const char16_t *begin, const char16_t *input, const char16_t *end, int add=0)
static bool isHex(char16_t c)